From 317c693978670789ea163bcea030f551805cc80b Mon Sep 17 00:00:00 2001 From: Dan Carpenter Date: Wed, 23 Apr 2025 20:22:05 +0300 Subject: [PATCH 001/857] rpmsg: qcom_smd: Fix uninitialized return variable in __qcom_smd_send() The "ret" variable isn't initialized if we don't enter the loop. For example, if "channel->state" is not SMD_CHANNEL_OPENED. Fixes: 33e3820dda88 ("rpmsg: smd: Use spinlock in tx path") Signed-off-by: Dan Carpenter Link: https://lore.kernel.org/r/aAkhvV0nSbrsef1P@stanley.mountain Signed-off-by: Bjorn Andersson --- drivers/rpmsg/qcom_smd.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/rpmsg/qcom_smd.c b/drivers/rpmsg/qcom_smd.c index 40d386809d6b78..bb161def317533 100644 --- a/drivers/rpmsg/qcom_smd.c +++ b/drivers/rpmsg/qcom_smd.c @@ -746,7 +746,7 @@ static int __qcom_smd_send(struct qcom_smd_channel *channel, const void *data, __le32 hdr[5] = { cpu_to_le32(len), }; int tlen = sizeof(hdr) + len; unsigned long flags; - int ret; + int ret = 0; /* Word aligned channels only accept word size aligned data */ if (channel->info_word && len % 4) From c922423ce66bcee6c61d1b5aadea09b29ca81400 Mon Sep 17 00:00:00 2001 From: Daniel Lezcano Date: Mon, 16 Mar 2026 18:14:13 +0100 Subject: [PATCH 002/857] slimbus: qcom-ngd-ctrl: Use the unified QMI service ID instead of defining it locally Instead of defining a local macro with a custom name for the QMI service identifier, use the one provided in qmi.h and remove the locally defined macro. Reviewed-by: Dmitry Baryshkov Signed-off-by: Daniel Lezcano Signed-off-by: Srinivas Kandagatla --- drivers/slimbus/qcom-ngd-ctrl.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/drivers/slimbus/qcom-ngd-ctrl.c b/drivers/slimbus/qcom-ngd-ctrl.c index 3071e46d03beaa..80877e951849f3 100644 --- a/drivers/slimbus/qcom-ngd-ctrl.c +++ b/drivers/slimbus/qcom-ngd-ctrl.c @@ -48,7 +48,6 @@ NGD_INT_RX_MSG_RCVD) /* Slimbus QMI service */ -#define SLIMBUS_QMI_SVC_ID 0x0301 #define SLIMBUS_QMI_SVC_V1 1 #define SLIMBUS_QMI_INS_ID 0 #define SLIMBUS_QMI_SELECT_INSTANCE_REQ_V01 0x0020 @@ -1408,8 +1407,8 @@ static int qcom_slim_ngd_qmi_svc_event_init(struct qcom_slim_ngd_ctrl *ctrl) return ret; } - ret = qmi_add_lookup(&qmi->svc_event_hdl, SLIMBUS_QMI_SVC_ID, - SLIMBUS_QMI_SVC_V1, SLIMBUS_QMI_INS_ID); + ret = qmi_add_lookup(&qmi->svc_event_hdl, QMI_SERVICE_ID_SLIMBUS, + SLIMBUS_QMI_SVC_V1, SLIMBUS_QMI_INS_ID); if (ret < 0) { dev_err(ctrl->dev, "qmi_add_lookup failed: %d\n", ret); qmi_handle_release(&qmi->svc_event_hdl); From e58af70a1ee4cddc36832a034ac76ea4c3c4418c Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Fri, 10 Jul 2026 16:45:03 +0200 Subject: [PATCH 003/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 29 ++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) create mode 100644 arch/arm/arm-soc-for-next-contents.txt diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt new file mode 100644 index 00000000000000..30c2012645df57 --- /dev/null +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -0,0 +1,29 @@ +soc/arm + +soc/dt + patch + ARM: dts: st: spear: Correct indentation + ARM: dts: st: ste: Correct indentation + +soc/drivers + +soc/defconfig + +soc/late + +arm/fixes + patch + MAINTAINERS: Update SpacemiT SoC git tree repository + (71827776667f4e4677a4fa806bcfb24d4b8dd9d7) + git://git.kernel.org/pub/scm/linux/kernel/git/pza/linux tags/reset-fixes-for-v7.2 + (813e034925814858cc52e7de321ec4848314e15d) + git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.2-pmc-fixes + (265d5d4032c5f6eb089a6e6241d37fdbde7da180) + git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.2-soc-fixes + (806a66f926c2b6652aeb88983d01f25081b41a73) + git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.2-arm64-dt-fixes + patch + ARM: Don't let ARMv5 platforms select USE_OF + (314c243b201b678fa89226b1eaea51a71340454e) + git://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee tags/tee-update-for-v7.2 + From 624bd3603c80f82c27a1b65357d2cd66bf1d385a Mon Sep 17 00:00:00 2001 From: Dinh Nguyen Date: Wed, 17 Jun 2026 11:43:03 -0500 Subject: [PATCH 004/857] EDAC/altera: Use parent device for devres in altr_portb_setup() Anchor the devres group and the devm-managed IRQ requests in altr_portb_setup() to the actual parent device (device->edac->dev) instead of the embedded struct device inside the copied per-port altr_edac_device_dev. This keeps devres_open_group(), devm_request_irq(), devres_remove_group() and devres_release_group() all referring to the same long-lived device so the group and the resources allocated inside it are torn down together. Fixes: 911049845d70 ("EDAC, altera: Add Arria10 SD-MMC EDAC support") Closes: https://sashiko.dev/#/patchset/20260503212558.2811480-1-dbgh9129%40gmail.com Assisted-by: Claude:claude-opus-4-7 Signed-off-by: Dinh Nguyen Signed-off-by: Borislav Petkov (AMD) Cc: stable@vger.kernel.org Link: https://patch.msgid.link/20260617164303.585555-1-dinguyen@kernel.org --- drivers/edac/altera_edac.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/drivers/edac/altera_edac.c b/drivers/edac/altera_edac.c index 4edd2088c2db6f..5914b2fd94d9c5 100644 --- a/drivers/edac/altera_edac.c +++ b/drivers/edac/altera_edac.c @@ -1533,7 +1533,7 @@ static int altr_portb_setup(struct altr_edac_device_dev *device) altdev = dci->pvt_info; *altdev = *device; - if (!devres_open_group(&altdev->ddev, altr_portb_setup, GFP_KERNEL)) + if (!devres_open_group(device->edac->dev, altr_portb_setup, GFP_KERNEL)) return -ENOMEM; /* Update PortB specific values */ @@ -1562,7 +1562,7 @@ static int altr_portb_setup(struct altr_edac_device_dev *device) rc = -ENODEV; goto err_release_group_1; } - rc = devm_request_irq(&altdev->ddev, altdev->sb_irq, + rc = devm_request_irq(device->edac->dev, altdev->sb_irq, prv->ecc_irq_handler, IRQF_TRIGGER_HIGH, ecc_name, altdev); if (rc) { @@ -1585,7 +1585,7 @@ static int altr_portb_setup(struct altr_edac_device_dev *device) rc = -ENODEV; goto err_release_group_1; } - rc = devm_request_irq(&altdev->ddev, altdev->db_irq, + rc = devm_request_irq(device->edac->dev, altdev->db_irq, prv->ecc_irq_handler, IRQF_TRIGGER_HIGH, ecc_name, altdev); if (rc) { @@ -1605,13 +1605,13 @@ static int altr_portb_setup(struct altr_edac_device_dev *device) list_add(&altdev->next, &altdev->edac->a10_ecc_devices); - devres_remove_group(&altdev->ddev, altr_portb_setup); + devres_remove_group(device->edac->dev, altr_portb_setup); return 0; err_release_group_1: edac_device_free_ctl_info(dci); - devres_release_group(&altdev->ddev, altr_portb_setup); + devres_release_group(device->edac->dev, altr_portb_setup); edac_printk(KERN_ERR, EDAC_DEVICE, "%s:Error setting up EDAC device: %d\n", ecc_name, rc); return rc; From bd3cbc42aa8cbde3fb565882bb1be7a20a2b527f Mon Sep 17 00:00:00 2001 From: AngeloGioacchino Del Regno Date: Wed, 1 Jul 2026 14:20:42 +0200 Subject: [PATCH 005/857] soc: mediatek: mtk-mmsys: Rework routes to specify component ID In preparation for a refactoring of multimedia related MediaTek drivers, including mmsys, mutex and mediatek-drm, rework all of the MMSYS routes to specify a hardware component instance number (or "SubID") alongside the hardware component type. This also is one step of preparation towards the removal of the catch-all mtk_ddp_comp_id enumeration and towards the migration from a predefined-coupling static hardware component IDSubID mapping (carrying around a very long enumeration and also some multiple big arrays in mediatek-drm) to a more flexible map of Component ID (Type) decoupled from Component SubID (HW Instance) as then, anyway, techniques to handle components are always the same on a type basis. Signed-off-by: AngeloGioacchino Del Regno --- drivers/soc/mediatek/mt8167-mmsys.h | 21 ++-- drivers/soc/mediatek/mt8173-mmsys.h | 28 ++--- drivers/soc/mediatek/mt8183-mmsys.h | 14 +-- drivers/soc/mediatek/mt8186-mmsys.h | 22 ++-- drivers/soc/mediatek/mt8188-mmsys.h | 78 ++++++------ drivers/soc/mediatek/mt8192-mmsys.h | 20 +-- drivers/soc/mediatek/mt8195-mmsys.h | 181 ++++++++++++++-------------- drivers/soc/mediatek/mt8365-mmsys.h | 20 +-- drivers/soc/mediatek/mtk-mmsys.h | 21 ++-- 9 files changed, 204 insertions(+), 201 deletions(-) diff --git a/drivers/soc/mediatek/mt8167-mmsys.h b/drivers/soc/mediatek/mt8167-mmsys.h index eef14083c47b5a..d579feee4212f0 100644 --- a/drivers/soc/mediatek/mt8167-mmsys.h +++ b/drivers/soc/mediatek/mt8167-mmsys.h @@ -10,29 +10,24 @@ #define MT8167_DISP_REG_CONFIG_DISP_RDMA0_SOUT_SEL_IN 0x06c #define MT8167_DITHER_MOUT_EN_RDMA0 0x1 -#define MT8167_DITHER_MOUT_EN_MASK 0x7 - #define MT8167_RDMA0_SOUT_DSI0 0x2 -#define MT8167_RDMA0_SOUT_MASK 0x3 - #define MT8167_DSI0_SEL_IN_RDMA0 0x1 -#define MT8167_DSI0_SEL_IN_MASK 0x3 static const struct mtk_mmsys_routes mt8167_mmsys_routing_table[] = { - MMSYS_ROUTE(OVL0, COLOR0, + MMSYS_ROUTE(OVL, 0, COLOR, 0, MT8167_DISP_REG_CONFIG_DISP_OVL0_MOUT_EN, OVL0_MOUT_EN_COLOR0, OVL0_MOUT_EN_COLOR0), - MMSYS_ROUTE(DITHER0, RDMA0, - MT8167_DISP_REG_CONFIG_DISP_DITHER_MOUT_EN, MT8167_DITHER_MOUT_EN_MASK, + MMSYS_ROUTE(DITHER, 0, RDMA, 0, + MT8167_DISP_REG_CONFIG_DISP_DITHER_MOUT_EN, MT8167_DITHER_MOUT_EN_RDMA0, MT8167_DITHER_MOUT_EN_RDMA0), - MMSYS_ROUTE(OVL0, COLOR0, + MMSYS_ROUTE(OVL, 0, COLOR, 0, MT8167_DISP_REG_CONFIG_DISP_COLOR0_SEL_IN, COLOR0_SEL_IN_OVL0, COLOR0_SEL_IN_OVL0), - MMSYS_ROUTE(RDMA0, DSI0, - MT8167_DISP_REG_CONFIG_DISP_DSI0_SEL_IN, MT8167_DSI0_SEL_IN_MASK, + MMSYS_ROUTE(RDMA, 0, DSI, 0, + MT8167_DISP_REG_CONFIG_DISP_DSI0_SEL_IN, MT8167_DSI0_SEL_IN_RDMA0, MT8167_DSI0_SEL_IN_RDMA0), - MMSYS_ROUTE(RDMA0, DSI0, - MT8167_DISP_REG_CONFIG_DISP_RDMA0_SOUT_SEL_IN, MT8167_RDMA0_SOUT_MASK, + MMSYS_ROUTE(RDMA, 0, DSI, 0, + MT8167_DISP_REG_CONFIG_DISP_RDMA0_SOUT_SEL_IN, MT8167_RDMA0_SOUT_DSI0, MT8167_RDMA0_SOUT_DSI0), }; diff --git a/drivers/soc/mediatek/mt8173-mmsys.h b/drivers/soc/mediatek/mt8173-mmsys.h index 957876d7c16661..af67879ff8b445 100644 --- a/drivers/soc/mediatek/mt8173-mmsys.h +++ b/drivers/soc/mediatek/mt8173-mmsys.h @@ -33,46 +33,46 @@ #define MT8173_RDMA0_SOUT_COLOR0 BIT(0) static const struct mtk_mmsys_routes mt8173_mmsys_routing_table[] = { - MMSYS_ROUTE(OVL0, COLOR0, + MMSYS_ROUTE(OVL, 0, COLOR, 0, MT8173_DISP_REG_CONFIG_DISP_OVL0_MOUT_EN, MT8173_OVL0_MOUT_EN_COLOR0, MT8173_OVL0_MOUT_EN_COLOR0), - MMSYS_ROUTE(OD0, RDMA0, + MMSYS_ROUTE(OD, 0, RDMA, 0, MT8173_DISP_REG_CONFIG_DISP_OD_MOUT_EN, MT8173_OD0_MOUT_EN_RDMA0, MT8173_OD0_MOUT_EN_RDMA0), - MMSYS_ROUTE(UFOE, DSI0, + MMSYS_ROUTE(UFOE, 0, DSI, 0, MT8173_DISP_REG_CONFIG_DISP_UFOE_MOUT_EN, MT8173_UFOE_MOUT_EN_DSI0, MT8173_UFOE_MOUT_EN_DSI0), - MMSYS_ROUTE(COLOR0, AAL0, + MMSYS_ROUTE(COLOR, 0, AAL, 0, MT8173_DISP_REG_CONFIG_DISP_COLOR0_SOUT_SEL_IN, MT8173_COLOR0_SOUT_MERGE, 0 /* SOUT to AAL */), - MMSYS_ROUTE(RDMA0, UFOE, + MMSYS_ROUTE(RDMA, 0, UFOE, 0, MT8173_DISP_REG_CONFIG_DISP_RDMA0_SOUT_SEL_IN, MT8173_RDMA0_SOUT_COLOR0, 0 /* SOUT to UFOE */), - MMSYS_ROUTE(OVL0, COLOR0, + MMSYS_ROUTE(OVL, 0, COLOR, 0, MT8173_DISP_REG_CONFIG_DISP_COLOR0_SEL_IN, MT8173_COLOR0_SEL_IN_OVL0, MT8173_COLOR0_SEL_IN_OVL0), - MMSYS_ROUTE(AAL0, COLOR0, + MMSYS_ROUTE(AAL, 0, COLOR, 0, MT8173_DISP_REG_CONFIG_DISP_AAL_SEL_IN, MT8173_AAL_SEL_IN_MERGE, 0 /* SEL_IN from COLOR0 */), - MMSYS_ROUTE(RDMA0, UFOE, + MMSYS_ROUTE(RDMA, 0, UFOE, 0, MT8173_DISP_REG_CONFIG_DISP_UFOE_SEL_IN, MT8173_UFOE_SEL_IN_RDMA0, 0 /* SEL_IN from RDMA0 */), - MMSYS_ROUTE(UFOE, DSI0, + MMSYS_ROUTE(UFOE, 0, DSI, 0, MT8173_DISP_REG_CONFIG_DSI0_SEL_IN, MT8173_DSI0_SEL_IN_UFOE, 0 /* SEL_IN from UFOE */), - MMSYS_ROUTE(OVL1, COLOR1, + MMSYS_ROUTE(OVL, 1, COLOR, 1, MT8173_DISP_REG_CONFIG_DISP_OVL1_MOUT_EN, MT8173_OVL1_MOUT_EN_COLOR1, MT8173_OVL1_MOUT_EN_COLOR1), - MMSYS_ROUTE(GAMMA, RDMA1, + MMSYS_ROUTE(GAMMA, 0, RDMA, 1, MT8173_DISP_REG_CONFIG_DISP_GAMMA_MOUT_EN, MT8173_GAMMA_MOUT_EN_RDMA1, MT8173_GAMMA_MOUT_EN_RDMA1), - MMSYS_ROUTE(RDMA1, DPI0, + MMSYS_ROUTE(RDMA, 1, DPI, 0, MT8173_DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, RDMA1_SOUT_DPI0), - MMSYS_ROUTE(OVL1, COLOR1, + MMSYS_ROUTE(OVL, 1, COLOR, 1, MT8173_DISP_REG_CONFIG_DISP_COLOR1_SEL_IN, COLOR1_SEL_IN_OVL1, COLOR1_SEL_IN_OVL1), - MMSYS_ROUTE(RDMA1, DPI0, + MMSYS_ROUTE(RDMA, 1, DPI, 0, MT8173_DISP_REG_CONFIG_DPI_SEL_IN, MT8173_DPI0_SEL_IN_MASK, MT8173_DPI0_SEL_IN_RDMA1), }; diff --git a/drivers/soc/mediatek/mt8183-mmsys.h b/drivers/soc/mediatek/mt8183-mmsys.h index 123384958c4b7f..cf221ef203d228 100644 --- a/drivers/soc/mediatek/mt8183-mmsys.h +++ b/drivers/soc/mediatek/mt8183-mmsys.h @@ -28,25 +28,25 @@ #define MT8183_MMSYS_SW0_RST_B 0x140 static const struct mtk_mmsys_routes mmsys_mt8183_routing_table[] = { - MMSYS_ROUTE(OVL0, OVL_2L0, + MMSYS_ROUTE(OVL, 0, OVL_2L, 0, MT8183_DISP_OVL0_MOUT_EN, MT8183_OVL0_MOUT_EN_OVL0_2L, MT8183_OVL0_MOUT_EN_OVL0_2L), - MMSYS_ROUTE(OVL_2L0, RDMA0, + MMSYS_ROUTE(OVL_2L, 0, RDMA, 0, MT8183_DISP_OVL0_2L_MOUT_EN, MT8183_OVL0_2L_MOUT_EN_DISP_PATH0, MT8183_OVL0_2L_MOUT_EN_DISP_PATH0), - MMSYS_ROUTE(OVL_2L1, RDMA1, + MMSYS_ROUTE(OVL_2L, 1, RDMA, 1, MT8183_DISP_OVL1_2L_MOUT_EN, MT8183_OVL1_2L_MOUT_EN_RDMA1, MT8183_OVL1_2L_MOUT_EN_RDMA1), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8183_DISP_DITHER0_MOUT_EN, MT8183_DITHER0_MOUT_IN_DSI0, MT8183_DITHER0_MOUT_IN_DSI0), - MMSYS_ROUTE(OVL_2L0, RDMA0, + MMSYS_ROUTE(OVL_2L, 0, RDMA, 0, MT8183_DISP_PATH0_SEL_IN, MT8183_DISP_PATH0_SEL_IN_OVL0_2L, MT8183_DISP_PATH0_SEL_IN_OVL0_2L), - MMSYS_ROUTE(RDMA1, DPI0, + MMSYS_ROUTE(RDMA, 1, DPI, 0, MT8183_DISP_DPI0_SEL_IN, MT8183_DPI0_SEL_IN_RDMA1, MT8183_DPI0_SEL_IN_RDMA1), - MMSYS_ROUTE(RDMA0, COLOR0, + MMSYS_ROUTE(RDMA, 0, COLOR, 0, MT8183_DISP_RDMA0_SOUT_SEL_IN, MT8183_RDMA0_SOUT_COLOR0, MT8183_RDMA0_SOUT_COLOR0), }; diff --git a/drivers/soc/mediatek/mt8186-mmsys.h b/drivers/soc/mediatek/mt8186-mmsys.h index 354664be72bd79..0c6941be6fa56b 100644 --- a/drivers/soc/mediatek/mt8186-mmsys.h +++ b/drivers/soc/mediatek/mt8186-mmsys.h @@ -63,37 +63,37 @@ #define MT8186_MMSYS_SW0_RST_B 0x160 static const struct mtk_mmsys_routes mmsys_mt8186_routing_table[] = { - MMSYS_ROUTE(OVL0, RDMA0, + MMSYS_ROUTE(OVL, 0, RDMA, 0, MT8186_DISP_OVL0_MOUT_EN, MT8186_OVL0_MOUT_EN_MASK, MT8186_OVL0_MOUT_TO_RDMA0), - MMSYS_ROUTE(OVL0, RDMA0, + MMSYS_ROUTE(OVL, 0, RDMA, 0, MT8186_DISP_RDMA0_SEL_IN, MT8186_RDMA0_SEL_IN_MASK, MT8186_RDMA0_FROM_OVL0), - MMSYS_ROUTE(OVL0, RDMA0, + MMSYS_ROUTE(OVL, 0, RDMA, 0, MT8186_MMSYS_OVL_CON, MT8186_MMSYS_OVL0_CON_MASK, MT8186_OVL0_GO_BLEND), - MMSYS_ROUTE(RDMA0, COLOR0, + MMSYS_ROUTE(RDMA, 0, COLOR, 0, MT8186_DISP_RDMA0_SOUT_SEL, MT8186_RDMA0_SOUT_SEL_MASK, MT8186_RDMA0_SOUT_TO_COLOR0), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8186_DISP_DITHER0_MOUT_EN, MT8186_DITHER0_MOUT_EN_MASK, MT8186_DITHER0_MOUT_TO_DSI0), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8186_DISP_DSI0_SEL_IN, MT8186_DSI0_SEL_IN_MASK, MT8186_DSI0_FROM_DITHER0), - MMSYS_ROUTE(OVL_2L0, RDMA1, + MMSYS_ROUTE(OVL_2L, 0, RDMA, 1, MT8186_DISP_OVL0_2L_MOUT_EN, MT8186_OVL0_2L_MOUT_EN_MASK, MT8186_OVL0_2L_MOUT_TO_RDMA1), - MMSYS_ROUTE(OVL_2L0, RDMA1, + MMSYS_ROUTE(OVL_2L, 0, RDMA, 1, MT8186_DISP_RDMA1_SEL_IN, MT8186_RDMA1_SEL_IN_MASK, MT8186_RDMA1_FROM_OVL0_2L), - MMSYS_ROUTE(OVL_2L0, RDMA1, + MMSYS_ROUTE(OVL_2L, 0, RDMA, 1, MT8186_MMSYS_OVL_CON, MT8186_MMSYS_OVL0_2L_CON_MASK, MT8186_OVL0_2L_GO_BLEND), - MMSYS_ROUTE(RDMA1, DPI0, + MMSYS_ROUTE(RDMA, 1, DPI, 0, MT8186_DISP_RDMA1_MOUT_EN, MT8186_RDMA1_MOUT_EN_MASK, MT8186_RDMA1_MOUT_TO_DPI0_SEL), - MMSYS_ROUTE(RDMA1, DPI0, + MMSYS_ROUTE(RDMA, 1, DPI, 0, MT8186_DISP_DPI0_SEL_IN, MT8186_DPI0_SEL_IN_MASK, MT8186_DPI0_FROM_RDMA1), }; diff --git a/drivers/soc/mediatek/mt8188-mmsys.h b/drivers/soc/mediatek/mt8188-mmsys.h index 99080afead7e7a..c70c4b46238124 100644 --- a/drivers/soc/mediatek/mt8188-mmsys.h +++ b/drivers/soc/mediatek/mt8188-mmsys.h @@ -202,124 +202,124 @@ static const u8 mmsys_mt8188_vdo1_rst_tb[] = { }; static const struct mtk_mmsys_routes mmsys_mt8188_routing_table[] = { - MMSYS_ROUTE(OVL0, RDMA0, + MMSYS_ROUTE(OVL, 0, RDMA, 0, MT8188_VDO0_OVL_MOUT_EN, MT8188_MOUT_DISP_OVL0_TO_DISP_RDMA0, MT8188_MOUT_DISP_OVL0_TO_DISP_RDMA0), - MMSYS_ROUTE(OVL0, WDMA0, + MMSYS_ROUTE(OVL, 0, WDMA, 0, MT8188_VDO0_OVL_MOUT_EN, MT8188_MOUT_DISP_OVL0_TO_DISP_WDMA0, MT8188_MOUT_DISP_OVL0_TO_DISP_WDMA0), - MMSYS_ROUTE(OVL0, RDMA0, + MMSYS_ROUTE(OVL, 0, RDMA, 0, MT8188_VDO0_DISP_RDMA_SEL, MT8188_SEL_IN_DISP_RDMA0_FROM_MASK, MT8188_SEL_IN_DISP_RDMA0_FROM_DISP_OVL0), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8188_VDO0_DSI0_SEL_IN, MT8188_SEL_IN_DSI0_FROM_MASK, MT8188_SEL_IN_DSI0_FROM_DISP_DITHER0), - MMSYS_ROUTE(DITHER0, MERGE0, + MMSYS_ROUTE(DITHER, 0, MERGE, 0, MT8188_VDO0_VPP_MERGE_SEL, MT8188_SEL_IN_VPP_MERGE_FROM_MASK, MT8188_SEL_IN_DP_INTF0_FROM_DISP_DITHER0), - MMSYS_ROUTE(DITHER0, DSC0, + MMSYS_ROUTE(DITHER, 0, DSC, 0, MT8188_VDO0_DSC_WARP_SEL, MT8188_SEL_IN_DSC_WRAP0C0_IN_FROM_MASK, MT8188_SEL_IN_DSC_WRAP0C0_IN_FROM_DISP_DITHER0), - MMSYS_ROUTE(DITHER0, DP_INTF0, + MMSYS_ROUTE(DITHER, 0, DP_INTF, 0, MT8188_VDO0_DP_INTF0_SEL_IN, MT8188_SEL_IN_DP_INTF0_FROM_MASK, MT8188_SEL_IN_DP_INTF0_FROM_DISP_DITHER0), - MMSYS_ROUTE(DSC0, MERGE0, + MMSYS_ROUTE(DSC, 0, MERGE, 0, MT8188_VDO0_VPP_MERGE_SEL, MT8188_SEL_IN_VPP_MERGE_FROM_MASK, MT8188_SEL_IN_VPP_MERGE_FROM_DSC_WRAP0_OUT), - MMSYS_ROUTE(MERGE0, DP_INTF0, + MMSYS_ROUTE(MERGE, 0, DP_INTF, 0, MT8188_VDO0_DP_INTF0_SEL_IN, MT8188_SEL_IN_DP_INTF0_FROM_MASK, MT8188_SEL_IN_DP_INTF0_FROM_VPP_MERGE), - MMSYS_ROUTE(DSC0, DSI0, + MMSYS_ROUTE(DSC, 0, DSI, 0, MT8188_VDO0_DSI0_SEL_IN, MT8188_SEL_IN_DSI0_FROM_MASK, MT8188_SEL_IN_DSI0_FROM_DSC_WRAP0_OUT), - MMSYS_ROUTE(RDMA0, COLOR0, + MMSYS_ROUTE(RDMA, 0, COLOR, 0, MT8188_VDO0_DISP_RDMA_SEL, GENMASK(1, 0), MT8188_SOUT_DISP_RDMA0_TO_DISP_COLOR0), - MMSYS_ROUTE(DITHER0, DSC0, + MMSYS_ROUTE(DITHER, 0, DSC, 0, MT8188_VDO0_DISP_DITHER0_SEL_OUT, MT8188_SOUT_DISP_DITHER0_TO_MASK, MT8188_SOUT_DISP_DITHER0_TO_DSC_WRAP0_IN), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8188_VDO0_DISP_DITHER0_SEL_OUT, MT8188_SOUT_DISP_DITHER0_TO_MASK, MT8188_SOUT_DISP_DITHER0_TO_DSI0), - MMSYS_ROUTE(DITHER0, MERGE0, + MMSYS_ROUTE(DITHER, 0, MERGE, 0, MT8188_VDO0_DISP_DITHER0_SEL_OUT, MT8188_SOUT_DISP_DITHER0_TO_MASK, MT8188_SOUT_DISP_DITHER0_TO_VPP_MERGE0), - MMSYS_ROUTE(DITHER0, DP_INTF0, + MMSYS_ROUTE(DITHER, 0, DP_INTF, 0, MT8188_VDO0_DISP_DITHER0_SEL_OUT, MT8188_SOUT_DISP_DITHER0_TO_MASK, MT8188_SOUT_DISP_DITHER0_TO_DP_INTF0), - MMSYS_ROUTE(MERGE0, DP_INTF0, + MMSYS_ROUTE(MERGE, 0, DP_INTF, 0, MT8188_VDO0_VPP_MERGE_SEL, MT8188_SOUT_VPP_MERGE_TO_MASK, MT8188_SOUT_VPP_MERGE_TO_DP_INTF0), - MMSYS_ROUTE(MERGE0, DPI0, + MMSYS_ROUTE(MERGE, 0, DPI, 0, MT8188_VDO0_VPP_MERGE_SEL, MT8188_SOUT_VPP_MERGE_TO_MASK, MT8188_SOUT_VPP_MERGE_TO_SINA_VIRTUAL0), - MMSYS_ROUTE(MERGE0, WDMA0, + MMSYS_ROUTE(MERGE, 0, WDMA, 0, MT8188_VDO0_VPP_MERGE_SEL, MT8188_SOUT_VPP_MERGE_TO_MASK, MT8188_SOUT_VPP_MERGE_TO_DISP_WDMA0), - MMSYS_ROUTE(MERGE0, DSC0, + MMSYS_ROUTE(MERGE, 0, DSC, 0, MT8188_VDO0_VPP_MERGE_SEL, MT8188_SOUT_VPP_MERGE_TO_MASK, MT8188_SOUT_VPP_MERGE_TO_DSC_WRAP0_IN), - MMSYS_ROUTE(DSC0, DSI0, + MMSYS_ROUTE(DSC, 0, DSI, 0, MT8188_VDO0_DSC_WARP_SEL, MT8188_SOUT_DSC_WRAP0_OUT_TO_MASK, MT8188_SOUT_DSC_WRAP0_OUT_TO_DSI0), - MMSYS_ROUTE(DSC0, MERGE0, + MMSYS_ROUTE(DSC, 0, MERGE, 0, MT8188_VDO0_DSC_WARP_SEL, MT8188_SOUT_DSC_WRAP0_OUT_TO_MASK, MT8188_SOUT_DSC_WRAP0_OUT_TO_VPP_MERGE), }; static const struct mtk_mmsys_routes mmsys_mt8188_vdo1_routing_table[] = { - MMSYS_ROUTE(MDP_RDMA0, MERGE1, + MMSYS_ROUTE(MDP_RDMA, 0, MERGE, 1, MT8188_VDO1_VPP_MERGE0_P0_SEL_IN, GENMASK(0, 0), MT8188_VPP_MERGE0_P0_SEL_IN_FROM_MDP_RDMA0), - MMSYS_ROUTE(MDP_RDMA1, MERGE1, + MMSYS_ROUTE(MDP_RDMA, 1, MERGE, 1, MT8188_VDO1_VPP_MERGE0_P1_SEL_IN, GENMASK(0, 0), MT8188_VPP_MERGE0_P1_SEL_IN_FROM_MDP_RDMA1), - MMSYS_ROUTE(MDP_RDMA2, MERGE2, + MMSYS_ROUTE(MDP_RDMA, 2, MERGE, 2, MT8188_VDO1_VPP_MERGE1_P0_SEL_IN, GENMASK(0, 0), MT8188_VPP_MERGE1_P0_SEL_IN_FROM_MDP_RDMA2), - MMSYS_ROUTE(MERGE1, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 1, ETHDR_MIXER, 0, MT8188_VDO1_MERGE0_ASYNC_SOUT_SEL, GENMASK(1, 0), MT8188_SOUT_TO_MIXER_IN1_SEL), - MMSYS_ROUTE(MERGE2, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 2, ETHDR_MIXER, 0, MT8188_VDO1_MERGE1_ASYNC_SOUT_SEL, GENMASK(1, 0), MT8188_SOUT_TO_MIXER_IN2_SEL), - MMSYS_ROUTE(MERGE3, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 3, ETHDR_MIXER, 0, MT8188_VDO1_MERGE2_ASYNC_SOUT_SEL, GENMASK(1, 0), MT8188_SOUT_TO_MIXER_IN3_SEL), - MMSYS_ROUTE(MERGE4, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 4, ETHDR_MIXER, 0, MT8188_VDO1_MERGE3_ASYNC_SOUT_SEL, GENMASK(1, 0), MT8188_SOUT_TO_MIXER_IN4_SEL), - MMSYS_ROUTE(ETHDR_MIXER, MERGE5, + MMSYS_ROUTE(ETHDR_MIXER, 0, MERGE, 5, MT8188_VDO1_MIXER_OUT_SOUT_SEL, GENMASK(0, 0), MT8188_MIXER_SOUT_TO_MERGE4_ASYNC_SEL), - MMSYS_ROUTE(MERGE1, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 1, ETHDR_MIXER, 0, MT8188_VDO1_MIXER_IN1_SEL_IN, GENMASK(0, 0), MT8188_MIXER_IN1_SEL_IN_FROM_MERGE0_ASYNC_SOUT), - MMSYS_ROUTE(MERGE2, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 2, ETHDR_MIXER, 0, MT8188_VDO1_MIXER_IN2_SEL_IN, GENMASK(0, 0), MT8188_MIXER_IN2_SEL_IN_FROM_MERGE1_ASYNC_SOUT), - MMSYS_ROUTE(MERGE3, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 3, ETHDR_MIXER, 0, MT8188_VDO1_MIXER_IN3_SEL_IN, GENMASK(0, 0), MT8188_MIXER_IN3_SEL_IN_FROM_MERGE2_ASYNC_SOUT), - MMSYS_ROUTE(MERGE4, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 4, ETHDR_MIXER, 0, MT8188_VDO1_MIXER_IN4_SEL_IN, GENMASK(0, 0), MT8188_MIXER_IN4_SEL_IN_FROM_MERGE3_ASYNC_SOUT), - MMSYS_ROUTE(ETHDR_MIXER, MERGE5, + MMSYS_ROUTE(ETHDR_MIXER, 0, MERGE, 5, MT8188_VDO1_MIXER_SOUT_SEL_IN, GENMASK(2, 0), MT8188_MIXER_SOUT_SEL_IN_FROM_DISP_MIXER), - MMSYS_ROUTE(ETHDR_MIXER, MERGE5, + MMSYS_ROUTE(ETHDR_MIXER, 0, MERGE, 5, MT8188_VDO1_MERGE4_ASYNC_SEL_IN, GENMASK(2, 0), MT8188_MERGE4_ASYNC_SEL_IN_FROM_MIXER_OUT_SOUT), - MMSYS_ROUTE(MERGE5, DPI1, + MMSYS_ROUTE(MERGE, 5, DPI, 1, MT8188_VDO1_DISP_DPI1_SEL_IN, GENMASK(1, 0), MT8188_DISP_DPI1_SEL_IN_FROM_VPP_MERGE4_MOUT), - MMSYS_ROUTE(MERGE5, DPI1, + MMSYS_ROUTE(MERGE, 5, DPI, 1, MT8188_VDO1_MERGE4_SOUT_SEL, GENMASK(3, 0), MT8188_MERGE4_SOUT_TO_DPI1_SEL), - MMSYS_ROUTE(MERGE5, DP_INTF1, + MMSYS_ROUTE(MERGE, 5, DP_INTF, 1, MT8188_VDO1_DISP_DP_INTF0_SEL_IN, GENMASK(1, 0), MT8188_DISP_DP_INTF0_SEL_IN_FROM_VPP_MERGE4_MOUT), - MMSYS_ROUTE(MERGE5, DP_INTF1, + MMSYS_ROUTE(MERGE, 5, DP_INTF, 1, MT8188_VDO1_MERGE4_SOUT_SEL, GENMASK(3, 0), MT8188_MERGE4_SOUT_TO_DP_INTF0_SEL), }; diff --git a/drivers/soc/mediatek/mt8192-mmsys.h b/drivers/soc/mediatek/mt8192-mmsys.h index 7cafa2455fd097..37ced5152ba7f5 100644 --- a/drivers/soc/mediatek/mt8192-mmsys.h +++ b/drivers/soc/mediatek/mt8192-mmsys.h @@ -31,34 +31,34 @@ #define MT8192_DSI0_SEL_IN_DITHER0 0x1 static const struct mtk_mmsys_routes mmsys_mt8192_routing_table[] = { - MMSYS_ROUTE(OVL_2L0, RDMA0, + MMSYS_ROUTE(OVL_2L, 0, RDMA, 0, MT8192_DISP_OVL0_2L_MOUT_EN, MT8192_OVL0_MOUT_EN_DISP_RDMA0, MT8192_OVL0_MOUT_EN_DISP_RDMA0), - MMSYS_ROUTE(OVL_2L2, RDMA4, + MMSYS_ROUTE(OVL_2L, 2, RDMA, 4, MT8192_DISP_OVL2_2L_MOUT_EN, MT8192_OVL2_2L_MOUT_EN_RDMA4, MT8192_OVL2_2L_MOUT_EN_RDMA4), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8192_DISP_DITHER0_MOUT_EN, MT8192_DITHER0_MOUT_IN_DSI0, MT8192_DITHER0_MOUT_IN_DSI0), - MMSYS_ROUTE(OVL_2L0, RDMA0, + MMSYS_ROUTE(OVL_2L, 0, RDMA, 0, MT8192_DISP_RDMA0_SEL_IN, MT8192_RDMA0_SEL_IN_OVL0_2L, MT8192_RDMA0_SEL_IN_OVL0_2L), - MMSYS_ROUTE(CCORR, AAL0, + MMSYS_ROUTE(CCORR, 0, AAL, 0, MT8192_DISP_AAL0_SEL_IN, MT8192_AAL0_SEL_IN_CCORR0, MT8192_AAL0_SEL_IN_CCORR0), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8192_DISP_DSI0_SEL_IN, MT8192_DSI0_SEL_IN_DITHER0, MT8192_DSI0_SEL_IN_DITHER0), - MMSYS_ROUTE(RDMA0, COLOR0, + MMSYS_ROUTE(RDMA, 0, COLOR, 0, MT8192_DISP_RDMA0_SOUT_SEL, MT8192_RDMA0_SOUT_COLOR0, MT8192_RDMA0_SOUT_COLOR0), - MMSYS_ROUTE(CCORR, AAL0, + MMSYS_ROUTE(CCORR, 0, AAL, 0, MT8192_DISP_CCORR0_SOUT_SEL, MT8192_CCORR0_SOUT_AAL0, MT8192_CCORR0_SOUT_AAL0), - MMSYS_ROUTE(OVL0, OVL_2L0, + MMSYS_ROUTE(OVL, 0, OVL_2L, 0, MT8192_MMSYS_OVL_MOUT_EN, MT8192_DISP_OVL0_GO_BG, MT8192_DISP_OVL0_GO_BG), - MMSYS_ROUTE(OVL_2L0, RDMA0, + MMSYS_ROUTE(OVL_2L, 0, RDMA, 0, MT8192_MMSYS_OVL_MOUT_EN, MT8192_DISP_OVL0_2L_GO_BLEND, MT8192_DISP_OVL0_2L_GO_BLEND), }; diff --git a/drivers/soc/mediatek/mt8195-mmsys.h b/drivers/soc/mediatek/mt8195-mmsys.h index f69929a2a4d4d1..3a58b9b74282ca 100644 --- a/drivers/soc/mediatek/mt8195-mmsys.h +++ b/drivers/soc/mediatek/mt8195-mmsys.h @@ -160,278 +160,279 @@ #define MT8195_SVPP3_MDP_RSZ BIT(5) static const struct mtk_mmsys_routes mmsys_mt8195_routing_table[] = { - MMSYS_ROUTE(OVL0, RDMA0, + MMSYS_ROUTE(OVL, 0, RDMA, 0, MT8195_VDO0_OVL_MOUT_EN, MT8195_MOUT_DISP_OVL0_TO_DISP_RDMA0, MT8195_MOUT_DISP_OVL0_TO_DISP_RDMA0), - MMSYS_ROUTE(OVL0, WDMA0, + MMSYS_ROUTE(OVL, 0, WDMA, 0, MT8195_VDO0_OVL_MOUT_EN, MT8195_MOUT_DISP_OVL0_TO_DISP_WDMA0, MT8195_MOUT_DISP_OVL0_TO_DISP_WDMA0), - MMSYS_ROUTE(OVL0, OVL1, + MMSYS_ROUTE(OVL, 0, OVL, 1, MT8195_VDO0_OVL_MOUT_EN, MT8195_MOUT_DISP_OVL0_TO_DISP_OVL1, MT8195_MOUT_DISP_OVL0_TO_DISP_OVL1), - MMSYS_ROUTE(OVL1, RDMA1, + MMSYS_ROUTE(OVL, 1, RDMA, 1, MT8195_VDO0_OVL_MOUT_EN, MT8195_MOUT_DISP_OVL1_TO_DISP_RDMA1, MT8195_MOUT_DISP_OVL1_TO_DISP_RDMA1), - MMSYS_ROUTE(OVL1, WDMA1, + MMSYS_ROUTE(OVL, 1, WDMA, 1, MT8195_VDO0_OVL_MOUT_EN, MT8195_MOUT_DISP_OVL1_TO_DISP_WDMA1, MT8195_MOUT_DISP_OVL1_TO_DISP_WDMA1), - MMSYS_ROUTE(OVL1, OVL0, + MMSYS_ROUTE(OVL, 1, OVL, 0, MT8195_VDO0_OVL_MOUT_EN, MT8195_MOUT_DISP_OVL1_TO_DISP_OVL0, MT8195_MOUT_DISP_OVL1_TO_DISP_OVL0), - MMSYS_ROUTE(DSC0, MERGE0, + MMSYS_ROUTE(DSC, 0, MERGE, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_VPP_MERGE_FROM_MASK, MT8195_SEL_IN_VPP_MERGE_FROM_DSC_WRAP0_OUT), - MMSYS_ROUTE(DITHER1, MERGE0, + MMSYS_ROUTE(DITHER, 1, MERGE, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_VPP_MERGE_FROM_MASK, MT8195_SEL_IN_VPP_MERGE_FROM_DISP_DITHER1), - MMSYS_ROUTE(MERGE5, MERGE0, + MMSYS_ROUTE(MERGE, 5, MERGE, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_VPP_MERGE_FROM_MASK, MT8195_SEL_IN_VPP_MERGE_FROM_VDO1_VIRTUAL0), - MMSYS_ROUTE(DITHER0, DSC0, + MMSYS_ROUTE(DITHER, 0, DSC, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP0_IN_FROM_MASK, MT8195_SEL_IN_DSC_WRAP0_IN_FROM_DISP_DITHER0), - MMSYS_ROUTE(MERGE0, DSC0, + MMSYS_ROUTE(MERGE, 0, DSC, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP0_IN_FROM_MASK, MT8195_SEL_IN_DSC_WRAP0_IN_FROM_VPP_MERGE), - MMSYS_ROUTE(DITHER1, DSC1, + MMSYS_ROUTE(DITHER, 1, DSC, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_IN_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_IN_FROM_DISP_DITHER1), - MMSYS_ROUTE(MERGE0, DSC1, + MMSYS_ROUTE(MERGE, 0, DSC, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_IN_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_IN_FROM_VPP_MERGE), - MMSYS_ROUTE(MERGE0, DP_INTF1, + MMSYS_ROUTE(MERGE, 0, DP_INTF, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_MASK, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_VPP_MERGE), - MMSYS_ROUTE(MERGE0, DPI0, + MMSYS_ROUTE(MERGE, 0, DPI, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_MASK, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_VPP_MERGE), - MMSYS_ROUTE(MERGE0, DPI1, + MMSYS_ROUTE(MERGE, 0, DPI, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_MASK, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_VPP_MERGE), - MMSYS_ROUTE(DSC1, DP_INTF1, + MMSYS_ROUTE(DSC, 1, DP_INTF, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_MASK, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_DSC_WRAP1_OUT), - MMSYS_ROUTE(DSC1, DPI0, + MMSYS_ROUTE(DSC, 1, DPI, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_MASK, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_DSC_WRAP1_OUT), - MMSYS_ROUTE(DSC1, DPI1, + MMSYS_ROUTE(DSC, 1, DPI, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_MASK, MT8195_SEL_IN_SINA_VIRTUAL0_FROM_DSC_WRAP1_OUT), - MMSYS_ROUTE(DSC0, DP_INTF1, + MMSYS_ROUTE(DSC, 0, DP_INTF, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_SINB_VIRTUAL0_FROM_MASK, MT8195_SEL_IN_SINB_VIRTUAL0_FROM_DSC_WRAP0_OUT), - MMSYS_ROUTE(DSC0, DPI0, + MMSYS_ROUTE(DSC, 0, DPI, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_SINB_VIRTUAL0_FROM_MASK, MT8195_SEL_IN_SINB_VIRTUAL0_FROM_DSC_WRAP0_OUT), - MMSYS_ROUTE(DSC0, DPI1, + MMSYS_ROUTE(DSC, 0, DPI, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_SINB_VIRTUAL0_FROM_MASK, MT8195_SEL_IN_SINB_VIRTUAL0_FROM_DSC_WRAP0_OUT), - MMSYS_ROUTE(DSC1, DP_INTF0, + MMSYS_ROUTE(DSC, 1, DP_INTF, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DP_INTF0_FROM_MASK, MT8195_SEL_IN_DP_INTF0_FROM_DSC_WRAP1_OUT), - MMSYS_ROUTE(MERGE0, DP_INTF0, + MMSYS_ROUTE(MERGE, 0, DP_INTF, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DP_INTF0_FROM_MASK, MT8195_SEL_IN_DP_INTF0_FROM_VPP_MERGE), - MMSYS_ROUTE(MERGE5, DP_INTF0, + MMSYS_ROUTE(MERGE, 5, DP_INTF, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DP_INTF0_FROM_MASK, MT8195_SEL_IN_DP_INTF0_FROM_VDO1_VIRTUAL0), - MMSYS_ROUTE(DSC0, DSI0, + MMSYS_ROUTE(DSC, 0, DSI, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSI0_FROM_MASK, MT8195_SEL_IN_DSI0_FROM_DSC_WRAP0_OUT), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSI0_FROM_MASK, MT8195_SEL_IN_DSI0_FROM_DISP_DITHER0), - MMSYS_ROUTE(DSC1, DSI1, + MMSYS_ROUTE(DSC, 1, DSI, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSI1_FROM_MASK, MT8195_SEL_IN_DSI1_FROM_DSC_WRAP1_OUT), - MMSYS_ROUTE(MERGE0, DSI1, + MMSYS_ROUTE(MERGE, 0, DSI, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSI1_FROM_MASK, MT8195_SEL_IN_DSI1_FROM_VPP_MERGE), - MMSYS_ROUTE(OVL1, WDMA1, + MMSYS_ROUTE(OVL, 1, WDMA, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DISP_WDMA1_FROM_MASK, MT8195_SEL_IN_DISP_WDMA1_FROM_DISP_OVL1), - MMSYS_ROUTE(MERGE0, WDMA1, + MMSYS_ROUTE(MERGE, 0, WDMA, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DISP_WDMA1_FROM_MASK, MT8195_SEL_IN_DISP_WDMA1_FROM_VPP_MERGE), - MMSYS_ROUTE(DSC1, DSI1, + MMSYS_ROUTE(DSC, 1, DSI, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DSC_WRAP1_IN), - MMSYS_ROUTE(DSC1, DP_INTF0, + MMSYS_ROUTE(DSC, 1, DP_INTF, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DSC_WRAP1_IN), - MMSYS_ROUTE(DSC1, DP_INTF1, + MMSYS_ROUTE(DSC, 1, DP_INTF, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DSC_WRAP1_IN), - MMSYS_ROUTE(DSC1, DPI0, + MMSYS_ROUTE(DSC, 1, DPI, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DSC_WRAP1_IN), - MMSYS_ROUTE(DSC1, DPI1, + MMSYS_ROUTE(DSC, 1, DPI, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DSC_WRAP1_IN), - MMSYS_ROUTE(DSC1, MERGE0, + MMSYS_ROUTE(DSC, 1, MERGE, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DSC_WRAP1_IN), - MMSYS_ROUTE(DITHER1, DSI1, + MMSYS_ROUTE(DITHER, 1, DSI, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DISP_DITHER1), - MMSYS_ROUTE(DITHER1, DP_INTF0, + MMSYS_ROUTE(DITHER, 1, DP_INTF, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DISP_DITHER1), - MMSYS_ROUTE(DITHER1, DPI0, + MMSYS_ROUTE(DITHER, 1, DPI, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DISP_DITHER1), - MMSYS_ROUTE(DITHER1, DPI1, + MMSYS_ROUTE(DITHER, 1, DPI, 1, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DSC_WRAP1_FROM_MASK, MT8195_SEL_IN_DSC_WRAP1_OUT_FROM_DISP_DITHER1), - MMSYS_ROUTE(OVL0, WDMA0, + MMSYS_ROUTE(OVL, 0, WDMA, 0, MT8195_VDO0_SEL_IN, MT8195_SEL_IN_DISP_WDMA0_FROM_MASK, MT8195_SEL_IN_DISP_WDMA0_FROM_DISP_OVL0), - MMSYS_ROUTE(DITHER0, DSC0, + MMSYS_ROUTE(DITHER, 0, DSC, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DISP_DITHER0_TO_MASK, MT8195_SOUT_DISP_DITHER0_TO_DSC_WRAP0_IN), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DISP_DITHER0_TO_MASK, MT8195_SOUT_DISP_DITHER0_TO_DSI0), - MMSYS_ROUTE(DITHER1, DSC1, + MMSYS_ROUTE(DITHER, 1, DSC, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DISP_DITHER1_TO_MASK, MT8195_SOUT_DISP_DITHER1_TO_DSC_WRAP1_IN), - MMSYS_ROUTE(DITHER1, MERGE0, + MMSYS_ROUTE(DITHER, 1, MERGE, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DISP_DITHER1_TO_MASK, MT8195_SOUT_DISP_DITHER1_TO_VPP_MERGE), - MMSYS_ROUTE(DITHER1, DSI1, + MMSYS_ROUTE(DITHER, 1, DSI, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DISP_DITHER1_TO_MASK, MT8195_SOUT_DISP_DITHER1_TO_DSC_WRAP1_OUT), - MMSYS_ROUTE(DITHER1, DP_INTF0, + MMSYS_ROUTE(DITHER, 1, DP_INTF, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DISP_DITHER1_TO_MASK, MT8195_SOUT_DISP_DITHER1_TO_DSC_WRAP1_OUT), - MMSYS_ROUTE(DITHER1, DP_INTF1, + MMSYS_ROUTE(DITHER, 1, DP_INTF, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DISP_DITHER1_TO_MASK, MT8195_SOUT_DISP_DITHER1_TO_DSC_WRAP1_OUT), - MMSYS_ROUTE(DITHER1, DPI0, + MMSYS_ROUTE(DITHER, 1, DPI, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DISP_DITHER1_TO_MASK, MT8195_SOUT_DISP_DITHER1_TO_DSC_WRAP1_OUT), - MMSYS_ROUTE(DITHER1, DPI1, + MMSYS_ROUTE(DITHER, 1, DPI, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DISP_DITHER1_TO_MASK, MT8195_SOUT_DISP_DITHER1_TO_DSC_WRAP1_OUT), - MMSYS_ROUTE(MERGE5, MERGE0, + MMSYS_ROUTE(MERGE, 5, MERGE, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VDO1_VIRTUAL0_TO_MASK, MT8195_SOUT_VDO1_VIRTUAL0_TO_VPP_MERGE), - MMSYS_ROUTE(MERGE5, DP_INTF0, + MMSYS_ROUTE(MERGE, 5, DP_INTF, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VDO1_VIRTUAL0_TO_MASK, MT8195_SOUT_VDO1_VIRTUAL0_TO_DP_INTF0), - MMSYS_ROUTE(MERGE0, DSI1, + MMSYS_ROUTE(MERGE, 0, DSI, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VPP_MERGE_TO_MASK, MT8195_SOUT_VPP_MERGE_TO_DSI1), - MMSYS_ROUTE(MERGE0, DP_INTF0, + MMSYS_ROUTE(MERGE, 0, DP_INTF, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VPP_MERGE_TO_MASK, MT8195_SOUT_VPP_MERGE_TO_DP_INTF0), - MMSYS_ROUTE(MERGE0, DP_INTF1, + MMSYS_ROUTE(MERGE, 0, DP_INTF, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VPP_MERGE_TO_MASK, MT8195_SOUT_VPP_MERGE_TO_SINA_VIRTUAL0), - MMSYS_ROUTE(MERGE0, DPI0, + MMSYS_ROUTE(MERGE, 0, DPI, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VPP_MERGE_TO_MASK, MT8195_SOUT_VPP_MERGE_TO_SINA_VIRTUAL0), - MMSYS_ROUTE(MERGE0, DPI1, + MMSYS_ROUTE(MERGE, 0, DPI, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VPP_MERGE_TO_MASK, MT8195_SOUT_VPP_MERGE_TO_SINA_VIRTUAL0), - MMSYS_ROUTE(MERGE0, WDMA1, + MMSYS_ROUTE(MERGE, 0, WDMA, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VPP_MERGE_TO_MASK, MT8195_SOUT_VPP_MERGE_TO_DISP_WDMA1), - MMSYS_ROUTE(MERGE0, DSC0, + MMSYS_ROUTE(MERGE, 0, DSC, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VPP_MERGE_TO_MASK, MT8195_SOUT_VPP_MERGE_TO_DSC_WRAP0_IN), - MMSYS_ROUTE(MERGE0, DSC1, + MMSYS_ROUTE(MERGE, 0, DSC, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_VPP_MERGE_TO_DSC_WRAP1_IN_MASK, MT8195_SOUT_VPP_MERGE_TO_DSC_WRAP1_IN), - MMSYS_ROUTE(DSC0, DSI0, + MMSYS_ROUTE(DSC, 0, DSI, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP0_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP0_OUT_TO_DSI0), - MMSYS_ROUTE(DSC0, DP_INTF1, + MMSYS_ROUTE(DSC, 0, DP_INTF, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP0_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP0_OUT_TO_SINB_VIRTUAL0), - MMSYS_ROUTE(DSC0, DPI0, + MMSYS_ROUTE(DSC, 0, DPI, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP0_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP0_OUT_TO_SINB_VIRTUAL0), - MMSYS_ROUTE(DSC0, DPI1, + MMSYS_ROUTE(DSC, 0, DPI, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP0_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP0_OUT_TO_SINB_VIRTUAL0), - MMSYS_ROUTE(DSC0, MERGE0, + MMSYS_ROUTE(DSC, 0, MERGE, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP0_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP0_OUT_TO_VPP_MERGE), - MMSYS_ROUTE(DSC1, DSI1, + MMSYS_ROUTE(DSC, 1, DSI, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP1_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP1_OUT_TO_DSI1), - MMSYS_ROUTE(DSC1, DP_INTF0, + MMSYS_ROUTE(DSC, 1, DP_INTF, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP1_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP1_OUT_TO_DP_INTF0), - MMSYS_ROUTE(DSC1, DP_INTF1, + MMSYS_ROUTE(DSC, 1, DP_INTF, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP1_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP1_OUT_TO_SINA_VIRTUAL0), - MMSYS_ROUTE(DSC1, DPI0, + MMSYS_ROUTE(DSC, 1, DPI, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP1_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP1_OUT_TO_SINA_VIRTUAL0), - MMSYS_ROUTE(DSC1, DPI1, + MMSYS_ROUTE(DSC, 1, DPI, 1, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP1_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP1_OUT_TO_SINA_VIRTUAL0), - MMSYS_ROUTE(DSC1, MERGE0, + MMSYS_ROUTE(DSC, 1, MERGE, 0, MT8195_VDO0_SEL_OUT, MT8195_SOUT_DSC_WRAP1_OUT_TO_MASK, MT8195_SOUT_DSC_WRAP1_OUT_TO_VPP_MERGE), }; static const struct mtk_mmsys_routes mmsys_mt8195_vdo1_routing_table[] = { - MMSYS_ROUTE(MDP_RDMA0, MERGE1, + MMSYS_ROUTE(MDP_RDMA, 0, MERGE, 1, MT8195_VDO1_VPP_MERGE0_P0_SEL_IN, GENMASK(0, 0), MT8195_VPP_MERGE0_P0_SEL_IN_FROM_MDP_RDMA0), - MMSYS_ROUTE(MDP_RDMA1, MERGE1, + MMSYS_ROUTE(MDP_RDMA, 1, MERGE, 1, MT8195_VDO1_VPP_MERGE0_P1_SEL_IN, GENMASK(0, 0), MT8195_VPP_MERGE0_P1_SEL_IN_FROM_MDP_RDMA1), - MMSYS_ROUTE(MDP_RDMA2, MERGE2, + MMSYS_ROUTE(MDP_RDMA, 2, MERGE, 2, MT8195_VDO1_VPP_MERGE1_P0_SEL_IN, GENMASK(0, 0), MT8195_VPP_MERGE1_P0_SEL_IN_FROM_MDP_RDMA2), - MMSYS_ROUTE(MERGE1, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 1, ETHDR_MIXER, 0, MT8195_VDO1_MERGE0_ASYNC_SOUT_SEL, GENMASK(1, 0), MT8195_SOUT_TO_MIXER_IN1_SEL), - MMSYS_ROUTE(MERGE2, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 2, ETHDR_MIXER, 0, MT8195_VDO1_MERGE1_ASYNC_SOUT_SEL, GENMASK(1, 0), MT8195_SOUT_TO_MIXER_IN2_SEL), - MMSYS_ROUTE(MERGE3, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 3, ETHDR_MIXER, 0, MT8195_VDO1_MERGE2_ASYNC_SOUT_SEL, GENMASK(1, 0), MT8195_SOUT_TO_MIXER_IN3_SEL), - MMSYS_ROUTE(MERGE4, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 4, ETHDR_MIXER, 0, MT8195_VDO1_MERGE3_ASYNC_SOUT_SEL, GENMASK(1, 0), MT8195_SOUT_TO_MIXER_IN4_SEL), - MMSYS_ROUTE(ETHDR_MIXER, MERGE5, + MMSYS_ROUTE(ETHDR_MIXER, 0, MERGE, 5, MT8195_VDO1_MIXER_OUT_SOUT_SEL, GENMASK(0, 0), MT8195_MIXER_SOUT_TO_MERGE4_ASYNC_SEL), - MMSYS_ROUTE(MERGE1, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 1, ETHDR_MIXER, 0, MT8195_VDO1_MIXER_IN1_SEL_IN, GENMASK(0, 0), MT8195_MIXER_IN1_SEL_IN_FROM_MERGE0_ASYNC_SOUT), - MMSYS_ROUTE(MERGE2, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 2, ETHDR_MIXER, 0, MT8195_VDO1_MIXER_IN2_SEL_IN, GENMASK(0, 0), MT8195_MIXER_IN2_SEL_IN_FROM_MERGE1_ASYNC_SOUT), - MMSYS_ROUTE(MERGE3, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 3, ETHDR_MIXER, 0, MT8195_VDO1_MIXER_IN3_SEL_IN, GENMASK(0, 0), MT8195_MIXER_IN3_SEL_IN_FROM_MERGE2_ASYNC_SOUT), - MMSYS_ROUTE(MERGE4, ETHDR_MIXER, + MMSYS_ROUTE(MERGE, 4, ETHDR_MIXER, 0, MT8195_VDO1_MIXER_IN4_SEL_IN, GENMASK(0, 0), MT8195_MIXER_IN4_SEL_IN_FROM_MERGE3_ASYNC_SOUT), - MMSYS_ROUTE(ETHDR_MIXER, MERGE5, + MMSYS_ROUTE(ETHDR_MIXER, 0, MERGE, 5, MT8195_VDO1_MIXER_SOUT_SEL_IN, GENMASK(2, 0), MT8195_MIXER_SOUT_SEL_IN_FROM_DISP_MIXER), - MMSYS_ROUTE(ETHDR_MIXER, MERGE5, + MMSYS_ROUTE(ETHDR_MIXER, 0, MERGE, 5, MT8195_VDO1_MERGE4_ASYNC_SEL_IN, GENMASK(2, 0), MT8195_MERGE4_ASYNC_SEL_IN_FROM_MIXER_OUT_SOUT), - MMSYS_ROUTE(MERGE5, DPI1, + MMSYS_ROUTE(MERGE, 5, DPI, 1, MT8195_VDO1_DISP_DPI1_SEL_IN, GENMASK(1, 0), MT8195_DISP_DPI1_SEL_IN_FROM_VPP_MERGE4_MOUT), - MMSYS_ROUTE(MERGE5, DPI1, + MMSYS_ROUTE(MERGE, 5, DPI, 1, MT8195_VDO1_MERGE4_SOUT_SEL, GENMASK(1, 0), MT8195_MERGE4_SOUT_TO_DPI1_SEL), - MMSYS_ROUTE(MERGE5, DP_INTF1, + MMSYS_ROUTE(MERGE, 5, DP_INTF, 1, MT8195_VDO1_DISP_DP_INTF0_SEL_IN, GENMASK(1, 0), MT8195_DISP_DP_INTF0_SEL_IN_FROM_VPP_MERGE4_MOUT), - MMSYS_ROUTE(MERGE5, DP_INTF1, + MMSYS_ROUTE(MERGE, 5, DP_INTF, 1, MT8195_VDO1_MERGE4_SOUT_SEL, GENMASK(1, 0), MT8195_MERGE4_SOUT_TO_DP_INTF0_SEL), }; + #endif /* __SOC_MEDIATEK_MT8195_MMSYS_H */ diff --git a/drivers/soc/mediatek/mt8365-mmsys.h b/drivers/soc/mediatek/mt8365-mmsys.h index 533a3fd0923b68..b438ab7ae00bc3 100644 --- a/drivers/soc/mediatek/mt8365-mmsys.h +++ b/drivers/soc/mediatek/mt8365-mmsys.h @@ -28,35 +28,35 @@ #define MT8365_DPI0_SEL_IN_RDMA1 0x0 static const struct mtk_mmsys_routes mt8365_mmsys_routing_table[] = { - MMSYS_ROUTE(OVL0, RDMA0, + MMSYS_ROUTE(OVL, 0, RDMA, 0, MT8365_DISP_REG_CONFIG_DISP_OVL0_MOUT_EN, MT8365_DISP_MS_IN_OUT_MASK, MT8365_OVL0_MOUT_PATH0_SEL), - MMSYS_ROUTE(OVL0, RDMA0, + MMSYS_ROUTE(OVL, 0, RDMA, 0, MT8365_DISP_REG_CONFIG_DISP_RDMA0_SEL_IN, MT8365_DISP_MS_IN_OUT_MASK, MT8365_RDMA0_SEL_IN_OVL0), - MMSYS_ROUTE(RDMA0, COLOR0, + MMSYS_ROUTE(RDMA, 0, COLOR, 0, MT8365_DISP_REG_CONFIG_DISP_RDMA0_SOUT_SEL, MT8365_DISP_MS_IN_OUT_MASK, MT8365_RDMA0_SOUT_COLOR0), - MMSYS_ROUTE(COLOR0, CCORR, + MMSYS_ROUTE(COLOR, 0, CCORR, 0, MT8365_DISP_REG_CONFIG_DISP_COLOR0_SEL_IN, MT8365_DISP_MS_IN_OUT_MASK, MT8365_DISP_COLOR_SEL_IN_COLOR0), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8365_DISP_REG_CONFIG_DISP_DITHER0_MOUT_EN, MT8365_DISP_MS_IN_OUT_MASK, MT8365_DITHER_MOUT_EN_DSI0), - MMSYS_ROUTE(DITHER0, DSI0, + MMSYS_ROUTE(DITHER, 0, DSI, 0, MT8365_DISP_REG_CONFIG_DISP_DSI0_SEL_IN, MT8365_DISP_MS_IN_OUT_MASK, MT8365_DSI0_SEL_IN_DITHER), - MMSYS_ROUTE(RDMA0, COLOR0, + MMSYS_ROUTE(RDMA, 0, COLOR, 0, MT8365_DISP_REG_CONFIG_DISP_RDMA0_RSZ0_SEL_IN, MT8365_DISP_MS_IN_OUT_MASK, MT8365_RDMA0_RSZ0_SEL_IN_RDMA0), - MMSYS_ROUTE(RDMA1, DPI0, + MMSYS_ROUTE(RDMA, 1, DPI, 0, MT8365_DISP_REG_CONFIG_DISP_LVDS_SYS_CFG_00, MT8365_LVDS_SYS_CFG_00_SEL_LVDS_PXL_CLK, MT8365_LVDS_SYS_CFG_00_SEL_LVDS_PXL_CLK), - MMSYS_ROUTE(RDMA1, DPI0, + MMSYS_ROUTE(RDMA, 1, DPI, 0, MT8365_DISP_REG_CONFIG_DISP_DPI0_SEL_IN, MT8365_DISP_MS_IN_OUT_MASK, MT8365_DPI0_SEL_IN_RDMA1), - MMSYS_ROUTE(RDMA1, DPI0, + MMSYS_ROUTE(RDMA, 1, DPI, 0, MT8365_DISP_REG_CONFIG_DISP_RDMA1_SOUT_SEL, MT8365_DISP_MS_IN_OUT_MASK, MT8365_RDMA1_SOUT_DPI0), }; diff --git a/drivers/soc/mediatek/mtk-mmsys.h b/drivers/soc/mediatek/mtk-mmsys.h index fe628d5f519861..a655cf0062171c 100644 --- a/drivers/soc/mediatek/mtk-mmsys.h +++ b/drivers/soc/mediatek/mtk-mmsys.h @@ -80,18 +80,25 @@ #define MMSYS_RST_NR(bank, bit) (((bank) * 32) + (bit)) +/* Temporary compatibility definitions */ +#define DDP_COMPONENT_BLS0 DDP_COMPONENT_BLS +#define DDP_COMPONENT_CCORR0 DDP_COMPONENT_CCORR +#define DDP_COMPONENT_UFOE0 DDP_COMPONENT_UFOE +#define DDP_COMPONENT_GAMMA0 DDP_COMPONENT_GAMMA +#define DDP_COMPONENT_ETHDR_MIXER0 DDP_COMPONENT_ETHDR_MIXER + /* * This macro adds a compile time check to make sure that the in/out * selection bit(s) fit in the register mask, similar to bitfield * macros, but this does not transform the value. */ -#define MMSYS_ROUTE(from, to, reg_addr, reg_mask, selection) \ - { DDP_COMPONENT_##from, DDP_COMPONENT_##to, reg_addr, reg_mask, \ - (__BUILD_BUG_ON_ZERO_MSG((reg_mask) == 0, "Invalid mask") + \ - __BUILD_BUG_ON_ZERO_MSG(~(reg_mask) & (selection), \ - #selection " does not fit in " \ - #reg_mask) + \ - (selection)) \ +#define MMSYS_ROUTE(from, fsid, to, tsid, reg_addr, reg_mask, selection) \ + { DDP_COMPONENT_##from##fsid, DDP_COMPONENT_##to##tsid, reg_addr, reg_mask, \ + (__BUILD_BUG_ON_ZERO_MSG((reg_mask) == 0, "Invalid mask") + \ + __BUILD_BUG_ON_ZERO_MSG(~(reg_mask) & (selection), \ + #selection " does not fit in " \ + #reg_mask) + \ + (selection)) \ } struct mtk_mmsys_routes { From 7a5e16e5dffa19d87b7f0aee924c02583d5f4ab8 Mon Sep 17 00:00:00 2001 From: AngeloGioacchino Del Regno Date: Wed, 1 Jul 2026 14:20:43 +0200 Subject: [PATCH 006/857] soc: mediatek: mtk-mmsys: Use MMSYS_ROUTE() in default routing table All of the mtk_mmsys_routes tables for all SoCs were converted to use the MMSYS_ROUTE() macro but the default one used for MT2701, MT2712 and SoCs from that generation was not: convert this one as well. This brings no functional change. Signed-off-by: AngeloGioacchino Del Regno --- drivers/soc/mediatek/mtk-mmsys.h | 279 +++++++++++++------------------ 1 file changed, 114 insertions(+), 165 deletions(-) diff --git a/drivers/soc/mediatek/mtk-mmsys.h b/drivers/soc/mediatek/mtk-mmsys.h index a655cf0062171c..c51a5e334e2917 100644 --- a/drivers/soc/mediatek/mtk-mmsys.h +++ b/drivers/soc/mediatek/mtk-mmsys.h @@ -158,171 +158,120 @@ struct mtk_mmsys_driver_data { * to an independent table. */ static const struct mtk_mmsys_routes mmsys_default_routing_table[] = { - { - DDP_COMPONENT_BLS, DDP_COMPONENT_DSI0, - DISP_REG_CONFIG_OUT_SEL, BLS_RDMA1_DSI_DPI_MASK, - BLS_TO_DSI_RDMA1_TO_DPI1 - }, { - DDP_COMPONENT_BLS, DDP_COMPONENT_DSI0, - DISP_REG_CONFIG_DSI_SEL, DSI_SEL_IN_MASK, - DSI_SEL_IN_BLS - }, { - DDP_COMPONENT_BLS, DDP_COMPONENT_DPI0, - DISP_REG_CONFIG_OUT_SEL, BLS_RDMA1_DSI_DPI_MASK, - BLS_TO_DPI_RDMA1_TO_DSI - }, { - DDP_COMPONENT_BLS, DDP_COMPONENT_DPI0, - DISP_REG_CONFIG_DSI_SEL, DSI_SEL_IN_MASK, - DSI_SEL_IN_RDMA - }, { - DDP_COMPONENT_BLS, DDP_COMPONENT_DPI0, - DISP_REG_CONFIG_DPI_SEL, DPI_SEL_IN_MASK, - DPI_SEL_IN_BLS - }, { - DDP_COMPONENT_GAMMA, DDP_COMPONENT_RDMA1, - DISP_REG_CONFIG_DISP_GAMMA_MOUT_EN, GAMMA_MOUT_EN_RDMA1, - GAMMA_MOUT_EN_RDMA1 - }, { - DDP_COMPONENT_OD0, DDP_COMPONENT_RDMA0, - DISP_REG_CONFIG_DISP_OD_MOUT_EN, OD_MOUT_EN_RDMA0, - OD_MOUT_EN_RDMA0 - }, { - DDP_COMPONENT_OD1, DDP_COMPONENT_RDMA1, - DISP_REG_CONFIG_DISP_OD_MOUT_EN, OD1_MOUT_EN_RDMA1, - OD1_MOUT_EN_RDMA1 - }, { - DDP_COMPONENT_OVL0, DDP_COMPONENT_COLOR0, - DISP_REG_CONFIG_DISP_OVL0_MOUT_EN, OVL0_MOUT_EN_COLOR0, - OVL0_MOUT_EN_COLOR0 - }, { - DDP_COMPONENT_OVL0, DDP_COMPONENT_COLOR0, - DISP_REG_CONFIG_DISP_COLOR0_SEL_IN, COLOR0_SEL_IN_OVL0, - COLOR0_SEL_IN_OVL0 - }, { - DDP_COMPONENT_OVL0, DDP_COMPONENT_RDMA0, - DISP_REG_CONFIG_DISP_OVL_MOUT_EN, OVL_MOUT_EN_RDMA, - OVL_MOUT_EN_RDMA - }, { - DDP_COMPONENT_OVL1, DDP_COMPONENT_COLOR1, - DISP_REG_CONFIG_DISP_OVL1_MOUT_EN, OVL1_MOUT_EN_COLOR1, - OVL1_MOUT_EN_COLOR1 - }, { - DDP_COMPONENT_OVL1, DDP_COMPONENT_COLOR1, - DISP_REG_CONFIG_DISP_COLOR1_SEL_IN, COLOR1_SEL_IN_OVL1, - COLOR1_SEL_IN_OVL1 - }, { - DDP_COMPONENT_RDMA0, DDP_COMPONENT_DPI0, - DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, - RDMA0_SOUT_DPI0 - }, { - DDP_COMPONENT_RDMA0, DDP_COMPONENT_DPI1, - DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, - RDMA0_SOUT_DPI1 - }, { - DDP_COMPONENT_RDMA0, DDP_COMPONENT_DSI1, - DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, - RDMA0_SOUT_DSI1 - }, { - DDP_COMPONENT_RDMA0, DDP_COMPONENT_DSI2, - DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, - RDMA0_SOUT_DSI2 - }, { - DDP_COMPONENT_RDMA0, DDP_COMPONENT_DSI3, - DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, - RDMA0_SOUT_DSI3 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DPI0, - DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, - RDMA1_SOUT_DPI0 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DPI0, - DISP_REG_CONFIG_DPI_SEL_IN, DPI0_SEL_IN_MASK, - DPI0_SEL_IN_RDMA1 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DPI1, - DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, - RDMA1_SOUT_DPI1 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DPI1, - DISP_REG_CONFIG_DPI_SEL_IN, DPI1_SEL_IN_MASK, - DPI1_SEL_IN_RDMA1 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DSI0, - DISP_REG_CONFIG_DSIE_SEL_IN, DSI0_SEL_IN_MASK, - DSI0_SEL_IN_RDMA1 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DSI1, - DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, - RDMA1_SOUT_DSI1 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DSI1, - DISP_REG_CONFIG_DSIO_SEL_IN, DSI1_SEL_IN_MASK, - DSI1_SEL_IN_RDMA1 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DSI2, - DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, - RDMA1_SOUT_DSI2 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DSI2, - DISP_REG_CONFIG_DSIE_SEL_IN, DSI2_SEL_IN_MASK, - DSI2_SEL_IN_RDMA1 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DSI3, - DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, - RDMA1_SOUT_DSI3 - }, { - DDP_COMPONENT_RDMA1, DDP_COMPONENT_DSI3, - DISP_REG_CONFIG_DSIO_SEL_IN, DSI3_SEL_IN_MASK, - DSI3_SEL_IN_RDMA1 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DPI0, - DISP_REG_CONFIG_DISP_RDMA2_SOUT, RDMA2_SOUT_MASK, - RDMA2_SOUT_DPI0 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DPI0, - DISP_REG_CONFIG_DPI_SEL_IN, DPI0_SEL_IN_MASK, - DPI0_SEL_IN_RDMA2 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DPI1, - DISP_REG_CONFIG_DISP_RDMA2_SOUT, RDMA2_SOUT_MASK, - RDMA2_SOUT_DPI1 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DPI1, - DISP_REG_CONFIG_DPI_SEL_IN, DPI1_SEL_IN_MASK, - DPI1_SEL_IN_RDMA2 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DSI0, - DISP_REG_CONFIG_DSIE_SEL_IN, DSI0_SEL_IN_MASK, - DSI0_SEL_IN_RDMA2 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DSI1, - DISP_REG_CONFIG_DISP_RDMA2_SOUT, RDMA2_SOUT_MASK, - RDMA2_SOUT_DSI1 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DSI1, - DISP_REG_CONFIG_DSIO_SEL_IN, DSI1_SEL_IN_MASK, - DSI1_SEL_IN_RDMA2 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DSI2, - DISP_REG_CONFIG_DISP_RDMA2_SOUT, RDMA2_SOUT_MASK, - RDMA2_SOUT_DSI2 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DSI2, - DISP_REG_CONFIG_DSIE_SEL_IN, DSI2_SEL_IN_MASK, - DSI2_SEL_IN_RDMA2 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DSI3, - DISP_REG_CONFIG_DISP_RDMA2_SOUT, RDMA2_SOUT_MASK, - RDMA2_SOUT_DSI3 - }, { - DDP_COMPONENT_RDMA2, DDP_COMPONENT_DSI3, - DISP_REG_CONFIG_DSIO_SEL_IN, DSI3_SEL_IN_MASK, - DSI3_SEL_IN_RDMA2 - }, { - DDP_COMPONENT_UFOE, DDP_COMPONENT_DSI0, - DISP_REG_CONFIG_DISP_UFOE_MOUT_EN, UFOE_MOUT_EN_DSI0, - UFOE_MOUT_EN_DSI0 - } + MMSYS_ROUTE(BLS, 0, DSI, 0, + DISP_REG_CONFIG_OUT_SEL, BLS_RDMA1_DSI_DPI_MASK, + BLS_TO_DSI_RDMA1_TO_DPI1), + MMSYS_ROUTE(BLS, 0, DSI, 0, + DISP_REG_CONFIG_DSI_SEL, DSI_SEL_IN_MASK, + DSI_SEL_IN_BLS), + MMSYS_ROUTE(BLS, 0, DPI, 0, + DISP_REG_CONFIG_OUT_SEL, BLS_RDMA1_DSI_DPI_MASK, + BLS_TO_DPI_RDMA1_TO_DSI), + MMSYS_ROUTE(BLS, 0, DPI, 0, + DISP_REG_CONFIG_DSI_SEL, DSI_SEL_IN_MASK, + DSI_SEL_IN_RDMA), + MMSYS_ROUTE(BLS, 0, DPI, 0, + DISP_REG_CONFIG_DPI_SEL, DPI_SEL_IN_MASK, + DPI_SEL_IN_BLS), + MMSYS_ROUTE(GAMMA, 0, RDMA, 1, + DISP_REG_CONFIG_DISP_GAMMA_MOUT_EN, GAMMA_MOUT_EN_RDMA1, + GAMMA_MOUT_EN_RDMA1), + MMSYS_ROUTE(OD, 0, RDMA, 0, + DISP_REG_CONFIG_DISP_OD_MOUT_EN, OD_MOUT_EN_RDMA0, + OD_MOUT_EN_RDMA0), + MMSYS_ROUTE(OD, 1, RDMA, 1, + DISP_REG_CONFIG_DISP_OD_MOUT_EN, OD1_MOUT_EN_RDMA1, + OD1_MOUT_EN_RDMA1), + MMSYS_ROUTE(OVL, 0, COLOR, 0, + DISP_REG_CONFIG_DISP_OVL0_MOUT_EN, OVL0_MOUT_EN_COLOR0, + OVL0_MOUT_EN_COLOR0), + MMSYS_ROUTE(OVL, 0, COLOR, 0, + DISP_REG_CONFIG_DISP_COLOR0_SEL_IN, COLOR0_SEL_IN_OVL0, + COLOR0_SEL_IN_OVL0), + MMSYS_ROUTE(OVL, 0, RDMA, 0, + DISP_REG_CONFIG_DISP_OVL_MOUT_EN, OVL_MOUT_EN_RDMA, + OVL_MOUT_EN_RDMA), + MMSYS_ROUTE(OVL, 1, COLOR, 1, + DISP_REG_CONFIG_DISP_OVL1_MOUT_EN, OVL1_MOUT_EN_COLOR1, + OVL1_MOUT_EN_COLOR1), + MMSYS_ROUTE(OVL, 1, COLOR, 1, + DISP_REG_CONFIG_DISP_COLOR1_SEL_IN, COLOR1_SEL_IN_OVL1, + COLOR1_SEL_IN_OVL1), + MMSYS_ROUTE(RDMA, 0, DPI, 0, + DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, + RDMA0_SOUT_DPI0), + MMSYS_ROUTE(RDMA, 0, DPI, 1, + DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, + RDMA0_SOUT_DPI1), + MMSYS_ROUTE(RDMA, 0, DSI, 1, + DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, + RDMA0_SOUT_DSI1), + MMSYS_ROUTE(RDMA, 0, DSI, 2, + DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, + RDMA0_SOUT_DSI2), + MMSYS_ROUTE(RDMA, 0, DSI, 3, + DISP_REG_CONFIG_DISP_RDMA0_SOUT_EN, RDMA0_SOUT_MASK, + RDMA0_SOUT_DSI3), + MMSYS_ROUTE(RDMA, 1, DPI, 0, + DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, + RDMA1_SOUT_DPI0), + MMSYS_ROUTE(RDMA, 1, DPI, 0, + DISP_REG_CONFIG_DPI_SEL_IN, DPI0_SEL_IN_MASK, + DPI0_SEL_IN_RDMA1), + MMSYS_ROUTE(RDMA, 1, DPI, 1, + DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, + RDMA1_SOUT_DPI1), + MMSYS_ROUTE(RDMA, 1, DPI, 1, + DISP_REG_CONFIG_DPI_SEL_IN, DPI1_SEL_IN_MASK, + DPI1_SEL_IN_RDMA1), + MMSYS_ROUTE(RDMA, 1, DSI, 0, + DISP_REG_CONFIG_DSIE_SEL_IN, DSI0_SEL_IN_MASK, + DSI0_SEL_IN_RDMA1), + MMSYS_ROUTE(RDMA, 1, DSI, 1, + DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, + RDMA1_SOUT_DSI1), + MMSYS_ROUTE(RDMA, 1, DSI, 1, + DISP_REG_CONFIG_DSIO_SEL_IN, DSI1_SEL_IN_MASK, + DSI1_SEL_IN_RDMA1), + MMSYS_ROUTE(RDMA, 1, DSI, 2, + DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, + RDMA1_SOUT_DSI2), + MMSYS_ROUTE(RDMA, 1, DSI, 2, + DISP_REG_CONFIG_DSIE_SEL_IN, DSI2_SEL_IN_MASK, + DSI2_SEL_IN_RDMA1), + MMSYS_ROUTE(RDMA, 1, DSI, 3, + DISP_REG_CONFIG_DISP_RDMA1_SOUT_EN, RDMA1_SOUT_MASK, + RDMA1_SOUT_DSI3), + MMSYS_ROUTE(RDMA, 1, DSI, 3, + DISP_REG_CONFIG_DSIO_SEL_IN, DSI3_SEL_IN_MASK, + DSI3_SEL_IN_RDMA1), + MMSYS_ROUTE(RDMA, 2, DPI, 0, + DISP_REG_CONFIG_DISP_RDMA2_SOUT, RDMA2_SOUT_MASK, + RDMA2_SOUT_DPI0), + MMSYS_ROUTE(RDMA, 2, DPI, 0, + DISP_REG_CONFIG_DPI_SEL_IN, DPI0_SEL_IN_MASK, + DPI0_SEL_IN_RDMA2), + MMSYS_ROUTE(RDMA, 2, DPI, 1, + DISP_REG_CONFIG_DISP_RDMA2_SOUT, RDMA2_SOUT_MASK, + RDMA2_SOUT_DPI1), + MMSYS_ROUTE(RDMA, 2, DPI, 1, + DISP_REG_CONFIG_DPI_SEL_IN, DPI1_SEL_IN_MASK, + DPI1_SEL_IN_RDMA2), + MMSYS_ROUTE(RDMA, 2, DSI, 0, + DISP_REG_CONFIG_DSIE_SEL_IN, DSI0_SEL_IN_MASK, + DSI0_SEL_IN_RDMA2), + MMSYS_ROUTE(RDMA, 2, DSI, 1, + DISP_REG_CONFIG_DISP_RDMA2_SOUT, RDMA2_SOUT_MASK, + RDMA2_SOUT_DSI1), + MMSYS_ROUTE(RDMA, 2, DSI, 1, + DISP_REG_CONFIG_DSIO_SEL_IN, DSI1_SEL_IN_MASK, + DSI1_SEL_IN_RDMA2), + MMSYS_ROUTE(RDMA, 2, DSI, 2, + DISP_REG_CONFIG_DISP_RDMA2_SOUT, RDMA2_SOUT_MASK, + RDMA2_SOUT_DSI2), + MMSYS_ROUTE(RDMA, 2, DSI, 2, + DISP_REG_CONFIG_DSIE_SEL_IN, DSI2_SEL_IN_MASK, + DSI2_SEL_IN_RDMA2), }; #endif /* __SOC_MEDIATEK_MTK_MMSYS_H */ From 0ef07f74aff2db93450a01fea3296c32b991e72a Mon Sep 17 00:00:00 2001 From: Pengpeng Hou Date: Sat, 4 Jul 2026 20:43:08 +0800 Subject: [PATCH 007/857] soc: mediatek: add missing MODULE_DEVICE_TABLE() The driver has an OF match table wired to .of_match_table, but does not export the table with MODULE_DEVICE_TABLE(). Add the missing MODULE_DEVICE_TABLE(of, ...) entry so module alias information is generated for OF based module autoloading. This is a source-level fix. It does not claim dynamic hardware reproduction; the evidence is the driver-owned match table, its use by the platform driver, and the missing module alias publication. Signed-off-by: Pengpeng Hou Signed-off-by: AngeloGioacchino Del Regno --- drivers/soc/mediatek/mtk-dvfsrc.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/soc/mediatek/mtk-dvfsrc.c b/drivers/soc/mediatek/mtk-dvfsrc.c index 548a28f5024262..e8cdee8aed353c 100644 --- a/drivers/soc/mediatek/mtk-dvfsrc.c +++ b/drivers/soc/mediatek/mtk-dvfsrc.c @@ -862,6 +862,7 @@ static const struct of_device_id mtk_dvfsrc_of_match[] = { { .compatible = "mediatek,mt8196-dvfsrc", .data = &mt8196_data }, { /* sentinel */ } }; +MODULE_DEVICE_TABLE(of, mtk_dvfsrc_of_match); static struct platform_driver mtk_dvfsrc_driver = { .probe = mtk_dvfsrc_probe, From 5ad59d4b626343901cf52078f118baa960e89868 Mon Sep 17 00:00:00 2001 From: Alexandre Belloni Date: Mon, 13 Jul 2026 16:23:52 +0200 Subject: [PATCH 008/857] soc: document merges Signed-off-by: Alexandre Belloni --- arch/arm/arm-soc-for-next-contents.txt | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 30c2012645df57..e50a879a7c4053 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -1,9 +1,30 @@ soc/arm + patch + ARM: use CONFIG_AEABI by default everywhere + ARM: limit OABI support to StrongARM CPUs + ARM: rework ARM11 CPU selection logic + ARM: deprecate support for ARM1136r0 + ARM: turn CONFIG_ATAGS off by default + ARM: mark CPU_ENDIAN_BE8 as deprecated + ARM: update DEPRECATED_PARAM_STRUCT removal timeline + ARM: s3c64xx: extend deprecation schedule + ARM: update FPE_NWFPE help text + ARM: mark IWMMXT as deprecated + ARM: mark ARCH_DOVE as deprecated + ARM: PXA: mark remaining board files as deprecated + ARM: orion5x: mark all board files as deprecated + ARM: mark mach-sa1100 as deprecated + ARM: mark RiscPC as deprecated + ARM: mark footbridge as deprecated + ARM: mark Cortex-M3/M4/M7 based boards as deprecated + ARM: mark axxia platform as deprecated + ARM: mark mv78xx0 support as deprecated soc/dt patch ARM: dts: st: spear: Correct indentation ARM: dts: st: ste: Correct indentation + ARM: dts: st: spear13xx: Drop unused/incorrect usbh0_id and usbh1_id soc/drivers From 900d72f1052c1028d6e8c1f182529a403ba50a72 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Wed, 15 Jul 2026 23:00:45 +0200 Subject: [PATCH 009/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 26 ++++++-------------------- 1 file changed, 6 insertions(+), 20 deletions(-) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index e50a879a7c4053..bf41208df00136 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -1,24 +1,6 @@ soc/arm - patch - ARM: use CONFIG_AEABI by default everywhere - ARM: limit OABI support to StrongARM CPUs - ARM: rework ARM11 CPU selection logic - ARM: deprecate support for ARM1136r0 - ARM: turn CONFIG_ATAGS off by default - ARM: mark CPU_ENDIAN_BE8 as deprecated - ARM: update DEPRECATED_PARAM_STRUCT removal timeline - ARM: s3c64xx: extend deprecation schedule - ARM: update FPE_NWFPE help text - ARM: mark IWMMXT as deprecated - ARM: mark ARCH_DOVE as deprecated - ARM: PXA: mark remaining board files as deprecated - ARM: orion5x: mark all board files as deprecated - ARM: mark mach-sa1100 as deprecated - ARM: mark RiscPC as deprecated - ARM: mark footbridge as deprecated - ARM: mark Cortex-M3/M4/M7 based boards as deprecated - ARM: mark axxia platform as deprecated - ARM: mark mv78xx0 support as deprecated + arm/deprecation + https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc tags/arm-feature-deprecation-for-7.3 soc/dt patch @@ -47,4 +29,8 @@ arm/fixes ARM: Don't let ARMv5 platforms select USE_OF (314c243b201b678fa89226b1eaea51a71340454e) git://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee tags/tee-update-for-v7.2 + patch + MAINTAINERS: Update maintainer and git tree for CIX SoC + (980a8bfe7baec9b9ee0d5443b0b204552e41c407) + https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux tags/scmi-ffa-fixes-7.2 From 11f8710f99a18b13bd1868202813f4d5dad7cbcf Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Fri, 17 Jul 2026 16:13:23 +0200 Subject: [PATCH 010/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index bf41208df00136..f64cbc376259fb 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -1,6 +1,8 @@ soc/arm arm/deprecation https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc tags/arm-feature-deprecation-for-7.3 + patch + ARM: replace linux/gpio.h inclusions soc/dt patch @@ -33,4 +35,6 @@ arm/fixes MAINTAINERS: Update maintainer and git tree for CIX SoC (980a8bfe7baec9b9ee0d5443b0b204552e41c407) https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux tags/scmi-ffa-fixes-7.2 + (6fa6ee724d8dadf392139e242ac936b5da730c4b) + https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-fixes-for-v7.2-tag1 From b4f1115ec568160c0ac859c3464b5319afee785f Mon Sep 17 00:00:00 2001 From: Krzysztof Kozlowski Date: Mon, 6 Jul 2026 12:19:21 +0200 Subject: [PATCH 011/857] ARM: dts: broadcom: bcm7445: Correct indentation Correct spaces or mix of tabs+spaces into proper tab-indented lines. No functional impact (same DTB). Signed-off-by: Krzysztof Kozlowski Link: https://lore.kernel.org/r/20260706101920.341586-2-krzysztof.kozlowski@oss.qualcomm.com Signed-off-by: Florian Fainelli --- arch/arm/boot/dts/broadcom/bcm7445.dtsi | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm/boot/dts/broadcom/bcm7445.dtsi b/arch/arm/boot/dts/broadcom/bcm7445.dtsi index c6307c7437e3bf..7488781e0301f0 100644 --- a/arch/arm/boot/dts/broadcom/bcm7445.dtsi +++ b/arch/arm/boot/dts/broadcom/bcm7445.dtsi @@ -132,7 +132,7 @@ interrupt-names = "hif"; }; - aon_pm_l2_intc: interrupt-controller@410640 { + aon_pm_l2_intc: interrupt-controller@410640 { compatible = "brcm,l2-intc"; reg = <0x410640 0x30>; interrupt-controller; From dacbe5d56fd3707316b44153eb15a9fd149d56f9 Mon Sep 17 00:00:00 2001 From: Stanislave Kinsburskii Date: Wed, 22 Jul 2026 23:56:46 +0000 Subject: [PATCH 012/857] mshv: Use kfree_rcu in mshv_portid_free mshv_portid_free() uses synchronize_rcu() followed by kfree() to reclaim port table entries. This blocks the caller until a full RCU grace period elapses, which is unnecessary since the same module already uses the non-blocking kfree_rcu() pattern in mshv_port_table_fini(). Replace with kfree_rcu() to avoid the blocking wait and keep the reclamation strategy consistent across the file. Signed-off-by: Stanislav Kinsburskii Reviewed-by: Anirudh Rayabharam (Microsoft) Signed-off-by: Wei Liu --- drivers/hv/mshv_portid_table.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/drivers/hv/mshv_portid_table.c b/drivers/hv/mshv_portid_table.c index 6f59b3e3762473..0d632507ed2144 100644 --- a/drivers/hv/mshv_portid_table.c +++ b/drivers/hv/mshv_portid_table.c @@ -62,8 +62,7 @@ mshv_portid_free(int port_id) WARN_ON(!info); idr_unlock(&port_table_idr); - synchronize_rcu(); - kfree(info); + kfree_rcu(info, portbl_rcu); } int From a4bc97d5bd01d58912270468b3ad1fcfd57d241c Mon Sep 17 00:00:00 2001 From: Stanislav Kinsburskii Date: Thu, 7 May 2026 15:43:15 +0000 Subject: [PATCH 013/857] mshv: Fix race in mshv_irqfd_deassign mshv_irqfd_deactivate() and the hlist traversal of pt_irqfds_list require pt->pt_irqfds_lock to be held, but mshv_irqfd_deassign() omits it. This races with the EPOLLHUP path in mshv_irqfd_wakeup(), which does take the lock before calling mshv_irqfd_deactivate(). Additionally, mshv_irqfd_deactivate() uses hlist_del() which poisons the node pointers rather than resetting them. Since mshv_irqfd_is_active() relies on hlist_unhashed() (checks pprev == NULL), a poisoned node still appears active. If a concurrent path calls mshv_irqfd_deactivate() again on the same irqfd, the guard fails to prevent a double hlist_del() on poisoned pointers. Fix both issues: - Add the missing spin_lock_irq/spin_unlock_irq around the list traversal in mshv_irqfd_deassign(), matching mshv_irqfd_release(). - Use hlist_del_init() instead of hlist_del() so the node is properly marked as unhashed after removal, making the is_active guard reliable. Fixes: 621191d709b14 ("Drivers: hv: Introduce mshv_root module to expose /dev/mshv to VMMs") Signed-off-by: Stanislav Kinsburskii Reviewed-by: Anirudh Rayabharam (Microsoft) Signed-off-by: Wei Liu --- drivers/hv/mshv_eventfd.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/drivers/hv/mshv_eventfd.c b/drivers/hv/mshv_eventfd.c index 90959f639dc329..5995a62aff8d8d 100644 --- a/drivers/hv/mshv_eventfd.c +++ b/drivers/hv/mshv_eventfd.c @@ -284,7 +284,7 @@ static void mshv_irqfd_deactivate(struct mshv_irqfd *irqfd) if (!mshv_irqfd_is_active(irqfd)) return; - hlist_del(&irqfd->irqfd_hnode); + hlist_del_init(&irqfd->irqfd_hnode); queue_work(irqfd_cleanup_wq, &irqfd->irqfd_shutdown); } @@ -541,13 +541,14 @@ static int mshv_irqfd_deassign(struct mshv_partition *pt, if (IS_ERR(eventfd)) return PTR_ERR(eventfd); + spin_lock_irq(&pt->pt_irqfds_lock); hlist_for_each_entry_safe(irqfd, n, &pt->pt_irqfds_list, irqfd_hnode) { if (irqfd->irqfd_eventfd_ctx == eventfd && irqfd->irqfd_irqnum == args->gsi) - mshv_irqfd_deactivate(irqfd); } + spin_unlock_irq(&pt->pt_irqfds_lock); eventfd_ctx_put(eventfd); From ced1cad146f0d893920d2cf39e08200a08333920 Mon Sep 17 00:00:00 2001 From: Stanislav Kinsburskii Date: Thu, 7 May 2026 15:43:43 +0000 Subject: [PATCH 014/857] mshv: Fix level-triggered check on uninitialized data MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit In mshv_irqfd_assign(), the level-triggered validation for resample irqfds checks irqfd_lapic_irq.lapic_control.level_triggered before mshv_irqfd_update() has populated the field. Since the irqfd struct is zero-allocated, level_triggered is always 0 at that point, causing the check to always reject resample irqfds with -EINVAL. This makes level-triggered interrupt resampling — used to avoid interrupt storms with assigned devices — completely non-functional. Move the check after the mshv_irqfd_update() call, which resolves the IRQ routing entry and populates irqfd_lapic_irq with the actual trigger mode. Fixes: 621191d709b14 ("Drivers: hv: Introduce mshv_root module to expose /dev/mshv to VMMs") Signed-off-by: Stanislav Kinsburskii Reviewed-by: Anirudh Rayabharam (Microsoft) Signed-off-by: Wei Liu --- drivers/hv/mshv_eventfd.c | 25 ++++++++++++++----------- 1 file changed, 14 insertions(+), 11 deletions(-) diff --git a/drivers/hv/mshv_eventfd.c b/drivers/hv/mshv_eventfd.c index 5995a62aff8d8d..047e5bd432381e 100644 --- a/drivers/hv/mshv_eventfd.c +++ b/drivers/hv/mshv_eventfd.c @@ -473,6 +473,19 @@ static int mshv_irqfd_assign(struct mshv_partition *pt, init_poll_funcptr(&irqfd->irqfd_polltbl, mshv_irqfd_queue_proc); spin_lock_irq(&pt->pt_irqfds_lock); + ret = 0; + hlist_for_each_entry(tmp, &pt->pt_irqfds_list, irqfd_hnode) { + if (irqfd->irqfd_eventfd_ctx != tmp->irqfd_eventfd_ctx) + continue; + /* This fd is used for another irq already. */ + ret = -EBUSY; + spin_unlock_irq(&pt->pt_irqfds_lock); + goto fail; + } + + idx = srcu_read_lock(&pt->pt_irq_srcu); + mshv_irqfd_update(pt, irqfd); + #if IS_ENABLED(CONFIG_X86) if (args->flags & BIT(MSHV_IRQFD_BIT_RESAMPLE) && !irqfd->irqfd_lapic_irq.lapic_control.level_triggered) { @@ -481,22 +494,12 @@ static int mshv_irqfd_assign(struct mshv_partition *pt, * Otherwise return with failure */ spin_unlock_irq(&pt->pt_irqfds_lock); + srcu_read_unlock(&pt->pt_irq_srcu, idx); ret = -EINVAL; goto fail; } #endif - ret = 0; - hlist_for_each_entry(tmp, &pt->pt_irqfds_list, irqfd_hnode) { - if (irqfd->irqfd_eventfd_ctx != tmp->irqfd_eventfd_ctx) - continue; - /* This fd is used for another irq already. */ - ret = -EBUSY; - spin_unlock_irq(&pt->pt_irqfds_lock); - goto fail; - } - idx = srcu_read_lock(&pt->pt_irq_srcu); - mshv_irqfd_update(pt, irqfd); hlist_add_head(&irqfd->irqfd_hnode, &pt->pt_irqfds_list); spin_unlock_irq(&pt->pt_irqfds_lock); From ac092ae422d67ed071f216c9fa10df8006e51019 Mon Sep 17 00:00:00 2001 From: Stanislav Kinsburskii Date: Thu, 7 May 2026 15:44:37 +0000 Subject: [PATCH 015/857] mshv: Fix missing error code on VP allocation failure In mshv_partition_ioctl_create_vp(), when kzalloc for the VP struct fails, the code jumps to the cleanup path without setting ret. At that point ret is 0 from the preceding successful mshv_vp_stats_map() call, so the function returns success to userspace despite having failed to create the VP. No fd is installed and no VP is registered in pt_vp_array, but userspace has no way to know the operation failed. Set ret to -ENOMEM before jumping to the cleanup path. Fixes: 621191d709b14 ("Drivers: hv: Introduce mshv_root module to expose /dev/mshv to VMMs") Signed-off-by: Stanislav Kinsburskii Reviewed-by: Anirudh Rayabharam (Microsoft) Signed-off-by: Wei Liu --- drivers/hv/mshv_root_main.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/drivers/hv/mshv_root_main.c b/drivers/hv/mshv_root_main.c index 146726cc4e9ba8..644f9b10cbba41 100644 --- a/drivers/hv/mshv_root_main.c +++ b/drivers/hv/mshv_root_main.c @@ -1117,8 +1117,10 @@ mshv_partition_ioctl_create_vp(struct mshv_partition *partition, goto unmap_ghcb_page; vp = kzalloc_obj(*vp); - if (!vp) + if (!vp) { + ret = -ENOMEM; goto unmap_stats_pages; + } vp->vp_partition = mshv_partition_get(partition); if (!vp->vp_partition) { From 3522b14e26d3daae671c937650b784fd9dea67a3 Mon Sep 17 00:00:00 2001 From: Stanislave Kinsburskii Date: Thu, 23 Jul 2026 00:28:11 +0000 Subject: [PATCH 016/857] mshv: Order pt_vp_array publish against irqfd assertion path mshv_partition_ioctl_create_vp() initialises a VP struct (allocations, mutex_init, init_waitqueue_head, page mappings) and then publishes the pointer into partition->pt_vp_array. Several ISR paths read this array locklessly: the intercept ISR, the two scheduler ISRs, and mshv_try_assert_irq_fast() on the irqfd fast path. Of these, only mshv_try_assert_irq_fast() can structurally race the publish. It runs from an eventfd waker without holding pt_mutex, and MSHV_IRQFD does not require the target lapic_apic_id (== vp_index) to refer to an existing VP at registration time. A user can therefore register an irqfd targeting a yet-to-be-created VP, then trigger mshv_try_assert_irq_fast() concurrently with MSHV_CREATE_VP for the same index. On weakly-ordered architectures the reader can observe a non-NULL pointer in pt_vp_array before the initialising stores to the VP struct become visible, leading to use of partially-initialised fields (e.g. vp_register_page). The other ISR readers cannot reach this race: the hypervisor will not generate intercept or scheduler messages for a VP that has never been told to run, and the user can only call MSHV_RUN_VP on the VP fd returned by MSHV_CREATE_VP, which by construction is returned after the publish. Leave those readers as plain loads. Use smp_store_release() in mshv_partition_ioctl_create_vp() to publish the pointer, and pair it with smp_load_acquire() in mshv_try_assert_irq_fast(). On x86 these compile to plain accesses under TSO; on ARM64 they emit one-instruction acquire/release barriers, acceptable on this fast path. The destroy-side path (destroy_partition() clearing pt_vp_array[i] to NULL after kfree(vp)) has a separate ordering and lifetime concern that is out of scope here. Fixes: 621191d709b14 ("Drivers: hv: Introduce mshv_root module to expose /dev/mshv to VMMs") Signed-off-by: Stanislav Kinsburskii Reviewed-by: Anirudh Rayabharam (Microsoft) Signed-off-by: Wei Liu --- drivers/hv/mshv_eventfd.c | 9 ++++++++- drivers/hv/mshv_root_main.c | 8 +++++++- 2 files changed, 15 insertions(+), 2 deletions(-) diff --git a/drivers/hv/mshv_eventfd.c b/drivers/hv/mshv_eventfd.c index 047e5bd432381e..06aef99c8298d6 100644 --- a/drivers/hv/mshv_eventfd.c +++ b/drivers/hv/mshv_eventfd.c @@ -169,7 +169,14 @@ static int mshv_try_assert_irq_fast(struct mshv_irqfd *irqfd) return -EOPNOTSUPP; #endif - vp = partition->pt_vp_array[irq->lapic_apic_id]; + /* + * Pairs with smp_store_release() in mshv_partition_ioctl_create_vp(). + * MSHV_IRQFD does not require the target lapic_apic_id to refer to an + * existing VP, so this read can race a concurrent VP creation; the + * acquire ensures that a non-NULL pointer implies the VP's + * initialising stores are visible. + */ + vp = smp_load_acquire(&partition->pt_vp_array[irq->lapic_apic_id]); if (!vp->vp_register_page) return -EOPNOTSUPP; diff --git a/drivers/hv/mshv_root_main.c b/drivers/hv/mshv_root_main.c index 644f9b10cbba41..8a15448e2ace98 100644 --- a/drivers/hv/mshv_root_main.c +++ b/drivers/hv/mshv_root_main.c @@ -1157,7 +1157,13 @@ mshv_partition_ioctl_create_vp(struct mshv_partition *partition, /* already exclusive with the partition mutex for all ioctls */ partition->pt_vp_count++; - partition->pt_vp_array[args.vp_index] = vp; + /* + * Pairs with smp_load_acquire() in mshv_try_assert_irq_fast(), which + * can run concurrently from an irqfd waker without holding pt_mutex. + * The release ensures the VP's initialising stores are visible to any + * reader that observes a non-NULL pointer in pt_vp_array. + */ + smp_store_release(&partition->pt_vp_array[args.vp_index], vp); goto out; From 2225dc7722999c4746dddc1efd7ac0282975fd90 Mon Sep 17 00:00:00 2001 From: Hardik Garg Date: Fri, 17 Jul 2026 00:18:37 +0000 Subject: [PATCH 017/857] Drivers: hv: vmbus: add VTL2 redirect connection ID VMBus sends CHANNELMSG_INITIATE_CONTACT through a Hyper-V message connection ID. Older protocol versions use VMBUS_MESSAGE_CONNECTION_ID, while protocol version 5.0 and newer normally use VMBUS_MESSAGE_CONNECTION_ID_4. For a VTL2 kernel using VMBus protocol 5.0 or newer, the host may expect INITIATE_CONTACT on either the redirect connection ID or VMBUS_MESSAGE_CONNECTION_ID_4. There is no capability indication that identifies which ID is active, so the driver must determine it at runtime. During VMBus negotiation, the redirect ID is tried first because it is used by VTL2 configurations with VMBus redirection enabled. If the redirect ID is unavailable, the host rejects it synchronously with HV_STATUS_INVALID_CONNECTION_ID, allowing fallback to the standard ID. Return a distinct error for an invalid Initiate Contact connection ID so this fallback does not mask other post-message failures or protocol-version rejections. Preserve the existing connection ID selection for older protocol versions or when running below VTL2. Signed-off-by: Hardik Garg Reviewed-by: Tianyu Lan Reviewed-by: Saurabh Sengar Reviewed-by: Naman Jain Reviewed-by: Michael Kelley Signed-off-by: Wei Liu --- drivers/hv/connection.c | 47 +++++++++++++++++++++++---------------- drivers/hv/hyperv_vmbus.h | 2 ++ 2 files changed, 30 insertions(+), 19 deletions(-) diff --git a/drivers/hv/connection.c b/drivers/hv/connection.c index b5b322ce16df44..0fd50d4cb57398 100644 --- a/drivers/hv/connection.c +++ b/drivers/hv/connection.c @@ -72,7 +72,8 @@ module_param(max_version, uint, S_IRUGO); MODULE_PARM_DESC(max_version, "Maximal VMBus protocol version which can be negotiated"); -int vmbus_negotiate_version(struct vmbus_channel_msginfo *msginfo, u32 version) +static int vmbus_try_connection_id(struct vmbus_channel_msginfo *msginfo, + u32 version, u32 connection_id) { int ret = 0; struct vmbus_channel_initiate_contact *msg; @@ -87,20 +88,20 @@ int vmbus_negotiate_version(struct vmbus_channel_msginfo *msginfo, u32 version) msg->vmbus_version_requested = version; /* - * VMBus protocol 5.0 (VERSION_WIN10_V5) and higher require that we must - * use VMBUS_MESSAGE_CONNECTION_ID_4 for the Initiate Contact Message, - * and for subsequent messages, we must use the Message Connection ID - * field in the host-returned Version Response Message. And, with - * VERSION_WIN10_V5 and higher, we don't use msg->interrupt_page, but we - * tell the host explicitly that we still use VMBUS_MESSAGE_SINT(2) for - * compatibility. + * For VMBus protocol 5.0 (VERSION_WIN10_V5) and higher, use the + * caller-supplied connection_id for the Initiate Contact message so + * the caller can implement the required retry scheme. For subsequent + * messages, use the Message Connection ID field in the host-returned + * Version Response message. With VERSION_WIN10_V5 and higher, we don't + * use msg->interrupt_page, but tell the host explicitly that we still + * use VMBUS_MESSAGE_SINT(2) for compatibility. * * On old hosts, we should always use VMBUS_MESSAGE_CONNECTION_ID (1). */ if (version >= VERSION_WIN10_V5) { msg->msg_sint = VMBUS_MESSAGE_SINT; msg->msg_vtl = ms_hyperv.vtl; - vmbus_connection.msg_conn_id = VMBUS_MESSAGE_CONNECTION_ID_4; + vmbus_connection.msg_conn_id = connection_id; } else { msg->interrupt_page = virt_to_phys(vmbus_connection.int_page); vmbus_connection.msg_conn_id = VMBUS_MESSAGE_CONNECTION_ID; @@ -165,6 +166,22 @@ int vmbus_negotiate_version(struct vmbus_channel_msginfo *msginfo, u32 version) return ret; } +int vmbus_negotiate_version(struct vmbus_channel_msginfo *msginfo, u32 version) +{ + int ret; + + /* Try the redirect ID first for VTL2 with VMBus protocol 5.0+. */ + if (version >= VERSION_WIN10_V5 && ms_hyperv.vtl == 2) { + ret = vmbus_try_connection_id(msginfo, version, + VMBUS_MESSAGE_CONNECTION_ID_REDIRECT); + if (ret != -ENXIO) + return ret; + } + + return vmbus_try_connection_id(msginfo, version, + VMBUS_MESSAGE_CONNECTION_ID_4); +} + /* * vmbus_connect - Sends a connect request on the partition service connection */ @@ -457,18 +474,10 @@ int vmbus_post_msg(void *buffer, size_t buflen, bool can_sleep) switch (ret) { case HV_STATUS_INVALID_CONNECTION_ID: - /* - * See vmbus_negotiate_version(): VMBus protocol 5.0 - * and higher require that we must use - * VMBUS_MESSAGE_CONNECTION_ID_4 for the Initiate - * Contact message, but on old hosts that only - * support VMBus protocol 4.0 or lower, here we get - * HV_STATUS_INVALID_CONNECTION_ID and we should - * return an error immediately without retrying. - */ + /* Allow INITIATE_CONTACT to try another connection ID. */ hdr = buffer; if (hdr->msgtype == CHANNELMSG_INITIATE_CONTACT) - return -EINVAL; + return -ENXIO; /* * We could get this if we send messages too * frequently. diff --git a/drivers/hv/hyperv_vmbus.h b/drivers/hv/hyperv_vmbus.h index eb8bdd8bb1f582..33923621a5a3f2 100644 --- a/drivers/hv/hyperv_vmbus.h +++ b/drivers/hv/hyperv_vmbus.h @@ -110,6 +110,8 @@ struct hv_input_post_message { enum { VMBUS_MESSAGE_CONNECTION_ID = 1, VMBUS_MESSAGE_CONNECTION_ID_4 = 4, + /* VTL2 redirect connection ID for INITIATE_CONTACT. */ + VMBUS_MESSAGE_CONNECTION_ID_REDIRECT = 0x800074, VMBUS_MESSAGE_PORT_ID = 1, VMBUS_EVENT_CONNECTION_ID = 2, VMBUS_EVENT_PORT_ID = 2, From b7e92e14744131ca00720837dcdff944fa54fa10 Mon Sep 17 00:00:00 2001 From: Stanislav Kinsburskii Date: Thu, 7 May 2026 15:44:32 +0000 Subject: [PATCH 018/857] mshv: Publish VP to pt_vp_array before installing the file descriptor MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit mshv_partition_ioctl_create_vp() called anon_inode_getfd() before publishing the new VP into partition->pt_vp_array. anon_inode_getfd() includes fd_install(), so the fd was live in current->files before the publish ran. A concurrent MSHV_RUN_VP ioctl on that fd does not serialise against the in-progress MSHV_CREATE_VP — it takes vp->vp_mutex, not the partition mutex. Once the VP starts running and traps, mshv_intercept_isr() can look up partition->pt_vp_array[vp_index] and observe NULL, silently dropping the intercept message. Split the fd creation: reserve an fd with get_unused_fd_flags(), create the file with anon_inode_getfile(), publish the VP via smp_store_release(), and finally call fd_install() as the userspace-visibility commit point. Fixes: 621191d709b14 ("Drivers: hv: Introduce mshv_root module to expose /dev/mshv to VMMs") Signed-off-by: Stanislav Kinsburskii Reviewed-by: Anirudh Rayabharam (Microsoft) Signed-off-by: Wei Liu --- drivers/hv/mshv_root_main.c | 29 ++++++++++++++++++++++------- 1 file changed, 22 insertions(+), 7 deletions(-) diff --git a/drivers/hv/mshv_root_main.c b/drivers/hv/mshv_root_main.c index 8a15448e2ace98..cc2cfce2aefdb5 100644 --- a/drivers/hv/mshv_root_main.c +++ b/drivers/hv/mshv_root_main.c @@ -1072,6 +1072,8 @@ mshv_partition_ioctl_create_vp(struct mshv_partition *partition, struct mshv_vp *vp; struct page *intercept_msg_page, *register_page, *ghcb_page; struct hv_stats_page *stats_pages[2]; + struct file *file; + int fd; long ret; if (copy_from_user(&args, arg, sizeof(args))) @@ -1146,14 +1148,18 @@ mshv_partition_ioctl_create_vp(struct mshv_partition *partition, if (ret) goto put_partition; - /* - * Keep anon_inode_getfd last: it installs fd in the file struct and - * thus makes the state accessible in user space. - */ - ret = anon_inode_getfd("mshv_vp", &mshv_vp_fops, vp, - O_RDWR | O_CLOEXEC); - if (ret < 0) + fd = get_unused_fd_flags(O_RDWR | O_CLOEXEC); + if (fd < 0) { + ret = fd; goto remove_debugfs_vp; + } + + file = anon_inode_getfile("mshv_vp", &mshv_vp_fops, vp, + O_RDWR | O_CLOEXEC); + if (IS_ERR(file)) { + ret = PTR_ERR(file); + goto put_unused_vp_fd; + } /* already exclusive with the partition mutex for all ioctls */ partition->pt_vp_count++; @@ -1165,8 +1171,17 @@ mshv_partition_ioctl_create_vp(struct mshv_partition *partition, */ smp_store_release(&partition->pt_vp_array[args.vp_index], vp); + /* + * fd_install() is the userspace-visibility commit point. Must be the + * last operation that can fail or be observed. + */ + fd_install(fd, file); + ret = fd; + goto out; +put_unused_vp_fd: + put_unused_fd(fd); remove_debugfs_vp: mshv_debugfs_vp_remove(vp); put_partition: From 0fd49f7bfb8f1952ae157344ea9712da74734200 Mon Sep 17 00:00:00 2001 From: Yi Xie Date: Thu, 9 Jul 2026 10:19:47 +0800 Subject: [PATCH 019/857] mshv_vtl: bounds-check cpu index in vtl mmap fault handler cpu is taken from pgoff & 0xffff. cpu_online() does not reject cpu >= nr_cpu_ids, and per_cpu_ptr() can then walk off __per_cpu_offset. Signed-off-by: Yi Xie Reviewed-by: Naman Jain Signed-off-by: Wei Liu --- drivers/hv/mshv_vtl_main.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/hv/mshv_vtl_main.c b/drivers/hv/mshv_vtl_main.c index 5ba1efb3b4e7ea..6e3c11c681717b 100644 --- a/drivers/hv/mshv_vtl_main.c +++ b/drivers/hv/mshv_vtl_main.c @@ -802,7 +802,7 @@ static vm_fault_t mshv_vtl_fault(struct vm_fault *vmf) int cpu = vmf->pgoff & MSHV_PG_OFF_CPU_MASK; int real_off = vmf->pgoff >> MSHV_REAL_OFF_SHIFT; - if (!cpu_online(cpu)) + if (cpu >= nr_cpu_ids || !cpu_online(cpu)) return VM_FAULT_SIGBUS; /* * CPU Hotplug is not supported in VTL2 in OpenHCL, where this kernel driver exists. From 3edcea7f0413e84863300998ad3838eb87e6fca7 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Fri, 24 Jul 2026 18:29:48 +0200 Subject: [PATCH 020/857] Merge branch 'arm/fixes' into for-next * arm/fixes: (468 commits) arm64: dts: broadcom: bcm2712: Remove non-functional EL2 virtual timer Linux 7.2-rc4 Revert "drm/amd/display: Restore 5s vbl offdelay for NV3x+ DGPUs" drm/amd/display: check GRPH_FLIP status before sending event drm/amd/display: consolidate DCN vblank/flip handling onto vupdate_no_lock drm/amd: Create a device link between APU display and XHCI devices drm/amd/display: wire DCN42B mcache programming callback drm/amd/display: set new_stream to NULL after release drm/amd/display: Force PWM backlight on Lenovo Legion 5 15ARH05 drm/amdkfd: free MQD managers on DQM init failures drm/amdgpu/ttm: Consider concurrent VM flushes for buffer entities drm/amd/pm/smu7: Fix AC/DC switch notification drm/amdgpu: Disable PCIe dynamic speed switching on Ryzen Pinnacle Ridge drm/amdgpu: always emit the job vm fence drm/amd/pm/si: Fix AC/DC switch notification drm/amd/pm/si: Don't schedule thermal work when queue isn't initialized drm/amd/display: dce100: skip non-DP stream encoders for DP MST drm/amd/display: Set native cursor mode for disabled CRTCs drm/amd/pm/ci: Don't disable MCLK DPM on Bonaire 0x6658 (R7 260X) drm/amd/display: fix __udivdi3 link error ... From 2e00a93fbb4f7f2f713ae4900bd108f4f7413ff8 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Fri, 24 Jul 2026 18:31:47 +0200 Subject: [PATCH 021/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 39 +++++++++++++------------- 1 file changed, 19 insertions(+), 20 deletions(-) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index f64cbc376259fb..a96bc8edf487fa 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -3,38 +3,37 @@ soc/arm https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc tags/arm-feature-deprecation-for-7.3 patch ARM: replace linux/gpio.h inclusions + renesas/soc + https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-arm-soc-for-v7.3-tag1 + ixp4xx/arm + https://git.kernel.org/pub/scm/linux/kernel/git/linusw/linux-integrator tags/ixp4xx-arm-v7.3 soc/dt patch ARM: dts: st: spear: Correct indentation ARM: dts: st: ste: Correct indentation ARM: dts: st: spear13xx: Drop unused/incorrect usbh0_id and usbh1_id + ARM: dts: ixp4xx: Drop the reg-offset hack + canaan/dt + https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux tags/riscv-dt-for-v7.3-early-k230 soc/drivers + ixp4xx/soc-drivers + https://git.kernel.org/pub/scm/linux/kernel/git/linusw/linux-integrator tags/ixp4xx-soc-v7.3 + firmware/scmi + https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux tags/scmi-updates-7.3 + rockchip/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip tags/v7.3-rockchip-drivers1 + renesas/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-drivers-for-v7.3-tag1 soc/defconfig + riscv/defconfig + https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux tags/riscv-config-for-v7.3 soc/late arm/fixes - patch - MAINTAINERS: Update SpacemiT SoC git tree repository - (71827776667f4e4677a4fa806bcfb24d4b8dd9d7) - git://git.kernel.org/pub/scm/linux/kernel/git/pza/linux tags/reset-fixes-for-v7.2 - (813e034925814858cc52e7de321ec4848314e15d) - git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.2-pmc-fixes - (265d5d4032c5f6eb089a6e6241d37fdbde7da180) - git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.2-soc-fixes - (806a66f926c2b6652aeb88983d01f25081b41a73) - git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.2-arm64-dt-fixes - patch - ARM: Don't let ARMv5 platforms select USE_OF - (314c243b201b678fa89226b1eaea51a71340454e) - git://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee tags/tee-update-for-v7.2 - patch - MAINTAINERS: Update maintainer and git tree for CIX SoC - (980a8bfe7baec9b9ee0d5443b0b204552e41c407) - https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux tags/scmi-ffa-fixes-7.2 - (6fa6ee724d8dadf392139e242ac936b5da730c4b) - https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-fixes-for-v7.2-tag1 + broadcom/fixes + https://github.com/Broadcom/stblinux tags/arm-soc/for-7.2/devicetree-arm64-fixes From 632e1109192b920a3e94f844d926e9c2cc960998 Mon Sep 17 00:00:00 2001 From: "Rob Herring (Arm)" Date: Fri, 12 Jun 2026 16:52:51 -0500 Subject: [PATCH 022/857] clk: at91: Read "reg" with helper The "reg" property is an address-sized DT cell property. The AT91 compat clock parser only uses a small bus id from it, but reading it with the u8 helper does not match the property encoding. Use of_property_read_reg() so the code goes through the helper for "reg" properties, then keep the existing range check before passing the bus id to the clock registration code. Assisted-by: Codex:gpt-5-5 Signed-off-by: Rob Herring (Arm) Reviewed-by: Brian Masney Link: https://patch.msgid.link/20260612215251.1888345-1-robh@kernel.org Signed-off-by: Claudiu Beznea --- drivers/clk/at91/dt-compat.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/drivers/clk/at91/dt-compat.c b/drivers/clk/at91/dt-compat.c index 5d543e80784311..dad26dd5e71df9 100644 --- a/drivers/clk/at91/dt-compat.c +++ b/drivers/clk/at91/dt-compat.c @@ -2,6 +2,7 @@ #include #include #include +#include #include #include #include @@ -217,7 +218,7 @@ CLK_OF_DECLARE(of_sama5d4_clk_h32mx_setup, "atmel,sama5d4-clk-h32mx", static void __init of_sama5d2_clk_i2s_mux_setup(struct device_node *np) { struct regmap *regmap_sfr; - u8 bus_id; + u64 bus_id; const char *parent_names[2]; struct device_node *i2s_mux_np; struct clk_hw *hw; @@ -228,7 +229,7 @@ static void __init of_sama5d2_clk_i2s_mux_setup(struct device_node *np) return; for_each_child_of_node(np, i2s_mux_np) { - if (of_property_read_u8(i2s_mux_np, "reg", &bus_id)) + if (of_property_read_reg(i2s_mux_np, 0, &bus_id, NULL)) continue; if (bus_id > I2S_BUS_NR) From 88f46b866fe93b18abbca52085971ee382ece771 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Mon, 27 Jul 2026 10:47:04 +0200 Subject: [PATCH 023/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index a96bc8edf487fa..7b531fb5d11ee8 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -7,6 +7,13 @@ soc/arm https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-arm-soc-for-v7.3-tag1 ixp4xx/arm https://git.kernel.org/pub/scm/linux/kernel/git/linusw/linux-integrator tags/ixp4xx-arm-v7.3 + patch + gpio: sa1100: register software node for GPIO controller + ARM: sa1100: assabet: convert gpio-keys to use software nodes + ARM: sa1100: collie: convert gpio-keys to use software nodes + ARM: sa1100: h3xxx: convert gpio-keys to use software nodes + lpc32xx/soc + https://github.com/vzapolskiy/linux-lpc32xx tags/lpc32xx-arm-for-7.3 soc/dt patch @@ -16,6 +23,8 @@ soc/dt ARM: dts: ixp4xx: Drop the reg-offset hack canaan/dt https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux tags/riscv-dt-for-v7.3-early-k230 + renesas/dts + https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-dts-for-v7.3-tag1 soc/drivers ixp4xx/soc-drivers From df5317e6cd9ebc5a2af71c62531f4c7661378ae1 Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Sat, 25 Jul 2026 14:58:09 -0700 Subject: [PATCH 024/857] ARM: dts: BCM5301X: EA6300: fix USB3 USB3 needs to have a GPIO pulled HIGH in order to function. Add vcc-gpio to do so. Signed-off-by: Rosen Penev Link: https://lore.kernel.org/r/20260725215809.9464-1-rosenp@gmail.com Signed-off-by: Florian Fainelli --- arch/arm/boot/dts/broadcom/bcm4708-linksys-ea6300-v1.dts | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/arch/arm/boot/dts/broadcom/bcm4708-linksys-ea6300-v1.dts b/arch/arm/boot/dts/broadcom/bcm4708-linksys-ea6300-v1.dts index 0ed25bf71f0dfa..03d5546a147c4b 100644 --- a/arch/arm/boot/dts/broadcom/bcm4708-linksys-ea6300-v1.dts +++ b/arch/arm/boot/dts/broadcom/bcm4708-linksys-ea6300-v1.dts @@ -46,3 +46,7 @@ &usb3_phy { status = "okay"; }; + +&usb3 { + vcc-gpio = <&chipcommon 10 GPIO_ACTIVE_HIGH>; +}; From 90a77291ac0995738eb4abad8838c393ab4abe30 Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Sat, 25 Jul 2026 14:56:34 -0700 Subject: [PATCH 025/857] ARM: dts: BCM5301X: R8000 add NVRAM with MAC address for WAN port The R8000 stores the WAN MAC address at a fixed offset in NVRAM. Define the NVRAM region and attach nvmem-cells to the WAN port so the MAC is assigned automatically by the DSA framework. Assisted-by: opencode:big-pickle Signed-off-by: Rosen Penev Link: https://lore.kernel.org/r/20260725215634.9181-1-rosenp@gmail.com Signed-off-by: Florian Fainelli --- arch/arm/boot/dts/broadcom/bcm4709-netgear-r8000.dts | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/arch/arm/boot/dts/broadcom/bcm4709-netgear-r8000.dts b/arch/arm/boot/dts/broadcom/bcm4709-netgear-r8000.dts index e85693fba16af4..a4b135d3765936 100644 --- a/arch/arm/boot/dts/broadcom/bcm4709-netgear-r8000.dts +++ b/arch/arm/boot/dts/broadcom/bcm4709-netgear-r8000.dts @@ -36,6 +36,15 @@ <0x88000000 0x08000000>; }; + nvram@1c080000 { + compatible = "brcm,nvram"; + reg = <0x1c080000 0x180000>; + + et2macaddr: et2macaddr { + #nvmem-cell-cells = <1>; + }; + }; + leds { compatible = "gpio-leds"; @@ -211,6 +220,9 @@ port@4 { label = "wan"; + + nvmem-cells = <&et2macaddr 1>; + nvmem-cell-names = "mac-address"; }; port@5 { From 83dcde478d1ed0b108cf9b440dcc0b137fe1a59d Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Sun, 28 Jun 2026 16:10:49 -0700 Subject: [PATCH 026/857] ARM: dts: BCM5301X: EA9200: fix nvram size Fixes: [ 0.182121] WARNING: CPU: 0 PID: 1 at drivers/nvmem/brcm_nvram.c:85 brcm_nvram_probe+0x400/0x480 [ 0.182159] Unexpected (big) NVRAM size: 1056112 B Signed-off-by: Rosen Penev Link: https://lore.kernel.org/r/20260628231049.1248899-1-rosenp@gmail.com Signed-off-by: Florian Fainelli --- arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts b/arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts index 5bbc2ba0f95989..87569408bb6959 100644 --- a/arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts +++ b/arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts @@ -26,7 +26,7 @@ nvram@1c080000 { compatible = "brcm,nvram"; - reg = <0x1c080000 0x180000>; + reg = <0x1c080000 0x100000>; et2macaddr: et2macaddr { #nvmem-cell-cells = <1>; From 42c57c049054dfaa0be83f6721f7c4ce4a880e56 Mon Sep 17 00:00:00 2001 From: Ilya Sorochan Date: Fri, 6 Mar 2026 20:19:39 +0300 Subject: [PATCH 027/857] riscv: dts: starfive: jh7110-common: fix jh7110 SoC boot from SD-card. Add bootph-pre-ram to mmc1_pins clk-pins. U-Boot pruned their overrides recently in favor of Linux which broke booting from sd-card for me and Heinrich Schuchardt [1]. Pruning commit: 27f617019dd070cb61f2 ("riscv: dts: starfive: prune redundant jh7110-common overrides") [1] https://lore.kernel.org/all/ffdfc550-559b-4c59-9873-3f040fc3bb0e@canonical.com/ Signed-off-by: Ilya Sorochan Signed-off-by: Conor Dooley --- arch/riscv/boot/dts/starfive/jh7110-common.dtsi | 1 + 1 file changed, 1 insertion(+) diff --git a/arch/riscv/boot/dts/starfive/jh7110-common.dtsi b/arch/riscv/boot/dts/starfive/jh7110-common.dtsi index a7a1c09a2c9075..7d1eda67aeb5b9 100644 --- a/arch/riscv/boot/dts/starfive/jh7110-common.dtsi +++ b/arch/riscv/boot/dts/starfive/jh7110-common.dtsi @@ -438,6 +438,7 @@ input-disable; input-schmitt-disable; slew-rate = <0>; + bootph-pre-ram; }; mmc-pins { From 0c642cefc072fc2e94b3c6531529ff352a855e50 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Wed, 29 Jul 2026 17:57:44 +0200 Subject: [PATCH 028/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 7b531fb5d11ee8..661b7b3b33c62d 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -25,6 +25,18 @@ soc/dt https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux tags/riscv-dt-for-v7.3-early-k230 renesas/dts https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-dts-for-v7.3-tag1 + realtek/dts + https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux tags/realtek-dt-v7.3 + exynos/dt + https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux tags/samsung-dt64-7.3 + thead/dts + https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux tags/thead-dt-for-v7.3 + rockchip/dt32 + https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip tags/v7.3-rockchip-dts32-1 + socfpga/dts + https://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux tags/socfpga_dts_updates_for_v7.3 + cix/dts + https://github.com/cixtech/linux-mainline tags/cix-dt-v7.3-rc1 soc/drivers ixp4xx/soc-drivers @@ -35,6 +47,10 @@ soc/drivers https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip tags/v7.3-rockchip-drivers1 renesas/drivers https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-drivers-for-v7.3-tag1 + drivers/memory + https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl tags/memory-controller-drv-7.3 + exynos/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux tags/samsung-drivers-7.3 soc/defconfig riscv/defconfig From ef681b1ee43da67c9280f776ca8d70c7c3678935 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Thu, 30 Jul 2026 11:40:09 +0200 Subject: [PATCH 029/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 661b7b3b33c62d..627e782b02af54 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -14,6 +14,8 @@ soc/arm ARM: sa1100: h3xxx: convert gpio-keys to use software nodes lpc32xx/soc https://github.com/vzapolskiy/linux-lpc32xx tags/lpc32xx-arm-for-7.3 + imx/soc + https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux tags/imx-soc-7.3 soc/dt patch @@ -37,6 +39,12 @@ soc/dt https://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux tags/socfpga_dts_updates_for_v7.3 cix/dts https://github.com/cixtech/linux-mainline tags/cix-dt-v7.3-rc1 + imx/dt + https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux tags/imx-dt-7.3 + mediatek/dt32 + https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-dts32-for-v7.3 + mediatek/dt64 + https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-dts64-for-v7.3 soc/drivers ixp4xx/soc-drivers @@ -51,6 +59,8 @@ soc/drivers https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl tags/memory-controller-drv-7.3 exynos/drivers https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux tags/samsung-drivers-7.3 + mediatek/soc + https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-soc-for-v7.3 soc/defconfig riscv/defconfig @@ -61,4 +71,8 @@ soc/late arm/fixes broadcom/fixes https://github.com/Broadcom/stblinux tags/arm-soc/for-7.2/devicetree-arm64-fixes + imx/maintainers + https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux tags/imx-maintainers-7.3 + qcom/dt-fixes + https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm64-fixes-for-7.2 From 06dc86890613afaf20649cbc08f611efcc496091 Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Mon, 27 Jul 2026 13:26:30 -0700 Subject: [PATCH 030/857] ARM: dts: BCM5301X: AC9: set WAN MAC from nvram The WAN MAC is offset by 1. Set in dts to avoid having to handle this in userspace. nvram size found from a random bootlog online. Signed-off-by: Rosen Penev Link: https://lore.kernel.org/r/20260727202630.21572-1-rosenp@gmail.com Signed-off-by: Florian Fainelli --- .../boot/dts/broadcom/bcm47189-tenda-ac9.dts | 20 +++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/arch/arm/boot/dts/broadcom/bcm47189-tenda-ac9.dts b/arch/arm/boot/dts/broadcom/bcm47189-tenda-ac9.dts index 3ac6cac541cacf..8b2ad7683529bd 100644 --- a/arch/arm/boot/dts/broadcom/bcm47189-tenda-ac9.dts +++ b/arch/arm/boot/dts/broadcom/bcm47189-tenda-ac9.dts @@ -20,6 +20,16 @@ reg = <0x00000000 0x08000000>; }; + nvram@1eff0000 { + compatible = "brcm,nvram"; + reg = <0x1eff0000 0x10000>; + + et0macaddr: et0macaddr { + #nvmem-cell-cells = <1>; + }; + }; + + leds-0 { compatible = "gpio-leds"; @@ -106,6 +116,16 @@ }; }; +&gmac0 { + nvmem-cells = <&et0macaddr 0>; + nvmem-cell-names = "mac-address"; +}; + +&gmac1 { + nvmem-cells = <&et0macaddr 1>; + nvmem-cell-names = "mac-address"; +}; + &switch { status = "okay"; From a597183e14549b99f640c99db31b1490d825d639 Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Mon, 27 Jul 2026 13:27:37 -0700 Subject: [PATCH 031/857] ARM: dts: BCM5301X: DIR-890L: set WAN MAC from nvram The WAN MAC is offset by 1 from et0macaddr, same as DIR-885L. Signed-off-by: Rosen Penev Link: https://lore.kernel.org/r/20260727202737.21862-1-rosenp@gmail.com Signed-off-by: Florian Fainelli --- arch/arm/boot/dts/broadcom/bcm47094-dlink-dir-890l.dts | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/arch/arm/boot/dts/broadcom/bcm47094-dlink-dir-890l.dts b/arch/arm/boot/dts/broadcom/bcm47094-dlink-dir-890l.dts index 3124dfd01b9447..944d592dee2c24 100644 --- a/arch/arm/boot/dts/broadcom/bcm47094-dlink-dir-890l.dts +++ b/arch/arm/boot/dts/broadcom/bcm47094-dlink-dir-890l.dts @@ -114,6 +114,7 @@ reg = <0x1e1f0000 0x00010000>; et0macaddr: et0macaddr { + #nvmem-cell-cells = <1>; }; }; }; @@ -125,7 +126,7 @@ * actually in use on the platform, we use this et0 MAC * address for et2. */ - nvmem-cells = <&et0macaddr>; + nvmem-cells = <&et0macaddr 0>; nvmem-cell-names = "mac-address"; }; @@ -190,6 +191,9 @@ port@4 { label = "wan"; + + nvmem-cells = <&et0macaddr 1>; + nvmem-cell-names = "mac-address"; }; port@5 { From 179a27876b963ccadb8e1f68f51bf3aabeb14d29 Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Mon, 27 Jul 2026 13:28:17 -0700 Subject: [PATCH 032/857] ARM: dts: BCM5301X: SR400AC: set WAN MAC from nvram The WAN MAC is offset by 1. Set in dts to avoid having to handle this in userspace. nvram size found from a random bootlog online. Signed-off-by: Rosen Penev Link: https://lore.kernel.org/r/20260727202817.22085-1-rosenp@gmail.com Signed-off-by: Florian Fainelli --- .../dts/broadcom/bcm4708-smartrg-sr400ac.dts | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/arch/arm/boot/dts/broadcom/bcm4708-smartrg-sr400ac.dts b/arch/arm/boot/dts/broadcom/bcm4708-smartrg-sr400ac.dts index b226bef3369cf7..03f4a6d37a674b 100644 --- a/arch/arm/boot/dts/broadcom/bcm4708-smartrg-sr400ac.dts +++ b/arch/arm/boot/dts/broadcom/bcm4708-smartrg-sr400ac.dts @@ -25,6 +25,16 @@ <0x88000000 0x08000000>; }; + nvram@1eff0000 { + compatible = "brcm,nvram"; + reg = <0x1eff0000 0x10000>; + + et0macaddr: et0macaddr { + #nvmem-cell-cells = <1>; + }; + }; + + leds { compatible = "gpio-leds"; @@ -118,6 +128,11 @@ status = "okay"; }; +&gmac0 { + nvmem-cells = <&et0macaddr 0>; + nvmem-cell-names = "mac-address"; +}; + &srab { status = "okay"; @@ -140,6 +155,9 @@ port@4 { label = "wan"; + + nvmem-cells = <&et0macaddr 1>; + nvmem-cell-names = "mac-address"; }; port@5 { From f4342f9e537c02ef6e2f9b3da44ed83fc1fdb518 Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Mon, 27 Jul 2026 15:17:28 -0700 Subject: [PATCH 033/857] ARM: dts: BCM5301X: Add missing phy-mode and fixed-link to DSA switch ports Ports 5, 7 and 8 in the BCM5301x SRAB switch node carry an 'ethernet' property linking them to GMACs, but lack the required 'phy-mode' and 'fixed-link' properties. The DSA port schema (dsa-port.yaml) mandates that any port with an 'ethernet' property must also have 'phy-mode' and one of 'fixed-link', 'phy-handle', or 'managed'. Add phy-mode = "internal" and a fixed-link subnode to ports 5, 7 and 8, matching the properties already present on their respective GMAC nodes (gmac0, gmac1, gmac2). This resolves dtbs_check schema validation warnings for all BCM5301x-based board DTBs. Assisted-by: Opencode:Big-Pickle Signed-off-by: Rosen Penev Link: https://lore.kernel.org/r/20260727221728.40811-1-rosenp@gmail.com Signed-off-by: Florian Fainelli --- arch/arm/boot/dts/broadcom/bcm-ns.dtsi | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/arch/arm/boot/dts/broadcom/bcm-ns.dtsi b/arch/arm/boot/dts/broadcom/bcm-ns.dtsi index 392a2571366964..b671d3f4575659 100644 --- a/arch/arm/boot/dts/broadcom/bcm-ns.dtsi +++ b/arch/arm/boot/dts/broadcom/bcm-ns.dtsi @@ -442,16 +442,29 @@ port@5 { reg = <5>; ethernet = <&gmac0>; + phy-mode = "internal"; + + fixed-link { + speed = <1000>; + full-duplex; + }; }; port@7 { reg = <7>; ethernet = <&gmac1>; + phy-mode = "internal"; + + fixed-link { + speed = <1000>; + full-duplex; + }; }; port@8 { reg = <8>; ethernet = <&gmac2>; + phy-mode = "internal"; fixed-link { speed = <1000>; From c62bbf2a690b55965d22e414e5adbfd5dc76d701 Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Mon, 27 Jul 2026 15:54:46 -0700 Subject: [PATCH 034/857] ARM: dts: BCM5301X: EA9200: add Wi-Fi definitions brcm,ccode-map and ieee80211-freq-limit are needed to be specified on some of them for proper operation. Also add an nvmem definition for the MAC addresses. Signed-off-by: Rosen Penev Link: https://lore.kernel.org/r/20260727225446.267097-1-rosenp@gmail.com Signed-off-by: Florian Fainelli --- .../dts/broadcom/bcm4709-linksys-ea9200.dts | 57 +++++++++++++++++++ 1 file changed, 57 insertions(+) diff --git a/arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts b/arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts index 87569408bb6959..9041dc6bfa8d15 100644 --- a/arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts +++ b/arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts @@ -93,6 +93,63 @@ }; }; +&pcie_bridge0 { + pcie@0,0 { + device_type = "pci"; + reg = <0x0000 0 0 0 0>; + + #address-cells = <3>; + #size-cells = <2>; + ranges; + + pcie@1,0 { + device_type = "pci"; + reg = <0x800 0 0 0 0>; + + #address-cells = <3>; + #size-cells = <2>; + ranges; + + wifi@0,0 { + compatible = "brcm,bcm4366-fmac", "brcm,bcm4329-fmac"; + reg = <0x0000 0 0 0 0>; + ieee80211-freq-limit = <5170000 5250000>; + brcm,ccode-map = "JP-JP-78", "US-Q2-86"; + nvmem-cells = <&et2macaddr 2>; + nvmem-cell-names = "mac-address"; + }; + }; + + pcie@2,0 { + device_type = "pci"; + reg = <0x1000 0 0 0 0>; + + #address-cells = <3>; + #size-cells = <2>; + ranges; + + wifi@0,0 { + compatible = "brcm,bcm4366-fmac", "brcm,bcm4329-fmac"; + reg = <0x0000 0 0 0 0>; + brcm,ccode-map = "JP-JP-78", "US-Q2-86"; + nvmem-cells = <&et2macaddr 3>; + nvmem-cell-names = "mac-address"; + }; + }; + }; +}; + +&pcie_bridge1 { + wifi@0,0 { + compatible = "brcm,bcm4366-fmac", "brcm,bcm4329-fmac"; + reg = <0x0000 0 0 0 0>; + ieee80211-freq-limit = <5735000 5835000>; + brcm,ccode-map = "JP-JP-78", "US-Q2-86"; + nvmem-cells = <&et2macaddr 4>; + nvmem-cell-names = "mac-address"; + }; +}; + &usb3_phy { status = "okay"; }; From fca5276165993e9ca18e746f9b41af7627cbe349 Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Wed, 29 Jul 2026 15:54:02 -0700 Subject: [PATCH 035/857] ARM: dts: BCM5301X: panamera: add phy-mode to switch phy-mode is needed for CPU ports based on the documentation. Fixes dtbs_checks warning: switch@0 (brcm,bcm53125): ports:port@8: 'phy-mode' is a required property from schema $id: http://devicetree.org/schemas/net/dsa/brcm,b53.yaml Signed-off-by: Rosen Penev Link: https://lore.kernel.org/r/20260729225402.775442-1-rosenp@gmail.com Signed-off-by: Florian Fainelli --- arch/arm/boot/dts/broadcom/bcm47094-linksys-panamera.dts | 1 + 1 file changed, 1 insertion(+) diff --git a/arch/arm/boot/dts/broadcom/bcm47094-linksys-panamera.dts b/arch/arm/boot/dts/broadcom/bcm47094-linksys-panamera.dts index 74161b76008ad0..8d9f640981ec56 100644 --- a/arch/arm/boot/dts/broadcom/bcm47094-linksys-panamera.dts +++ b/arch/arm/boot/dts/broadcom/bcm47094-linksys-panamera.dts @@ -184,6 +184,7 @@ sw1_p8: port@8 { reg = <8>; ethernet = <&sw0_p0>; + phy-mode = "rgmii"; label = "cpu"; fixed-link { From 60039faf8135ff05714a2014e54597ebfe07340f Mon Sep 17 00:00:00 2001 From: "Michael S. Tsirkin" Date: Sun, 5 Jul 2026 05:38:57 -0400 Subject: [PATCH 036/857] virtio_balloon: prime stats vq after virtio_device_ready() The virtio spec requires the driver not to kick the device before DRIVER_OK is set. init_vqs() primes the stats virtqueue with a buffer and kicks the device before virtio_device_ready() is called in virtballoon_probe(), violating this requirement. Further, if the device responds to the early kick by processing the buffer before DRIVER_OK, stats_request() fires and queues update_balloon_stats_work. Should probe then fail and free vb, the work runs against freed memory. To fix, move buffer setup to after DRIVER_OK. Be careful to disable update_balloon_stats_work while this is going on, to make sure it does not race with the setup. setup_vqs() warns but does not fail probe or restore if virtqueue_add_outbuf() fails; the call never actually fails in these contexts since the queue is freshly initialized and empty. Testing: tested that stats still work after the change. Fixes: 9564e138b1f6 ("virtio: Add memory statistics reporting to the balloon driver (V4)") Reported-by: Sashiko:gemini-3.1-pro-preview Cc: David Hildenbrand Assisted-by: Claude:claude-sonnet-4-6 Message-ID: Signed-off-by: Michael S. Tsirkin --- drivers/virtio/virtio_balloon.c | 51 +++++++++++++++++++++------------ 1 file changed, 33 insertions(+), 18 deletions(-) diff --git a/drivers/virtio/virtio_balloon.c b/drivers/virtio/virtio_balloon.c index 581ac799d97495..7c5ef4e5c87915 100644 --- a/drivers/virtio/virtio_balloon.c +++ b/drivers/virtio/virtio_balloon.c @@ -611,25 +611,9 @@ static int init_vqs(struct virtio_balloon *vb) vb->inflate_vq = vqs[VIRTIO_BALLOON_VQ_INFLATE]; vb->deflate_vq = vqs[VIRTIO_BALLOON_VQ_DEFLATE]; if (virtio_has_feature(vb->vdev, VIRTIO_BALLOON_F_STATS_VQ)) { - struct scatterlist sg; - unsigned int num_stats; vb->stats_vq = vqs[VIRTIO_BALLOON_VQ_STATS]; - - /* - * Prime this virtqueue with one buffer so the hypervisor can - * use it to signal us later (it can't be broken yet!). - */ - num_stats = update_balloon_stats(vb); - - sg_init_one(&sg, vb->stats, sizeof(vb->stats[0]) * num_stats); - err = virtqueue_add_outbuf(vb->stats_vq, &sg, 1, vb, - GFP_KERNEL); - if (err) { - dev_warn(&vb->vdev->dev, "%s: add stat_vq failed\n", - __func__); - return err; - } - virtqueue_kick(vb->stats_vq); + /* Prevent update_balloon_stats_work from accessing the stats vq. */ + disable_work(&vb->update_balloon_stats_work); } if (virtio_has_feature(vb->vdev, VIRTIO_BALLOON_F_FREE_PAGE_HINT)) @@ -916,6 +900,33 @@ static int virtio_balloon_register_shrinker(struct virtio_balloon *vb) return 0; } +static void setup_vqs(struct virtio_balloon *vb) +{ + struct scatterlist sg; + unsigned int num_stats; + bool ret; + + if (!virtio_has_feature(vb->vdev, VIRTIO_BALLOON_F_STATS_VQ)) + return; + + /* + * Prime this virtqueue with one buffer so the hypervisor can + * use it to signal us later (it can't be broken yet!). + */ + num_stats = update_balloon_stats(vb); + sg_init_one(&sg, vb->stats, sizeof(vb->stats[0]) * num_stats); + if (virtqueue_add_outbuf(vb->stats_vq, &sg, 1, vb, GFP_KERNEL)) { + dev_warn(&vb->vdev->dev, "%s: add stat_vq failed\n", __func__); + return; + } + virtqueue_kick(vb->stats_vq); + + ret = enable_and_queue_work(system_freezable_wq, + &vb->update_balloon_stats_work); + /* Make sure we balanced enable/disable, or we won't report stats. */ + WARN_ON_ONCE(!ret); +} + static int virtballoon_probe(struct virtio_device *vdev) { struct virtio_balloon *vb; @@ -1056,6 +1067,8 @@ static int virtballoon_probe(struct virtio_device *vdev) virtio_device_ready(vdev); + setup_vqs(vb); + if (towards_target(vb)) virtballoon_changed(vdev); return 0; @@ -1145,6 +1158,8 @@ static int virtballoon_restore(struct virtio_device *vdev) virtio_device_ready(vdev); + setup_vqs(vb); + if (towards_target(vb)) virtballoon_changed(vdev); update_balloon_size(vb); From 6236e765f16d2648adbaa49bfc539df28cc255ac Mon Sep 17 00:00:00 2001 From: Pavel Tikhomirov Date: Mon, 20 Jul 2026 13:22:37 +0300 Subject: [PATCH 037/857] vhost/vsock: split out vhost_vsock_drop_backends helper Split the actual backend dropping part from vhost_vsock_stop. We're going to need it for the VHOST_RESET_OWNER implementation in the following patch, when vsock->dev.mutex is already taken and owner is checked. Signed-off-by: Pavel Tikhomirov Signed-off-by: Andrey Drobyshev Reviewed-by: Pavel Tikhomirov Reviewed-by: Stefano Garzarella Message-ID: <20260720102241.371610-2-andrey.drobyshev@virtuozzo.com> Signed-off-by: Michael S. Tsirkin --- drivers/vhost/vsock.c | 26 +++++++++++++++++--------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/drivers/vhost/vsock.c b/drivers/vhost/vsock.c index 9aaab6bb8061c1..b12221ce6faf22 100644 --- a/drivers/vhost/vsock.c +++ b/drivers/vhost/vsock.c @@ -664,9 +664,24 @@ static int vhost_vsock_start(struct vhost_vsock *vsock) return ret; } -static int vhost_vsock_stop(struct vhost_vsock *vsock, bool check_owner) +static void vhost_vsock_drop_backends(struct vhost_vsock *vsock) { + struct vhost_virtqueue *vq; size_t i; + + lockdep_assert_held(&vsock->dev.mutex); + + for (i = 0; i < ARRAY_SIZE(vsock->vqs); i++) { + vq = &vsock->vqs[i]; + + mutex_lock(&vq->mutex); + vhost_vq_set_backend(vq, NULL); + mutex_unlock(&vq->mutex); + } +} + +static int vhost_vsock_stop(struct vhost_vsock *vsock, bool check_owner) +{ int ret = 0; mutex_lock(&vsock->dev.mutex); @@ -677,14 +692,7 @@ static int vhost_vsock_stop(struct vhost_vsock *vsock, bool check_owner) goto err; } - for (i = 0; i < ARRAY_SIZE(vsock->vqs); i++) { - struct vhost_virtqueue *vq = &vsock->vqs[i]; - - mutex_lock(&vq->mutex); - vhost_vq_set_backend(vq, NULL); - mutex_unlock(&vq->mutex); - } - + vhost_vsock_drop_backends(vsock); err: mutex_unlock(&vsock->dev.mutex); return ret; From 7c4fedf10a831b471dd94ab7bed499c305db27c9 Mon Sep 17 00:00:00 2001 From: Andrey Drobyshev Date: Mon, 20 Jul 2026 13:22:38 +0300 Subject: [PATCH 038/857] vhost/vsock: suppress EHOSTUNREACH fast-fail during CPR pause Earlier commit bb26ed5f3a8b ("vhost/vsock: Refuse the connection immediately when guest isn't ready") added a fast-fail in vhost_transport_send_pkt(). It rejects every host send with -EHOSTUNREACH until the destination calls SET_RUNNING(1). The fast-fail condition checks whether device's backends are dropped, and if they're, the guest is considered to be not ready. However, there might be other reasons for backends to be nulled. In particular, when QEMU is performing CPR (checkpoint-restore) migration, device ownership is being RESET and SET again, which leads to backends drop and reattach. If we end up connecting during this window, an AF_VSOCK client gets -EHOSTUNREACH, which is wrong. Add an 'ever_started' flag which is set once in vhost_vsock_start() and is never cleared. The behaviour changes to: * When device was never started -> flag is unset -> no listener can exist yet -> fast-fail; * Once the device starts -> flag is set -> we don't fast-fail -> we queue and preserve during any later stop / CPR pause. The VHOST_RESET_OWNER ioctl is implemented in a following patch, and without RESET_OWNER the problem we fix here isn't manifesting - thus this patch is a preparation to support RESET_OWNER. Important caveat: after the first start, a connect during any stopped window is queued instead of fast-failed. That was the behaviour before the patch bb26ed5f3a8b, and we're restoring it now. However we still keep the behaviour originally intended by that commit (i.e. fast-fail if there's no real listener yet) while fixing the CPR path. Suggested-by: Stefano Garzarella Signed-off-by: Denis V. Lunev Signed-off-by: Andrey Drobyshev Reviewed-by: Pavel Tikhomirov Reviewed-by: Stefano Garzarella Message-ID: <20260720102241.371610-3-andrey.drobyshev@virtuozzo.com> Signed-off-by: Michael S. Tsirkin --- drivers/vhost/vsock.c | 22 ++++++++++++---------- 1 file changed, 12 insertions(+), 10 deletions(-) diff --git a/drivers/vhost/vsock.c b/drivers/vhost/vsock.c index b12221ce6faf22..27169a09e87ead 100644 --- a/drivers/vhost/vsock.c +++ b/drivers/vhost/vsock.c @@ -61,6 +61,7 @@ struct vhost_vsock { u32 guest_cid; bool seqpacket_allow; + bool ever_started; /* set on first SET_RUNNING(1); never cleared */ }; static u32 vhost_transport_get_local_cid(void) @@ -302,17 +303,12 @@ vhost_transport_send_pkt(struct sk_buff *skb, struct net *net) return -ENODEV; } - /* Fast-fail if the guest hasn't enabled the RX vq yet. Queuing the packet - * and making the caller wait is pointless: even if the guest manages to init - * within the timeout, it'll immediately reply with RST, because there's no - * listener on the port yet. - * - * vhost_vq_get_backend() without vq->mutex is acceptable here: locking - * the mutex would be too expensive in this hot path, and we already have - * all the outcomes covered: if the backend becomes NULL right after the check, - * vhost_transport_do_send_pkt() will check it under the mutex anyway. + /* Fast-fail until the guest first enables the device (SET_RUNNING(1)). + * Before that there is no listener, so queuing is pointless. + * 'ever_started' is never cleared, so once we're up we keep queuing + * across later stop / CPR-pause windows. */ - if (unlikely(!data_race(vhost_vq_get_backend(&vsock->vqs[VSOCK_VQ_RX])))) { + if (unlikely(!READ_ONCE(vsock->ever_started))) { rcu_read_unlock(); kfree_skb(skb); return -EHOSTUNREACH; @@ -640,6 +636,11 @@ static int vhost_vsock_start(struct vhost_vsock *vsock) mutex_unlock(&vq->mutex); } + /* Set 'ever_started' flag on the first start; never cleared, so send_pkt + * keeps queuing (instead of fast-failing) on later stop / CPR pauses. + */ + WRITE_ONCE(vsock->ever_started, true); + /* Some packets may have been queued before the device was started, * let's kick the send worker to send them. */ @@ -728,6 +729,7 @@ static int vhost_vsock_dev_open(struct inode *inode, struct file *file) vsock->guest_cid = 0; /* no CID assigned yet */ vsock->seqpacket_allow = false; + vsock->ever_started = false; atomic_set(&vsock->queued_replies, 0); From 863c6d539ac64ec3fb1e08a4e2fe7f5aa7d1a781 Mon Sep 17 00:00:00 2001 From: Andrey Drobyshev Date: Mon, 20 Jul 2026 13:22:39 +0300 Subject: [PATCH 039/857] vhost/vsock: re-scan TX virtqueue on device start During QEMU CPR live-update (and VHOST_RESET_OWNER in general) the guest keeps running while the host drops and later re-attaches vhost backends. If the guest adds a buffer to the TX virtqueue (guest->host) and kicks while the backend is temporarily NULL (between vhost_vsock_drop_backends() and the next vhost_vsock_start()), then the kick is delivered to the vhost worker, handle_tx_kick() sees a NULL backend and returns, and the kick signal is consumed. The buffer is then left in the ring. Then upon device start vhost_vsock_start() only re-kicks the RX send worker, never the TX VQ, so the buffer is processed only if the guest happens to kick again. But if the guest itself is now waiting for data from the host, it will never kick TX VQ again, and we end up in a deadlock. The issue itself is pre-existing, but it only manifests during a device pause caused by VHOST_RESET_OWNER. Namely, the deadlock is reproduced during active host->guest socat data transfer under multiple consecutive CPR live-update's. To fix this, in vhost_vsock_start(), after kicking the RX send worker, also queue the TX vq poll so any buffers the guest enqueued while we were paused get scanned. The VHOST_RESET_OWNER ioctl itself is implemented in the following patch, thus this patch is a preparation to support VHOST_RESET_OWNER. Signed-off-by: Andrey Drobyshev Reviewed-by: Pavel Tikhomirov Reviewed-by: Stefano Garzarella Message-ID: <20260720102241.371610-4-andrey.drobyshev@virtuozzo.com> Signed-off-by: Michael S. Tsirkin --- drivers/vhost/vsock.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/drivers/vhost/vsock.c b/drivers/vhost/vsock.c index 27169a09e87ead..d5022d21120b30 100644 --- a/drivers/vhost/vsock.c +++ b/drivers/vhost/vsock.c @@ -646,6 +646,13 @@ static int vhost_vsock_start(struct vhost_vsock *vsock) */ vhost_vq_work_queue(&vsock->vqs[VSOCK_VQ_RX], &vsock->send_pkt_work); + /* The guest may have added TX buffers while the device was stopped + * (e.g. across VHOST_RESET_OWNER) and their kicks got consumed by + * the NULL-backend window. Re-scan the TX VQ, mirroring the RX + * send-worker kick above. + */ + vhost_poll_queue(&vsock->vqs[VSOCK_VQ_TX].poll); + mutex_unlock(&vsock->dev.mutex); return 0; From 0088c10f7cb7adc585c1d3f47f4dc46b3b14ba02 Mon Sep 17 00:00:00 2001 From: Andrey Drobyshev Date: Mon, 20 Jul 2026 13:22:40 +0300 Subject: [PATCH 040/857] vhost: synchronize with RCU readers when freeing workers vhost_vq_work_queue() only holds the RCU read lock while it dereferences vq->worker and queues work on it. vhost_workers_free() however clears the vq->worker pointers and immediately frees the workers, without waiting for a grace period. A caller that fetched the worker right before the pointer was cleared can therefore still be queueing work on it while it is freed. And even when the queueing itself wins the race, the work is never run, so its VHOST_WORK_QUEUED bit stays set and all future attempts to queue it are silently skipped. None of the current callers can actually hit this: net and scsi stop their virtqueues before the workers are freed, and vsock unhashes the device and does synchronize_rcu() of its own in vhost_vsock_dev_release() before the workers go away. But the upcoming VHOST_RESET_OWNER support in vhost-vsock keeps the device hashed while its workers are freed, so the lockless send/cancel paths become able to race with the teardown. Fix this by clearing the vq->worker pointers, waiting for a grace period, and then flushing the workers so any work the last readers queued runs before the workers are freed. Fixes: 228a27cf78af ("vhost: Allow worker switching while work is queueing") Suggested-by: Stefano Garzarella Signed-off-by: Andrey Drobyshev Reviewed-by: Stefano Garzarella Message-ID: <20260720102241.371610-5-andrey.drobyshev@virtuozzo.com> Signed-off-by: Michael S. Tsirkin --- drivers/vhost/vhost.c | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/drivers/vhost/vhost.c b/drivers/vhost/vhost.c index 269efad90369e5..1be9c0c53bedbb 100644 --- a/drivers/vhost/vhost.c +++ b/drivers/vhost/vhost.c @@ -729,6 +729,17 @@ static void vhost_workers_free(struct vhost_dev *dev) for (i = 0; i < dev->nvqs; i++) rcu_assign_pointer(dev->vqs[i]->worker, NULL); + + /* + * vhost_vq_work_queue() reads vq->worker under rcu_read_lock(), so a + * reader that fetched a worker before we cleared the pointers above + * may still be queueing work on it. Wait for those readers to + * finish, then flush so any work they queued runs (clearing + * VHOST_WORK_QUEUED) before the workers are freed. + */ + synchronize_rcu(); + vhost_dev_flush(dev); + /* * Free the default worker we created and cleanup workers userspace * created but couldn't clean up (it forgot or crashed). From c7bfb8815d83a63c974878180499c52411839ded Mon Sep 17 00:00:00 2001 From: Pavel Tikhomirov Date: Mon, 20 Jul 2026 13:22:41 +0300 Subject: [PATCH 041/857] vhost/vsock: add VHOST_RESET_OWNER ioctl This ioctl is needed for QEMU's CPR (checkpoint-restore) migration of the guest with vhost-vsock device. For this to work, we need to reset the device ownership on the source side by calling RESET_OWNER, and then claim it on the dest side by calling SET_OWNER. We expect not to lose any AF_VSOCK connection while this happens. To that end, unlike the release path, RESET_OWNER keeps the guest CID hashed: established connections survive, and host sends issued while the device is between owners simply stay on send_pkt_queue until the next device start drains them. Since the device stays reachable through the CID hash, the lockless send/cancel paths can race with the worker teardown in vhost_workers_free(). The previous commit ("vhost: synchronize with RCU readers when freeing workers") makes that safe. Signed-off-by: Pavel Tikhomirov Signed-off-by: Andrey Drobyshev Reviewed-by: Stefano Garzarella Message-ID: <20260720102241.371610-6-andrey.drobyshev@virtuozzo.com> Signed-off-by: Michael S. Tsirkin --- drivers/vhost/vsock.c | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/drivers/vhost/vsock.c b/drivers/vhost/vsock.c index d5022d21120b30..86f25ff80722d3 100644 --- a/drivers/vhost/vsock.c +++ b/drivers/vhost/vsock.c @@ -903,6 +903,29 @@ static int vhost_vsock_set_features(struct vhost_vsock *vsock, u64 features) return -EFAULT; } +static long vhost_vsock_reset_owner(struct vhost_vsock *vsock) +{ + struct vhost_iotlb *umem; + long err; + + mutex_lock(&vsock->dev.mutex); + err = vhost_dev_check_owner(&vsock->dev); + if (err) + goto done; + umem = vhost_dev_reset_owner_prepare(); + if (!umem) { + err = -ENOMEM; + goto done; + } + vhost_vsock_drop_backends(vsock); + vhost_vsock_flush(vsock); + vhost_dev_stop(&vsock->dev); + vhost_dev_reset_owner(&vsock->dev, umem); +done: + mutex_unlock(&vsock->dev.mutex); + return err; +} + static long vhost_vsock_dev_ioctl(struct file *f, unsigned int ioctl, unsigned long arg) { @@ -946,6 +969,8 @@ static long vhost_vsock_dev_ioctl(struct file *f, unsigned int ioctl, return -EOPNOTSUPP; vhost_set_backend_features(&vsock->dev, features); return 0; + case VHOST_RESET_OWNER: + return vhost_vsock_reset_owner(vsock); default: mutex_lock(&vsock->dev.mutex); r = vhost_dev_ioctl(&vsock->dev, ioctl, argp); From 0556300369d8fcdc0db08f8b4984a01119fd315f Mon Sep 17 00:00:00 2001 From: Peter Hilber Date: Fri, 5 Jun 2026 16:29:21 +0200 Subject: [PATCH 042/857] virtio-mmio: add support for transport version 3 Virtio MMIO transport version 3 allows device reset to complete asynchronously. Unlike version 2, where writing zero to Status must complete the reset before the write returns, version 3 requires the driver to poll Status until it reads back zero before considering reset complete. Update virtio-mmio accordingly: accept transport version 3 and, during reset, wait for Status to become zero. Keep the polling loop unbounded, consistent with virtio-pci, since the reset callback does not return an error code. Signed-off-by: Peter Hilber Link: https://github.com/oasis-tcs/virtio-spec/commit/bb1dd2e1fe89b862f38f15873d835a698b196f89 Message-ID: <20260605142921.2824-1-peter.hilber@oss.qualcomm.com> Signed-off-by: Michael S. Tsirkin --- drivers/virtio/virtio_mmio.c | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/drivers/virtio/virtio_mmio.c b/drivers/virtio/virtio_mmio.c index 510b7c4efdff8b..316f03b9735610 100644 --- a/drivers/virtio/virtio_mmio.c +++ b/drivers/virtio/virtio_mmio.c @@ -55,6 +55,7 @@ #define pr_fmt(fmt) "virtio-mmio: " fmt #include +#include #include #include #include @@ -114,9 +115,9 @@ static int vm_finalize_features(struct virtio_device *vdev) vring_transport_features(vdev); /* Make sure there are no mixed devices */ - if (vm_dev->version == 2 && + if (vm_dev->version >= 2 && !__virtio_test_bit(vdev, VIRTIO_F_VERSION_1)) { - dev_err(&vdev->dev, "New virtio-mmio devices (version 2) must provide VIRTIO_F_VERSION_1 feature!\n"); + dev_err(&vdev->dev, "New virtio-mmio devices (version >= 2) must provide VIRTIO_F_VERSION_1 feature!\n"); return -EINVAL; } @@ -254,6 +255,12 @@ static void vm_reset(struct virtio_device *vdev) /* 0 status means a reset. */ writel(0, vm_dev->base + VIRTIO_MMIO_STATUS); + + if (vm_dev->version >= 3) { + /* Wait for reset to complete. */ + while (vm_get_status(vdev)) + fsleep(1000); + } } @@ -600,7 +607,7 @@ static int virtio_mmio_probe(struct platform_device *pdev) /* Check device version */ vm_dev->version = readl(vm_dev->base + VIRTIO_MMIO_VERSION); - if (vm_dev->version < 1 || vm_dev->version > 2) { + if (vm_dev->version < 1 || vm_dev->version > 3) { dev_err(&pdev->dev, "Version %ld not supported!\n", vm_dev->version); rc = -ENXIO; From 22172d588a7cfa61d347774cfc0b4419d6c6388e Mon Sep 17 00:00:00 2001 From: "Michael S. Tsirkin" Date: Sun, 5 Jul 2026 02:24:18 -0400 Subject: [PATCH 043/857] virtio_balloon: disable indirect descriptors The page reporting callback submits an sg list to the reporting virtqueue. With VIRTIO_RING_F_INDIRECT_DESC negotiated and total_sg > 1 (which it typically is), virtqueue_add reports it to the host by allocating an indirect descriptor via kmalloc(GFP_KERNEL). This is not pretty: the reporting worker isolates potentially hundreds of MB of free pages from the buddy allocator (reported pages are at least pageblock_order, and the sg can contain up to PAGE_REPORTING_CAPACITY entries of varying orders). As the result, very theoretically, the kmalloc might trigger OOM when we have in fact a ton of free memory. Clear VIRTIO_RING_F_INDIRECT_DESC, to avoid using indirect descriptors. Fixes: b0c504f15471 ("virtio-balloon: add support for providing free page reports to host") Assisted-by: Claude:claude-opus-4-6 Acked-by: David Hildenbrand (Arm) Signed-off-by: Michael S. Tsirkin Message-ID: <73fac8a629fd9aca7bb3265ac243a769c28af25d.1783232420.git.mst@redhat.com> --- drivers/virtio/virtio_balloon.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/drivers/virtio/virtio_balloon.c b/drivers/virtio/virtio_balloon.c index 7c5ef4e5c87915..9309f3198db0e4 100644 --- a/drivers/virtio/virtio_balloon.c +++ b/drivers/virtio/virtio_balloon.c @@ -7,6 +7,7 @@ */ #include +#include #include #include #include @@ -1180,6 +1181,11 @@ static int virtballoon_validate(struct virtio_device *vdev) else if (!virtio_has_feature(vdev, VIRTIO_BALLOON_F_PAGE_POISON)) __virtio_clear_bit(vdev, VIRTIO_BALLOON_F_REPORTING); + /* + * Disable indirect descriptors to avoid memory allocation in + * virtqueue_add during page reporting. + */ + __virtio_clear_bit(vdev, VIRTIO_RING_F_INDIRECT_DESC); __virtio_clear_bit(vdev, VIRTIO_F_ACCESS_PLATFORM); return 0; } From e5725e85d68541584e64e702785ce7445486df83 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Linfeng=20Sun=C2=A0?= Date: Sat, 20 Jun 2026 18:09:59 +0800 Subject: [PATCH 044/857] vdpa_sim: fix cleanup after worker creation failure MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit vdpasim_create() leaves vdpasim->worker as an ERR_PTR when kthread_run_worker() fails. The error path then drops the device reference, which releases the partially initialized simulator. vdpasim_free() unconditionally passes the worker pointer to kthread_destroy_worker(), so the ERR_PTR is dereferenced and can trigger a general protection fault. Store the worker error, clear the pointer, and only clean up the worker when it was successfully initialized. Also make the release path tolerate partially initialized objects by guarding virtqueue and IOTLB cleanup, since the same release path can be reached from other initialization failures. I found this bug myself, though the patch was written with AI assistance. Fixes: 76acfa7bc54f ("vdpa_sim: use kthread worker") Assisted-by: OpenAI-Codex:GPT-5 Reviewed-by: Eugenio Pérez Signed-off-by: Linfeng Sun  Message-ID: <20260620100959.2070316-1-slf@hdu.edu.cn> Signed-off-by: Michael S. Tsirkin --- drivers/vdpa/vdpa_sim/vdpa_sim.c | 25 +++++++++++++++++-------- 1 file changed, 17 insertions(+), 8 deletions(-) diff --git a/drivers/vdpa/vdpa_sim/vdpa_sim.c b/drivers/vdpa/vdpa_sim/vdpa_sim.c index 4d116644851d92..c748fe45116350 100644 --- a/drivers/vdpa/vdpa_sim/vdpa_sim.c +++ b/drivers/vdpa/vdpa_sim/vdpa_sim.c @@ -233,8 +233,11 @@ struct vdpasim *vdpasim_create(struct vdpasim_dev_attr *dev_attr, kthread_init_work(&vdpasim->work, vdpasim_work_fn); vdpasim->worker = kthread_run_worker(0, "vDPA sim worker: %s", dev_attr->name); - if (IS_ERR(vdpasim->worker)) + if (IS_ERR(vdpasim->worker)) { + ret = PTR_ERR(vdpasim->worker); + vdpasim->worker = NULL; goto err_iommu; + } mutex_init(&vdpasim->mutex); spin_lock_init(&vdpasim->iommu_lock); @@ -746,18 +749,24 @@ static void vdpasim_free(struct vdpa_device *vdpa) struct vdpasim *vdpasim = vdpa_to_sim(vdpa); int i; - kthread_cancel_work_sync(&vdpasim->work); - kthread_destroy_worker(vdpasim->worker); + if (vdpasim->worker) { + kthread_cancel_work_sync(&vdpasim->work); + kthread_destroy_worker(vdpasim->worker); + } - for (i = 0; i < vdpasim->dev_attr.nvqs; i++) { - vringh_kiov_cleanup(&vdpasim->vqs[i].out_iov); - vringh_kiov_cleanup(&vdpasim->vqs[i].in_iov); + if (vdpasim->vqs) { + for (i = 0; i < vdpasim->dev_attr.nvqs; i++) { + vringh_kiov_cleanup(&vdpasim->vqs[i].out_iov); + vringh_kiov_cleanup(&vdpasim->vqs[i].in_iov); + } } vdpasim->dev_attr.free(vdpasim); - for (i = 0; i < vdpasim->dev_attr.nas; i++) - vhost_iotlb_reset(&vdpasim->iommu[i]); + if (vdpasim->iommu) { + for (i = 0; i < vdpasim->dev_attr.nas; i++) + vhost_iotlb_reset(&vdpasim->iommu[i]); + } kfree(vdpasim->iommu); kfree(vdpasim->iommu_pt); kfree(vdpasim->vqs); From 4894182d24593d24f918292160501becf9a112bf Mon Sep 17 00:00:00 2001 From: Octavian Purdila Date: Mon, 22 Jun 2026 22:27:56 +0000 Subject: [PATCH 045/857] iov_iter: export iov_iter_restore Export iov_iter_restore so that it can be used by modules. This is needed by the virtio vsock transport (which can be built as a module) to restore the msg_iter state when transmission fails. Acked-by: Stefano Garzarella Signed-off-by: Octavian Purdila Message-ID: <20260622222757.2130402-2-tavip@google.com> Signed-off-by: Michael S. Tsirkin --- lib/iov_iter.c | 1 + 1 file changed, 1 insertion(+) diff --git a/lib/iov_iter.c b/lib/iov_iter.c index c2484551a4e863..227c6ee69da820 100644 --- a/lib/iov_iter.c +++ b/lib/iov_iter.c @@ -1491,6 +1491,7 @@ void iov_iter_restore(struct iov_iter *i, struct iov_iter_state *state) i->__iov -= state->nr_segs - i->nr_segs; i->nr_segs = state->nr_segs; } +EXPORT_SYMBOL_FOR_MODULES(iov_iter_restore, "vmw_vsock_virtio_transport_common"); /* * Extract a list of contiguous pages from an ITER_FOLIOQ iterator. This does From 915d2dee0e3c1a83ec1f9cc7e69c1023776587eb Mon Sep 17 00:00:00 2001 From: Octavian Purdila Date: Mon, 22 Jun 2026 22:27:57 +0000 Subject: [PATCH 046/857] vsock/virtio: restore msg_iter on transmission failure When transmission fails in virtio_transport_send_pkt_info, the msg_iter might have been partially advanced. If we don't restore it, the next attempt to send data will use an incorrect iterator state, leading to desync and warnings like "send_pkt() returns 0, but X expected". Specifically, this can happen in the following scenario, triggered by the syzkaller repro: 1. A write-only VMA (PROT_WRITE only) is partially populated by a prior TUN write that failed with -EIO but still faulted in some pages). 2. A vsock sendmmsg call with MSG_ZEROCOPY requests transmission of a buffer from this VMA. 3. The first packet (64KB) is sent successfully because the pages are populated. 4. The second packet allocation fails because GUP fast pins the first page but GUP slow fails on the next unpopulated page due to PROT_WRITE-only permissions. 5. The iterator is advanced by the partially successful GUP (68KB total advanced: 64KB from first packet + 4KB from second), but the send loop breaks and only reports 64KB sent. This creates a 4KB desync. 6. The next retry starts with a non-zero iov_offset, disabling zerocopy and falling back to copy mode. 7. In copy mode, the transmission succeeds for the next packets but exhausts the iterator early because of the desync. 8. The final retry sees an empty iterator but zerocopy is re-enabled (offset resets). It attempts to send the remaining bytes with zerocopy but pins 0 pages, creating an empty packet. 9. The transport sends the empty packet, triggering the warning because the returned bytes (header only) do not match the expected payload size. 10. The loop continues to spin, allocating ubuf_info each time, eventually exhausting sysctl_optmem_max and returning -ENOMEM to userspace. Restore msg_iter to its original state before the packet allocation and transmission attempt if they fail. Fixes: e0718bd82e27 ("vsock: enable setting SO_ZEROCOPY") Reported-by: syzbot+28e5f3d207b14bae122a@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=28e5f3d207b14bae122a Assisted-by: gemini:gemini-3.1-pro Reviewed-by: Stefano Garzarella Signed-off-by: Octavian Purdila Message-ID: <20260622222757.2130402-3-tavip@google.com> Signed-off-by: Michael S. Tsirkin --- net/vmw_vsock/virtio_transport_common.c | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/net/vmw_vsock/virtio_transport_common.c b/net/vmw_vsock/virtio_transport_common.c index 8becad81279c86..20984f48284c13 100644 --- a/net/vmw_vsock/virtio_transport_common.c +++ b/net/vmw_vsock/virtio_transport_common.c @@ -302,6 +302,7 @@ static int virtio_transport_send_pkt_info(struct vsock_sock *vsk, u32 max_skb_len = VIRTIO_VSOCK_MAX_PKT_BUF_SIZE; u32 src_cid, src_port, dst_cid, dst_port; const struct virtio_transport *t_ops; + struct iov_iter_state msg_iter_state; struct virtio_vsock_sock *vvs; struct ubuf_info *uarg = NULL; u32 pkt_len = info->pkt_len; @@ -375,8 +376,17 @@ static int virtio_transport_send_pkt_info(struct vsock_sock *vsk, struct sk_buff *skb; size_t skb_len; + /* Save iterator state in case allocation or transmission fails + * so we can restore it and retry. + */ + if (info->msg) + iov_iter_save_state(&info->msg->msg_iter, &msg_iter_state); + skb_len = min(max_skb_len, rest_len); + /* Note: virtio_transport_alloc_skb() can advance info->msg->msg_iter + * even if it fails (e.g. partial GUP success). + */ skb = virtio_transport_alloc_skb(info, skb_len, can_zcopy, uarg, src_cid, src_port, @@ -406,6 +416,9 @@ static int virtio_transport_send_pkt_info(struct vsock_sock *vsk, break; } while (rest_len); + if (info->msg && ret < 0) + iov_iter_restore(&info->msg->msg_iter, &msg_iter_state); + virtio_transport_put_credit(vvs, rest_len); /* msg_zerocopy_realloc() initializes the ubuf_info refcnt to 1. From 93a5af4ed28ff5f1a2999892e959587369746981 Mon Sep 17 00:00:00 2001 From: Bryam Vargas Date: Mon, 22 Jun 2026 01:52:15 -0500 Subject: [PATCH 047/857] crypto: virtio - bound the akcipher result length virtio_crypto_dataq_akcipher_callback() sets the result length from the device-reported response length without bounding it to the destination buffer, which was allocated for the original request length. sg_copy_from_buffer() then reads that many bytes from the destination buffer; a backend reporting a larger length over-reads adjacent kernel heap into the caller's scatterlist (an out-of-bounds read). Clamp the reported length to the originally requested destination length. A conforming device reports no more than that, so valid results are unaffected. Fixes: a36bd0ad9fbf ("virtio-crypto: adjust dst_len at ops callback") Cc: stable@vger.kernel.org Signed-off-by: Bryam Vargas Message-ID: <20260622-b4-disp-3a2c09a8-v2-1-d1a809281db4@proton.me> Signed-off-by: Michael S. Tsirkin --- drivers/crypto/virtio/virtio_crypto_akcipher_algs.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c b/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c index d8d452cac3910b..64ea141f018cb0 100644 --- a/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c +++ b/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c @@ -88,7 +88,8 @@ static void virtio_crypto_dataq_akcipher_callback(struct virtio_crypto_request * } /* actual length may be less than dst buffer */ - akcipher_req->dst_len = len - sizeof(vc_req->status); + akcipher_req->dst_len = min_t(unsigned int, len - sizeof(vc_req->status), + akcipher_req->dst_len); sg_copy_from_buffer(akcipher_req->dst, sg_nents(akcipher_req->dst), vc_akcipher_req->dst_buf, akcipher_req->dst_len); virtio_crypto_akcipher_finalize_req(vc_akcipher_req, akcipher_req, error); From 18b83b68e65574cc1435132575ab943a086d8bec Mon Sep 17 00:00:00 2001 From: Ben Dooks Date: Mon, 22 Jun 2026 16:03:22 +0100 Subject: [PATCH 048/857] crypto: virtio - fix missing le64_to_cpu() conversions There are two cases of sending a __le64 type to a print function so fix this by adding le64_to_cpu() which fixes the following (prototype) sparse warnings: drivers/crypto/virtio/virtio_crypto_skcipher_algs.c:234:17: warning: incorrect type in argument 3 (different base types) drivers/crypto/virtio/virtio_crypto_skcipher_algs.c:234:17: expected unsigned long long drivers/crypto/virtio/virtio_crypto_skcipher_algs.c:234:17: got restricted __le64 [usertype] session_id drivers/crypto/virtio/virtio_crypto_akcipher_algs.c:196:17: warning: incorrect type in argument 3 (different base types) drivers/crypto/virtio/virtio_crypto_akcipher_algs.c:196:17: expected unsigned long long drivers/crypto/virtio/virtio_crypto_akcipher_algs.c:196:17: got restricted __le64 [usertype] session_id Signed-off-by: Ben Dooks Message-ID: <20260622150322.526375-1-ben.dooks@codethink.co.uk> Signed-off-by: Michael S. Tsirkin --- drivers/crypto/virtio/virtio_crypto_akcipher_algs.c | 3 ++- drivers/crypto/virtio/virtio_crypto_skcipher_algs.c | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c b/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c index 64ea141f018cb0..9078f22978b7be 100644 --- a/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c +++ b/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c @@ -195,7 +195,8 @@ static int virtio_crypto_alg_akcipher_close_session(struct virtio_crypto_akciphe if (ctrl_status->status != VIRTIO_CRYPTO_OK) { pr_err("virtio_crypto: Close session failed status: %u, session_id: 0x%llx\n", - ctrl_status->status, destroy_session->session_id); + ctrl_status->status, + le64_to_cpu(destroy_session->session_id)); err = -EINVAL; goto out; } diff --git a/drivers/crypto/virtio/virtio_crypto_skcipher_algs.c b/drivers/crypto/virtio/virtio_crypto_skcipher_algs.c index e82fc16cab254f..3ca441ae275982 100644 --- a/drivers/crypto/virtio/virtio_crypto_skcipher_algs.c +++ b/drivers/crypto/virtio/virtio_crypto_skcipher_algs.c @@ -232,7 +232,8 @@ static int virtio_crypto_alg_skcipher_close_session( if (ctrl_status->status != VIRTIO_CRYPTO_OK) { pr_err("virtio_crypto: Close session failed status: %u, session_id: 0x%llx\n", - ctrl_status->status, destroy_session->session_id); + ctrl_status->status, + le64_to_cpu(destroy_session->session_id)); err = -EINVAL; goto out; From 709f34c2f8c7c1ade79622b38667b37ea64825d0 Mon Sep 17 00:00:00 2001 From: Albert Esteve Date: Tue, 10 Mar 2026 09:40:46 +0100 Subject: [PATCH 049/857] virtio: Add ID for virtio media Add VIRTIO_ID_MEDIA definition for virtio-media. Signed-off-by: Albert Esteve Message-ID: <20260310-virtio-media-id-v1-1-be211bcf682b@redhat.com> Signed-off-by: Michael S. Tsirkin --- include/uapi/linux/virtio_ids.h | 1 + 1 file changed, 1 insertion(+) diff --git a/include/uapi/linux/virtio_ids.h b/include/uapi/linux/virtio_ids.h index 6c12db16faa3ad..f9056af0c6223f 100644 --- a/include/uapi/linux/virtio_ids.h +++ b/include/uapi/linux/virtio_ids.h @@ -69,6 +69,7 @@ #define VIRTIO_ID_BT 40 /* virtio bluetooth */ #define VIRTIO_ID_GPIO 41 /* virtio gpio */ #define VIRTIO_ID_SPI 45 /* virtio spi */ +#define VIRTIO_ID_MEDIA 48 /* virtio media */ /* * Virtio Transitional IDs From aa47d9df13b7aa0b4ed81ccb58f1f9a6452235ba Mon Sep 17 00:00:00 2001 From: "Denis V. Lunev" Date: Wed, 24 Jun 2026 16:08:43 +0200 Subject: [PATCH 050/857] virtio: add virtio_device_shutdown() helper The generic virtio bus .shutdown handler, virtio_dev_shutdown(), breaks and resets a device once it has established that the driver has no .shutdown of its own. A driver that does implement .shutdown, to quiesce its own activity first, still needs the same break and reset afterwards and would otherwise have to open code it. Factor the break + synchronize_cbs + reset sequence out of virtio_dev_shutdown() into an exported virtio_device_shutdown() helper so such drivers can reuse it instead of duplicating the core logic. No functional change. Signed-off-by: Denis V. Lunev Reviewed-by: David Hildenbrand (Arm) Signed-off-by: Michael S. Tsirkin Message-ID: <20260624140846.2616797-2-den@openvz.org> --- drivers/virtio/virtio.c | 41 +++++++++++++++++++++++++++-------------- include/linux/virtio.h | 1 + 2 files changed, 28 insertions(+), 14 deletions(-) diff --git a/drivers/virtio/virtio.c b/drivers/virtio/virtio.c index 299fa83be1d5f9..75bb4ffe3b8774 100644 --- a/drivers/virtio/virtio.c +++ b/drivers/virtio/virtio.c @@ -401,6 +401,32 @@ static const struct cpumask *virtio_irq_get_affinity(struct device *_d, return dev->config->get_vq_affinity(dev, irq_vec); } +/** + * virtio_device_shutdown - break and reset a device on shutdown + * @dev: the device + * + * Drivers with their own .shutdown method should quiesce their activity and + * then call this to stop the device the way the generic shutdown path does. + */ +void virtio_device_shutdown(struct virtio_device *dev) +{ + /* + * Some devices get wedged if you kick them after they are + * reset. Mark all vqs as broken to make sure we don't. + */ + virtio_break_device(dev); + /* + * Guarantee that any callback will see vq->broken as true. + */ + virtio_synchronize_cbs(dev); + /* + * As IOMMUs are reset on shutdown, this will block device access to memory. + * Some devices get wedged if this happens, so reset to make sure it does not. + */ + dev->config->reset(dev); +} +EXPORT_SYMBOL_GPL(virtio_device_shutdown); + static void virtio_dev_shutdown(struct device *_d) { struct virtio_device *dev = dev_to_virtio(_d); @@ -419,20 +445,7 @@ static void virtio_dev_shutdown(struct device *_d) return; } - /* - * Some devices get wedged if you kick them after they are - * reset. Mark all vqs as broken to make sure we don't. - */ - virtio_break_device(dev); - /* - * Guarantee that any callback will see vq->broken as true. - */ - virtio_synchronize_cbs(dev); - /* - * As IOMMUs are reset on shutdown, this will block device access to memory. - * Some devices get wedged if this happens, so reset to make sure it does not. - */ - dev->config->reset(dev); + virtio_device_shutdown(dev); } static int virtio_dev_num_vf(struct device *dev) diff --git a/include/linux/virtio.h b/include/linux/virtio.h index 93e573c565635a..f923e42cfd0110 100644 --- a/include/linux/virtio.h +++ b/include/linux/virtio.h @@ -213,6 +213,7 @@ int virtio_device_freeze(struct virtio_device *dev); int virtio_device_restore(struct virtio_device *dev); #endif void virtio_reset_device(struct virtio_device *dev); +void virtio_device_shutdown(struct virtio_device *dev); int virtio_device_reset_prepare(struct virtio_device *dev); int virtio_device_reset_done(struct virtio_device *dev); From a6f917677af3495af3871a20a1aef4dc471667ae Mon Sep 17 00:00:00 2001 From: "Denis V. Lunev" Date: Wed, 24 Jun 2026 16:08:44 +0200 Subject: [PATCH 051/857] virtio_balloon: factor out virtballoon_quiesce() virtballoon_remove() stops all of the balloon's asynchronous work (the free page reporting worker, the inflate/deflate and stats workers, the OOM notifier and the free page shrinker) before tearing the device down. A following change needs the same teardown from a .shutdown handler, so move it into a virtballoon_quiesce() helper. No functional change. Signed-off-by: Denis V. Lunev Reviewed-by: David Hildenbrand (Arm) Signed-off-by: Michael S. Tsirkin Message-ID: <20260624140846.2616797-3-den@openvz.org> --- drivers/virtio/virtio_balloon.c | 27 ++++++++++++++++++++------- 1 file changed, 20 insertions(+), 7 deletions(-) diff --git a/drivers/virtio/virtio_balloon.c b/drivers/virtio/virtio_balloon.c index 9309f3198db0e4..d11d81e6b64414 100644 --- a/drivers/virtio/virtio_balloon.c +++ b/drivers/virtio/virtio_balloon.c @@ -1109,26 +1109,39 @@ static void remove_common(struct virtio_balloon *vb) vb->vdev->config->del_vqs(vb->vdev); } -static void virtballoon_remove(struct virtio_device *vdev) +/* + * Stop all asynchronous balloon work. The device must still be alive so that + * in-flight requests can drain via the host before it is reset or freed. + */ +static void virtballoon_quiesce(struct virtio_balloon *vb) { - struct virtio_balloon *vb = vdev->priv; + struct virtio_device *vdev = vb->vdev; - if (virtio_has_feature(vb->vdev, VIRTIO_BALLOON_F_REPORTING)) + if (virtio_has_feature(vdev, VIRTIO_BALLOON_F_REPORTING)) page_reporting_unregister(&vb->pr_dev_info); - if (virtio_has_feature(vb->vdev, VIRTIO_BALLOON_F_DEFLATE_ON_OOM)) + if (virtio_has_feature(vdev, VIRTIO_BALLOON_F_DEFLATE_ON_OOM)) unregister_oom_notifier(&vb->oom_nb); - if (virtio_has_feature(vb->vdev, VIRTIO_BALLOON_F_FREE_PAGE_HINT)) + if (virtio_has_feature(vdev, VIRTIO_BALLOON_F_FREE_PAGE_HINT)) virtio_balloon_unregister_shrinker(vb); + spin_lock_irq(&vb->stop_update_lock); vb->stop_update = true; spin_unlock_irq(&vb->stop_update_lock); cancel_work_sync(&vb->update_balloon_size_work); cancel_work_sync(&vb->update_balloon_stats_work); - if (virtio_has_feature(vdev, VIRTIO_BALLOON_F_FREE_PAGE_HINT)) { + if (virtio_has_feature(vdev, VIRTIO_BALLOON_F_FREE_PAGE_HINT)) cancel_work_sync(&vb->report_free_page_work); +} + +static void virtballoon_remove(struct virtio_device *vdev) +{ + struct virtio_balloon *vb = vdev->priv; + + virtballoon_quiesce(vb); + + if (virtio_has_feature(vdev, VIRTIO_BALLOON_F_FREE_PAGE_HINT)) destroy_workqueue(vb->balloon_wq); - } remove_common(vb); mutex_destroy(&vb->balloon_lock); From 009418c31c5a4162888ab9dfb0e44f0cbb6ad64b Mon Sep 17 00:00:00 2001 From: "Denis V. Lunev" Date: Wed, 24 Jun 2026 16:08:45 +0200 Subject: [PATCH 052/857] virtio_balloon: quiesce balloon work before device shutdown Commit 8bd2fa086a04 ("virtio: break and reset virtio devices on device_shutdown()") added a generic virtio bus .shutdown handler that breaks and resets every virtio device during device_shutdown(), i.e. on reboot and kexec. virtio_balloon provides no .shutdown of its own, so that generic path runs while the balloon's asynchronous work is still armed. Once the device has been broken, virtqueue_add_inbuf() in virtballoon_free_page_report() returns -EIO and trips its WARN_ON_ONCE(). On a kernel booted with panic_on_warn that turns an ordinary reboot, for example a kexec based upgrade, into a fatal panic in the middle of device_shutdown(), so the machine never reaches the new kernel. Relaxing that single WARN_ON_ONCE() would only hide the symptom: the inflate/deflate and OOM paths do not warn, they call wait_event(vb->acked, ...) and would instead block forever on a broken queue that can no longer complete. The device has to be quiesced, not just kept quiet. Add a .shutdown handler that quiesces the balloon via the shared virtballoon_quiesce() helper while the device is still alive, and only then breaks and resets it via virtio_device_shutdown(). Unlike virtballoon_remove() the balloon workqueue is not destroyed, as shutdown does not free the device and cancel_work_sync() together with stop_update already prevent any further work from being queued. Fixes: 8bd2fa086a04 ("virtio: break and reset virtio devices on device_shutdown()") Signed-off-by: Denis V. Lunev Reviewed-by: David Hildenbrand (Arm) Signed-off-by: Michael S. Tsirkin Message-ID: <20260624140846.2616797-4-den@openvz.org> --- drivers/virtio/virtio_balloon.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/drivers/virtio/virtio_balloon.c b/drivers/virtio/virtio_balloon.c index d11d81e6b64414..47b0fc63a65cd4 100644 --- a/drivers/virtio/virtio_balloon.c +++ b/drivers/virtio/virtio_balloon.c @@ -1148,6 +1148,12 @@ static void virtballoon_remove(struct virtio_device *vdev) kfree(vb); } +static void virtballoon_shutdown(struct virtio_device *vdev) +{ + virtballoon_quiesce(vdev->priv); + virtio_device_shutdown(vdev); +} + #ifdef CONFIG_PM_SLEEP static int virtballoon_freeze(struct virtio_device *vdev) { @@ -1220,6 +1226,7 @@ static struct virtio_driver virtio_balloon_driver = { .validate = virtballoon_validate, .probe = virtballoon_probe, .remove = virtballoon_remove, + .shutdown = virtballoon_shutdown, .config_changed = virtballoon_changed, #ifdef CONFIG_PM_SLEEP .freeze = virtballoon_freeze, From bb72524bca3a2aa2a422413b5c1c9b4cda0c843a Mon Sep 17 00:00:00 2001 From: "Denis V. Lunev" Date: Wed, 24 Jun 2026 16:08:46 +0200 Subject: [PATCH 053/857] virtio_balloon: warn on failed buffer add in tell_host() tell_host() ignores the return value of virtqueue_add_outbuf() and goes on to kick the queue and wait_event() for the host's ack. The comment claims "We should always be able to add one buffer to an empty queue", but that does not hold once the virtqueue has been broken (e.g. on device shutdown): the add then fails with -EIO and the following wait_event() would block forever on a buffer the host can never return. Warn and bail out on failure, mirroring virtballoon_free_page_report(). Suggested-by: David Hildenbrand Signed-off-by: Denis V. Lunev Signed-off-by: Michael S. Tsirkin Message-ID: <20260624140846.2616797-5-den@openvz.org> --- drivers/virtio/virtio_balloon.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/drivers/virtio/virtio_balloon.c b/drivers/virtio/virtio_balloon.c index 47b0fc63a65cd4..1c55e3c3e14c54 100644 --- a/drivers/virtio/virtio_balloon.c +++ b/drivers/virtio/virtio_balloon.c @@ -185,16 +185,18 @@ static void tell_host(struct virtio_balloon *vb, struct virtqueue *vq) { struct scatterlist sg; unsigned int len; + int err; sg_init_one(&sg, vb->pfns, sizeof(vb->pfns[0]) * vb->num_pfns); /* We should always be able to add one buffer to an empty queue. */ - virtqueue_add_outbuf(vq, &sg, 1, vb, GFP_KERNEL); + err = virtqueue_add_outbuf(vq, &sg, 1, vb, GFP_KERNEL); + if (WARN_ON_ONCE(err)) + return; virtqueue_kick(vq); /* When host has read buffer, this completes via balloon_ack */ wait_event(vb->acked, virtqueue_get_buf(vq, &len)); - } static int virtballoon_free_page_report(struct page_reporting_dev_info *pr_dev_info, From 46d447d703a8ce3f46352c71c7b281c99e416bc3 Mon Sep 17 00:00:00 2001 From: "Denis V. Lunev" Date: Wed, 24 Jun 2026 17:40:01 +0200 Subject: [PATCH 054/857] virtio_balloon: warn on failed buffer add in stats_handle_request() Like tell_host(), stats_handle_request() ignores the return value of virtqueue_add_outbuf() and kicks the queue regardless. The same "we should always be able to add one buffer to an empty queue" assumption does not hold once the virtqueue has been broken (e.g. on device shutdown), where the add fails with -EIO. Unlike tell_host() it does not wait_event() afterwards so it cannot hang, but it still kicks a queue with nothing queued. Warn and bail out on failure, mirroring tell_host() and virtballoon_free_page_report(). Suggested-by: David Hildenbrand Signed-off-by: Denis V. Lunev Reviewed-by: David Hildenbrand (Arm) Signed-off-by: Michael S. Tsirkin Message-ID: <20260624154001.2733242-1-den@openvz.org> --- drivers/virtio/virtio_balloon.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/drivers/virtio/virtio_balloon.c b/drivers/virtio/virtio_balloon.c index 1c55e3c3e14c54..5cc73dd1363c13 100644 --- a/drivers/virtio/virtio_balloon.c +++ b/drivers/virtio/virtio_balloon.c @@ -446,6 +446,7 @@ static void stats_handle_request(struct virtio_balloon *vb) struct virtqueue *vq; struct scatterlist sg; unsigned int len, num_stats; + int err; num_stats = update_balloon_stats(vb); @@ -453,7 +454,9 @@ static void stats_handle_request(struct virtio_balloon *vb) if (!virtqueue_get_buf(vq, &len)) return; sg_init_one(&sg, vb->stats, sizeof(vb->stats[0]) * num_stats); - virtqueue_add_outbuf(vq, &sg, 1, vb, GFP_KERNEL); + err = virtqueue_add_outbuf(vq, &sg, 1, vb, GFP_KERNEL); + if (WARN_ON_ONCE(err)) + return; virtqueue_kick(vq); } From d914650e946d137fb58dbb75fe476de295595097 Mon Sep 17 00:00:00 2001 From: Yufeng Wang Date: Fri, 26 Jun 2026 15:04:38 +0800 Subject: [PATCH 055/857] vhost/net: fix clear_user start address in VHOST_GET_FEATURES_ARRAY MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The clear_user() call in VHOST_GET_FEATURES_ARRAY incorrectly starts at argp, which is the beginning of the features array, overwriting the data just written by copy_to_user(). It should start after the copied elements at argp + copied * sizeof(u64) to only zero the trailing unused space. Use size_mul() for both the offset and length calculations so the arithmetic stays consistent with the surrounding code and remains overflow-safe. Fixes: 333c515d1896 ("vhost-net: allow configuring extended features") Signed-off-by: Yufeng Wang Acked-by: Eugenio Pérez Signed-off-by: Michael S. Tsirkin Message-ID: <20260626070438.59149-1-r4o5m6e8o@163.com> --- drivers/vhost/net.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/drivers/vhost/net.c b/drivers/vhost/net.c index 6949b704166d58..38d9c184082d0f 100644 --- a/drivers/vhost/net.c +++ b/drivers/vhost/net.c @@ -1777,7 +1777,8 @@ static long vhost_net_ioctl(struct file *f, unsigned int ioctl, return -EFAULT; /* Zero the trailing space provided by user-space, if any */ - if (clear_user(argp, size_mul(count - copied, sizeof(u64)))) + if (clear_user(argp + size_mul(copied, sizeof(u64)), + size_mul(count - copied, sizeof(u64)))) return -EFAULT; return 0; case VHOST_SET_FEATURES_ARRAY: From 1263264b5af268dba19f27520a7fb135a5c8f07e Mon Sep 17 00:00:00 2001 From: Yichong Chen Date: Thu, 18 Jun 2026 18:02:54 +0800 Subject: [PATCH 056/857] tools/virtio: Remove unsupported --batch option from vhost_net_test MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit vhost_net_test has --batch in longopts, but not in help. The parser never handles 'b', so --batch hits assert(0). Remove the unsupported option. Signed-off-by: Yichong Chen Acked-by: Eugenio Pérez Signed-off-by: Michael S. Tsirkin Message-ID: --- tools/virtio/vhost_net_test.c | 5 ----- 1 file changed, 5 deletions(-) diff --git a/tools/virtio/vhost_net_test.c b/tools/virtio/vhost_net_test.c index 389d99a6d7c76d..566e15420bb654 100644 --- a/tools/virtio/vhost_net_test.c +++ b/tools/virtio/vhost_net_test.c @@ -450,11 +450,6 @@ static const struct option longopts[] = { .val = 'n', .has_arg = required_argument, }, - { - .name = "batch", - .val = 'b', - .has_arg = required_argument, - }, { } }; From db09a0f62f888da11bca2f4aba65a7b60b4eb6f5 Mon Sep 17 00:00:00 2001 From: Xiong Weimin Date: Fri, 26 Jun 2026 10:05:44 +0800 Subject: [PATCH 057/857] vdpa_sim: clear pending_kick on device reset vdpasim_kick_vq() sets pending_kick when a virtqueue is kicked while the device is suspended (!running but DRIVER_OK). vdpasim_resume() later replays kicks for all virtqueues when pending_kick is set. vdpasim_do_reset() clears running and status but leaves pending_kick unchanged. If a kick is deferred during suspend and the device is reset before resume, a later resume can spuriously kick every virtqueue even though no new work was queued after reset. Clear pending_kick in vdpasim_do_reset() together with the other device state that must not survive a reset. Tested-on: openEuler VM (6.16.8, /usr/src/linux-6.16.8) Tested-by: Xiong Weimin Signed-off-by: Xiong Weimin Signed-off-by: Michael S. Tsirkin Message-ID: <20260626020545.607600-2-15927021679@163.com> --- drivers/vdpa/vdpa_sim/vdpa_sim.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/vdpa/vdpa_sim/vdpa_sim.c b/drivers/vdpa/vdpa_sim/vdpa_sim.c index c748fe45116350..aa3741b8778de7 100644 --- a/drivers/vdpa/vdpa_sim/vdpa_sim.c +++ b/drivers/vdpa/vdpa_sim/vdpa_sim.c @@ -161,6 +161,7 @@ static void vdpasim_do_reset(struct vdpasim *vdpasim, u32 flags) } vdpasim->running = false; + vdpasim->pending_kick = false; spin_unlock(&vdpasim->iommu_lock); vdpasim->features = 0; From e50521dcc5864356ae1933753bc96887eda160c7 Mon Sep 17 00:00:00 2001 From: Xiong Weimin Date: Fri, 26 Jun 2026 10:05:45 +0800 Subject: [PATCH 058/857] vdpa_sim: hold iommu_lock across dma_unmap passthrough transition vdpasim_dma_map() updates the IOTLB and the passthrough (iommu_pt) state under iommu_lock. vdpasim_dma_unmap() clears iommu_pt and resets the IOTLB before taking iommu_lock, then deletes the mapping while holding the lock. A concurrent dma_map(), dma_unmap(), or reset path that also touches the same address space can therefore observe or modify the IOTLB and iommu_pt state without consistent locking. Perform the passthrough transition and range deletion under the same iommu_lock scope, matching dma_map(). Tested-on: openEuler VM (6.16.8, /usr/src/linux-6.16.8) Tested-by: Xiong Weimin Signed-off-by: Xiong Weimin Signed-off-by: Michael S. Tsirkin Message-ID: <20260626020545.607600-3-15927021679@163.com> --- drivers/vdpa/vdpa_sim/vdpa_sim.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/drivers/vdpa/vdpa_sim/vdpa_sim.c b/drivers/vdpa/vdpa_sim/vdpa_sim.c index aa3741b8778de7..cfa88a60a38fd4 100644 --- a/drivers/vdpa/vdpa_sim/vdpa_sim.c +++ b/drivers/vdpa/vdpa_sim/vdpa_sim.c @@ -733,12 +733,11 @@ static int vdpasim_dma_unmap(struct vdpa_device *vdpa, unsigned int asid, if (asid >= vdpasim->dev_attr.nas) return -EINVAL; + spin_lock(&vdpasim->iommu_lock); if (vdpasim->iommu_pt[asid]) { vhost_iotlb_reset(&vdpasim->iommu[asid]); vdpasim->iommu_pt[asid] = false; } - - spin_lock(&vdpasim->iommu_lock); vhost_iotlb_del_range(&vdpasim->iommu[asid], iova, iova + size - 1); spin_unlock(&vdpasim->iommu_lock); From e62363cd022101b6e60507837e0060d158d22f84 Mon Sep 17 00:00:00 2001 From: Li RongQing Date: Mon, 29 Jun 2026 11:31:46 +0800 Subject: [PATCH 059/857] virtio_dma_buf: fix typo in kdoc comment: get_uid -> get_uuid The @get_uid tag in the virtio_dma_buf_ops kdoc comment is a typo; the actual field name is get_uuid. Fixes: a0308938ec81 ("virtio: add dma-buf support for exported objects") Signed-off-by: Li RongQing Signed-off-by: Michael S. Tsirkin Message-ID: <20260629033146.2209-1-lirongqing@baidu.com> --- include/linux/virtio_dma_buf.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/linux/virtio_dma_buf.h b/include/linux/virtio_dma_buf.h index a2fdf217ac6222..545ac5f17a54c9 100644 --- a/include/linux/virtio_dma_buf.h +++ b/include/linux/virtio_dma_buf.h @@ -17,7 +17,7 @@ * @ops: the base dma_buf_ops. ops.attach MUST be virtio_dma_buf_attach. * @device_attach: [optional] callback invoked by virtio_dma_buf_attach during * all attach operations. - * @get_uid: [required] callback to get the uuid of the exported object. + * @get_uuid: [required] callback to get the uuid of the exported object. */ struct virtio_dma_buf_ops { struct dma_buf_ops ops; From c2ed2f2a6415a84673485e5125be08ed6f732489 Mon Sep 17 00:00:00 2001 From: Li RongQing Date: Tue, 30 Jun 2026 12:59:52 +0800 Subject: [PATCH 060/857] virtio_mem: fix hardcoded 'vm' variable in bbm iteration macros virtio_mem_bbm_for_each_bb() and virtio_mem_bbm_for_each_bb_rev() accept a '_vm' parameter to allow callers to pass any variable name referring to the virtio_mem instance. However, the 'for' loop initializer and part of the loop condition use the bare name 'vm' instead of the macro parameter '_vm'. Fix by replacing all bare 'vm->' references inside the macros with the '_vm' parameter, and wrap in parentheses following kernel macro conventions. Signed-off-by: Li RongQing Acked-by: David Hildenbrand (Arm) Signed-off-by: Michael S. Tsirkin Message-ID: <20260630045952.2188-1-lirongqing@baidu.com> --- drivers/virtio/virtio_mem.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/drivers/virtio/virtio_mem.c b/drivers/virtio/virtio_mem.c index 11c4415015829d..82a285c0926de8 100644 --- a/drivers/virtio/virtio_mem.c +++ b/drivers/virtio/virtio_mem.c @@ -423,14 +423,14 @@ static int virtio_mem_bbm_bb_states_prepare_next_bb(struct virtio_mem *vm) } #define virtio_mem_bbm_for_each_bb(_vm, _bb_id, _state) \ - for (_bb_id = vm->bbm.first_bb_id; \ - _bb_id < vm->bbm.next_bb_id && _vm->bbm.bb_count[_state]; \ + for (_bb_id = (_vm)->bbm.first_bb_id; \ + _bb_id < (_vm)->bbm.next_bb_id && (_vm)->bbm.bb_count[_state]; \ _bb_id++) \ if (virtio_mem_bbm_get_bb_state(_vm, _bb_id) == _state) #define virtio_mem_bbm_for_each_bb_rev(_vm, _bb_id, _state) \ - for (_bb_id = vm->bbm.next_bb_id - 1; \ - _bb_id >= vm->bbm.first_bb_id && _vm->bbm.bb_count[_state]; \ + for (_bb_id = (_vm)->bbm.next_bb_id - 1; \ + _bb_id >= (_vm)->bbm.first_bb_id && (_vm)->bbm.bb_count[_state]; \ _bb_id--) \ if (virtio_mem_bbm_get_bb_state(_vm, _bb_id) == _state) From d57a0ea6b521435bb688b067a0f112df1d63191b Mon Sep 17 00:00:00 2001 From: Li RongQing Date: Mon, 29 Jun 2026 11:35:38 +0800 Subject: [PATCH 061/857] virtio_pci: fix wrong queue index for admin vq in intx path In vp_find_vqs_intx(), the admin vq was set up using the local queue_idx counter instead of avq->vq_index (the actual queue index obtained from the device). This differs from vp_find_vqs_msix() which correctly uses avq->vq_index. Using the wrong index causes the admin virtqueue to be mapped to an incorrect hardware queue. Fix it by using avq->vq_index consistent with the msix path. Fixes: af22bbe1f4a5 ("virtio: create admin queues alongside other virtqueues") Signed-off-by: Li RongQing Message-ID: <20260629033538.2476-1-lirongqing@baidu.com> Signed-off-by: Michael S. Tsirkin --- drivers/virtio/virtio_pci_common.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/virtio/virtio_pci_common.c b/drivers/virtio/virtio_pci_common.c index 164f480b18a6fe..10371ecbc054cb 100644 --- a/drivers/virtio/virtio_pci_common.c +++ b/drivers/virtio/virtio_pci_common.c @@ -499,7 +499,7 @@ static int vp_find_vqs_intx(struct virtio_device *vdev, unsigned int nvqs, if (!avq_num) return 0; sprintf(avq->name, "avq.%u", avq->vq_index); - vq = vp_setup_vq(vdev, queue_idx++, vp_modern_avq_done, avq->name, + vq = vp_setup_vq(vdev, avq->vq_index, vp_modern_avq_done, avq->name, false, VIRTIO_MSI_NO_VECTOR, &vp_dev->admin_vq.info); if (IS_ERR(vq)) { From 1d04623fb8d96fa0c2539239faf9c394d8cdf7c0 Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:26 +0800 Subject: [PATCH 062/857] nvdimm: preserve flush callback -ENOMEM nvdimm_flush() maps provider flush failures to -EIO. Keep that default because provider callbacks can report host-side or backend failures that should remain generic I/O errors to the guest. Guest-side allocation failures should not be reported as I/O errors. In the virtio-pmem path, the flush request allocation can fail with -ENOMEM before any request is submitted to the host. Mapping that to -EIO makes resource pressure look like media failure. Preserve -ENOMEM from provider callbacks and continue to map other non-zero provider failures to -EIO. The generic flush path still returns 0, and pmem_submit_bio() already converts errno values to block status for bio completion. Suggested-by: Pankaj Gupta Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-2-me@linux.beauty> --- drivers/nvdimm/region_devs.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/drivers/nvdimm/region_devs.c b/drivers/nvdimm/region_devs.c index 5e079d61cbaa32..39669eb4ce34e9 100644 --- a/drivers/nvdimm/region_devs.c +++ b/drivers/nvdimm/region_devs.c @@ -1093,7 +1093,8 @@ int nvdimm_flush(struct nd_region *nd_region, struct bio *bio) if (!nd_region->flush) rc = generic_nvdimm_flush(nd_region); else { - if (nd_region->flush(nd_region, bio)) + rc = nd_region->flush(nd_region, bio); + if (rc && rc != -ENOMEM) rc = -EIO; } From 42c5c8634aae258eb1a1cce8429de581d7046079 Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:27 +0800 Subject: [PATCH 063/857] nvdimm: pmem: keep PREFLUSH before data writes pmem_submit_bio() records a REQ_PREFLUSH error, but continues to copy the bio data and can later overwrite the error with a successful REQ_FUA flush. That lets data writes run after a failed preflush and can complete the bio successfully despite the failed ordering barrier. Run the REQ_PREFLUSH flush synchronously before touching the bio data and complete the bio with the flush error if it fails. Keep asynchronous flush chaining for REQ_FUA. At that point, data copy has completed and the parent bio can wait for the chained flush bio. Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-3-me@linux.beauty> --- drivers/nvdimm/pmem.c | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/drivers/nvdimm/pmem.c b/drivers/nvdimm/pmem.c index 92c67fbbc1c85d..05d3de33e2706e 100644 --- a/drivers/nvdimm/pmem.c +++ b/drivers/nvdimm/pmem.c @@ -208,8 +208,14 @@ static void pmem_submit_bio(struct bio *bio) struct pmem_device *pmem = bio->bi_bdev->bd_disk->private_data; struct nd_region *nd_region = to_region(pmem); - if (bio->bi_opf & REQ_PREFLUSH) - ret = nvdimm_flush(nd_region, bio); + if (bio->bi_opf & REQ_PREFLUSH) { + ret = nvdimm_flush(nd_region, NULL); + if (ret) { + bio->bi_status = errno_to_blk_status(ret); + bio_endio(bio); + return; + } + } do_acct = blk_queue_io_stat(bio->bi_bdev->bd_disk->queue); if (do_acct) @@ -229,7 +235,7 @@ static void pmem_submit_bio(struct bio *bio) if (do_acct) bio_end_io_acct(bio, start); - if (bio->bi_opf & REQ_FUA) + if ((bio->bi_opf & REQ_FUA) && !bio->bi_status) ret = nvdimm_flush(nd_region, bio); if (ret) From 0957511b09c37fd52400741ddce0423ff4026116 Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:28 +0800 Subject: [PATCH 064/857] nvdimm: pmem: guard data loop for dataless bios pmem_submit_bio() handles flush-only bios before and after the data loop. Keep dataless bios out of bio_for_each_segment() so the data path only walks bios that actually carry bvec data. Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-4-me@linux.beauty> --- drivers/nvdimm/pmem.c | 36 +++++++++++++++++++++--------------- 1 file changed, 21 insertions(+), 15 deletions(-) diff --git a/drivers/nvdimm/pmem.c b/drivers/nvdimm/pmem.c index 05d3de33e2706e..82ee1ddb3a4450 100644 --- a/drivers/nvdimm/pmem.c +++ b/drivers/nvdimm/pmem.c @@ -217,23 +217,29 @@ static void pmem_submit_bio(struct bio *bio) } } - do_acct = blk_queue_io_stat(bio->bi_bdev->bd_disk->queue); - if (do_acct) - start = bio_start_io_acct(bio); - bio_for_each_segment(bvec, bio, iter) { - if (op_is_write(bio_op(bio))) - rc = pmem_do_write(pmem, bvec.bv_page, bvec.bv_offset, - iter.bi_sector, bvec.bv_len); - else - rc = pmem_do_read(pmem, bvec.bv_page, bvec.bv_offset, - iter.bi_sector, bvec.bv_len); - if (rc) { - bio->bi_status = rc; - break; + if (bio_has_data(bio)) { + do_acct = blk_queue_io_stat(bio->bi_bdev->bd_disk->queue); + if (do_acct) + start = bio_start_io_acct(bio); + bio_for_each_segment(bvec, bio, iter) { + if (op_is_write(bio_op(bio))) + rc = pmem_do_write(pmem, bvec.bv_page, + bvec.bv_offset, + iter.bi_sector, + bvec.bv_len); + else + rc = pmem_do_read(pmem, bvec.bv_page, + bvec.bv_offset, + iter.bi_sector, + bvec.bv_len); + if (rc) { + bio->bi_status = rc; + break; + } } + if (do_acct) + bio_end_io_acct(bio, start); } - if (do_acct) - bio_end_io_acct(bio, start); if ((bio->bi_opf & REQ_FUA) && !bio->bi_status) ret = nvdimm_flush(nd_region, bio); From d330252c32cfaf8ace0d7f93aaa25f0c4f3698ca Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:29 +0800 Subject: [PATCH 065/857] nvdimm: virtio_pmem: stop allocating child flush bio pmem_submit_bio() passes the parent bio to nvdimm_flush() for REQ_FUA. For virtio-pmem this makes async_pmem_flush() allocate and submit a child PREFLUSH bio chained to the parent. That child allocation is in the block submit path. Making it blocking with GFP_NOIO can consume the same global bio mempool that submit_bio() uses, while making it GFP_ATOMIC can fail under pressure. A forced failure of the child allocation produced: virtio_pmem: forcing child bio allocation failure for test Buffer I/O error on dev pmem0, logical block 0, lost sync page write EXT4-fs (pmem0): I/O error while writing superblock EXT4-fs (pmem0): mount failed Avoid the child bio without turning REQ_FUA into a synchronous submit-path wait. Let provider flush callbacks return NVDIMM_FLUSH_ASYNC after taking ownership of parent bio completion. pmem_submit_bio() returns in that case, and virtio-pmem queues an ordered WQ_MEM_RECLAIM work item that runs the existing host flush path and completes the parent bio. This keeps the asynchronous completion model of the child-bio path while removing the child bio allocation from the submit path. Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-5-me@linux.beauty> --- drivers/nvdimm/nd_virtio.c | 54 +++++++++++++++++++++++++----------- drivers/nvdimm/pmem.c | 5 +++- drivers/nvdimm/region_devs.c | 2 ++ drivers/nvdimm/virtio_pmem.c | 17 +++++++++++- drivers/nvdimm/virtio_pmem.h | 4 +++ include/linux/libnvdimm.h | 9 ++++++ 6 files changed, 73 insertions(+), 18 deletions(-) diff --git a/drivers/nvdimm/nd_virtio.c b/drivers/nvdimm/nd_virtio.c index 4176046627beb3..8e16b7780be1a3 100644 --- a/drivers/nvdimm/nd_virtio.c +++ b/drivers/nvdimm/nd_virtio.c @@ -9,6 +9,12 @@ #include "virtio_pmem.h" #include "nd.h" +struct virtio_pmem_flush_work { + struct work_struct work; + struct nd_region *nd_region; + struct bio *bio; +}; + /* The interrupt handler */ void virtio_pmem_host_ack(struct virtqueue *vq) { @@ -107,30 +113,46 @@ static int virtio_pmem_flush(struct nd_region *nd_region) return err; }; +static void virtio_pmem_flush_work(struct work_struct *work) +{ + struct virtio_pmem_flush_work *flush; + int err; + + flush = container_of(work, struct virtio_pmem_flush_work, work); + err = virtio_pmem_flush(flush->nd_region); + if (err > 0) + err = -EIO; + if (err) + flush->bio->bi_status = errno_to_blk_status(err); + bio_endio(flush->bio); + kfree(flush); +} + /* The asynchronous flush callback function */ int async_pmem_flush(struct nd_region *nd_region, struct bio *bio) { - /* - * Create child bio for asynchronous flush and chain with - * parent bio. Otherwise directly call nd_region flush. - */ - if (bio && bio->bi_iter.bi_sector != -1) { - struct bio *child = bio_alloc(bio->bi_bdev, 0, - REQ_OP_WRITE | REQ_PREFLUSH, - GFP_ATOMIC); + struct virtio_device *vdev = nd_region->provider_data; + struct virtio_pmem *vpmem = vdev->priv; + struct virtio_pmem_flush_work *flush; + int err; - if (!child) + if (bio && bio->bi_iter.bi_sector != -1) { + flush = kmalloc_obj(*flush, GFP_NOIO); + if (!flush) return -ENOMEM; - bio_clone_blkg_association(child, bio); - child->bi_iter.bi_sector = -1; - bio_chain(child, bio); - submit_bio(child); - return 0; + + INIT_WORK(&flush->work, virtio_pmem_flush_work); + flush->nd_region = nd_region; + flush->bio = bio; + queue_work(vpmem->flush_wq, &flush->work); + return NVDIMM_FLUSH_ASYNC; } - if (virtio_pmem_flush(nd_region)) + + err = virtio_pmem_flush(nd_region); + if (err > 0) return -EIO; - return 0; + return err; }; EXPORT_SYMBOL_GPL(async_pmem_flush); MODULE_DESCRIPTION("Virtio Persistent Memory Driver"); diff --git a/drivers/nvdimm/pmem.c b/drivers/nvdimm/pmem.c index 82ee1ddb3a4450..30a51c365ce8ba 100644 --- a/drivers/nvdimm/pmem.c +++ b/drivers/nvdimm/pmem.c @@ -241,8 +241,11 @@ static void pmem_submit_bio(struct bio *bio) bio_end_io_acct(bio, start); } - if ((bio->bi_opf & REQ_FUA) && !bio->bi_status) + if ((bio->bi_opf & REQ_FUA) && !bio->bi_status) { ret = nvdimm_flush(nd_region, bio); + if (ret == NVDIMM_FLUSH_ASYNC) + return; + } if (ret) bio->bi_status = errno_to_blk_status(ret); diff --git a/drivers/nvdimm/region_devs.c b/drivers/nvdimm/region_devs.c index 39669eb4ce34e9..24f42b4650ba66 100644 --- a/drivers/nvdimm/region_devs.c +++ b/drivers/nvdimm/region_devs.c @@ -1094,6 +1094,8 @@ int nvdimm_flush(struct nd_region *nd_region, struct bio *bio) rc = generic_nvdimm_flush(nd_region); else { rc = nd_region->flush(nd_region, bio); + if (rc > 0) + return rc; if (rc && rc != -ENOMEM) rc = -EIO; } diff --git a/drivers/nvdimm/virtio_pmem.c b/drivers/nvdimm/virtio_pmem.c index 77b1966619059c..9cf822a6c0c38e 100644 --- a/drivers/nvdimm/virtio_pmem.c +++ b/drivers/nvdimm/virtio_pmem.c @@ -67,10 +67,17 @@ static int virtio_pmem_probe(struct virtio_device *vdev) mutex_init(&vpmem->flush_lock); vpmem->vdev = vdev; vdev->priv = vpmem; + vpmem->flush_wq = alloc_ordered_workqueue("virtio-pmem-flush", + WQ_MEM_RECLAIM); + if (!vpmem->flush_wq) { + err = -ENOMEM; + goto out_err; + } + err = init_vq(vpmem); if (err) { dev_err(&vdev->dev, "failed to initialize virtio pmem vq's\n"); - goto out_err; + goto out_wq; } if (virtio_has_feature(vdev, VIRTIO_PMEM_F_SHMEM_REGION)) { @@ -131,6 +138,8 @@ static int virtio_pmem_probe(struct virtio_device *vdev) nvdimm_bus_unregister(vpmem->nvdimm_bus); out_vq: vdev->config->del_vqs(vdev); +out_wq: + destroy_workqueue(vpmem->flush_wq); out_err: return err; } @@ -138,14 +147,20 @@ static int virtio_pmem_probe(struct virtio_device *vdev) static void virtio_pmem_remove(struct virtio_device *vdev) { struct nvdimm_bus *nvdimm_bus = dev_get_drvdata(&vdev->dev); + struct virtio_pmem *vpmem = vdev->priv; nvdimm_bus_unregister(nvdimm_bus); + drain_workqueue(vpmem->flush_wq); vdev->config->del_vqs(vdev); virtio_reset_device(vdev); + destroy_workqueue(vpmem->flush_wq); } static int virtio_pmem_freeze(struct virtio_device *vdev) { + struct virtio_pmem *vpmem = vdev->priv; + + drain_workqueue(vpmem->flush_wq); vdev->config->del_vqs(vdev); virtio_reset_device(vdev); diff --git a/drivers/nvdimm/virtio_pmem.h b/drivers/nvdimm/virtio_pmem.h index f72cf17f9518fb..e6dfc10ce0762a 100644 --- a/drivers/nvdimm/virtio_pmem.h +++ b/drivers/nvdimm/virtio_pmem.h @@ -15,6 +15,7 @@ #include #include #include +#include struct virtio_pmem_request { struct virtio_pmem_req req; @@ -39,6 +40,9 @@ struct virtio_pmem { /* Serialize flush requests to the device. */ struct mutex flush_lock; + /* Complete asynchronous FUA flushes outside the submit path. */ + struct workqueue_struct *flush_wq; + /* nvdimm bus registers virtio pmem device */ struct nvdimm_bus *nvdimm_bus; struct nvdimm_bus_descriptor nd_desc; diff --git a/include/linux/libnvdimm.h b/include/linux/libnvdimm.h index 28f086c4a1873c..d929d83abf3be1 100644 --- a/include/linux/libnvdimm.h +++ b/include/linux/libnvdimm.h @@ -126,6 +126,15 @@ struct nd_mapping_desc { struct bio; struct resource; struct nd_region; + +/* + * Provider flush callback return values: + * 0: flush completed synchronously + * <0: flush failed + * >0: flush completion was queued and @bio will be completed later + */ +#define NVDIMM_FLUSH_ASYNC 1 + struct nd_region_desc { struct resource *res; struct nd_mapping_desc *mapping; From 5ba542f94526fcc38efa4dff7cde93caa0ff5e0e Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:30 +0800 Subject: [PATCH 066/857] nvdimm: virtio_pmem: use GFP_NOIO for flush requests virtio_pmem_flush() can run from pmem_submit_bio() while filesystem IO is waiting on the flush completion. The request object allocation can sleep, but it should not enter filesystem or block IO reclaim from this flush path. Use GFP_NOIO for the request allocation. The virtqueue descriptor allocation still uses GFP_ATOMIC because it runs under pmem_lock. Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-6-me@linux.beauty> --- drivers/nvdimm/nd_virtio.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/nvdimm/nd_virtio.c b/drivers/nvdimm/nd_virtio.c index 8e16b7780be1a3..a35044afddf345 100644 --- a/drivers/nvdimm/nd_virtio.c +++ b/drivers/nvdimm/nd_virtio.c @@ -61,7 +61,7 @@ static int virtio_pmem_flush(struct nd_region *nd_region) return -EIO; } - req_data = kmalloc_obj(*req_data); + req_data = kmalloc_obj(*req_data, GFP_NOIO); if (!req_data) return -ENOMEM; From 5915b4ca83967fd6daf521c8599acefe62c0a80b Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:31 +0800 Subject: [PATCH 067/857] nvdimm: virtio_pmem: always wake -ENOSPC waiters virtio_pmem_host_ack() reclaims virtqueue descriptors with virtqueue_get_buf(). The -ENOSPC waiter wakeup is tied to completing the returned token. If token completion is skipped for any reason, reclaimed descriptors may not wake a waiter and the submitter may sleep forever waiting for a free slot. Always wake one -ENOSPC waiter for each virtqueue completion before touching the returned token. Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-7-me@linux.beauty> --- drivers/nvdimm/nd_virtio.c | 25 ++++++++++++++++--------- 1 file changed, 16 insertions(+), 9 deletions(-) diff --git a/drivers/nvdimm/nd_virtio.c b/drivers/nvdimm/nd_virtio.c index a35044afddf345..fcb26a595d7c66 100644 --- a/drivers/nvdimm/nd_virtio.c +++ b/drivers/nvdimm/nd_virtio.c @@ -15,26 +15,33 @@ struct virtio_pmem_flush_work { struct bio *bio; }; +static void virtio_pmem_wake_one_waiter(struct virtio_pmem *vpmem) +{ + struct virtio_pmem_request *req_buf; + + if (list_empty(&vpmem->req_list)) + return; + + req_buf = list_first_entry(&vpmem->req_list, + struct virtio_pmem_request, list); + req_buf->wq_buf_avail = true; + wake_up(&req_buf->wq_buf); + list_del(&req_buf->list); +} + /* The interrupt handler */ void virtio_pmem_host_ack(struct virtqueue *vq) { struct virtio_pmem *vpmem = vq->vdev->priv; - struct virtio_pmem_request *req_data, *req_buf; + struct virtio_pmem_request *req_data; unsigned long flags; unsigned int len; spin_lock_irqsave(&vpmem->pmem_lock, flags); while ((req_data = virtqueue_get_buf(vq, &len)) != NULL) { + virtio_pmem_wake_one_waiter(vpmem); req_data->done = true; wake_up(&req_data->host_acked); - - if (!list_empty(&vpmem->req_list)) { - req_buf = list_first_entry(&vpmem->req_list, - struct virtio_pmem_request, list); - req_buf->wq_buf_avail = true; - wake_up(&req_buf->wq_buf); - list_del(&req_buf->list); - } } spin_unlock_irqrestore(&vpmem->pmem_lock, flags); } From be595885fc537a0ac0cb83538593cccfaf7f7b71 Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:32 +0800 Subject: [PATCH 068/857] nvdimm: virtio_pmem: use READ_ONCE()/WRITE_ONCE() for wait flags Use READ_ONCE()/WRITE_ONCE() for the wait_event() flags (done and wq_buf_avail). They are observed by waiters without pmem_lock, so make the accesses explicit single loads/stores and avoid compiler reordering/caching across the wait/wake paths. Acked-by: Pankaj Gupta Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-8-me@linux.beauty> --- drivers/nvdimm/nd_virtio.c | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/drivers/nvdimm/nd_virtio.c b/drivers/nvdimm/nd_virtio.c index fcb26a595d7c66..8c0d4347938a1c 100644 --- a/drivers/nvdimm/nd_virtio.c +++ b/drivers/nvdimm/nd_virtio.c @@ -24,9 +24,9 @@ static void virtio_pmem_wake_one_waiter(struct virtio_pmem *vpmem) req_buf = list_first_entry(&vpmem->req_list, struct virtio_pmem_request, list); - req_buf->wq_buf_avail = true; + list_del_init(&req_buf->list); + WRITE_ONCE(req_buf->wq_buf_avail, true); wake_up(&req_buf->wq_buf); - list_del(&req_buf->list); } /* The interrupt handler */ @@ -40,7 +40,7 @@ void virtio_pmem_host_ack(struct virtqueue *vq) spin_lock_irqsave(&vpmem->pmem_lock, flags); while ((req_data = virtqueue_get_buf(vq, &len)) != NULL) { virtio_pmem_wake_one_waiter(vpmem); - req_data->done = true; + WRITE_ONCE(req_data->done, true); wake_up(&req_data->host_acked); } spin_unlock_irqrestore(&vpmem->pmem_lock, flags); @@ -72,7 +72,7 @@ static int virtio_pmem_flush(struct nd_region *nd_region) if (!req_data) return -ENOMEM; - req_data->done = false; + WRITE_ONCE(req_data->done, false); init_waitqueue_head(&req_data->host_acked); init_waitqueue_head(&req_data->wq_buf); INIT_LIST_HEAD(&req_data->list); @@ -93,12 +93,12 @@ static int virtio_pmem_flush(struct nd_region *nd_region) GFP_ATOMIC)) == -ENOSPC) { dev_info(&vdev->dev, "failed to send command to virtio pmem device, no free slots in the virtqueue\n"); - req_data->wq_buf_avail = false; + WRITE_ONCE(req_data->wq_buf_avail, false); list_add_tail(&req_data->list, &vpmem->req_list); spin_unlock_irqrestore(&vpmem->pmem_lock, flags); /* A host response results in "host_ack" getting called */ - wait_event(req_data->wq_buf, req_data->wq_buf_avail); + wait_event(req_data->wq_buf, READ_ONCE(req_data->wq_buf_avail)); spin_lock_irqsave(&vpmem->pmem_lock, flags); } err1 = virtqueue_kick(vpmem->req_vq); @@ -112,7 +112,7 @@ static int virtio_pmem_flush(struct nd_region *nd_region) err = -EIO; } else { /* A host response results in "host_ack" getting called */ - wait_event(req_data->host_acked, req_data->done); + wait_event(req_data->host_acked, READ_ONCE(req_data->done)); err = le32_to_cpu(req_data->resp.ret); } From 37ac728bd77557c7fec841345ee098407d081e5f Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:33 +0800 Subject: [PATCH 069/857] nvdimm: virtio_pmem: refcount requests for token lifetime KASAN reports slab-use-after-free in __wake_up_common(): BUG: KASAN: slab-use-after-free in __wake_up_common+0x114/0x160 Read of size 8 at addr ffff88810fdcb710 by task swapper/0/0 CPU: 0 UID: 0 PID: 0 Comm: swapper/0 Not tainted 6.19.0-next-20260220-00006-g1eae5f204ec3 #4 PREEMPT(full) Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS Arch Linux 1.17.0-2-2 04/01/2014 Call Trace: dump_stack_lvl+0x6d/0xb0 print_report+0x170/0x4e2 ? __pfx__raw_spin_lock_irqsave+0x10/0x10 ? __virt_addr_valid+0x1dc/0x380 kasan_report+0xbc/0xf0 ? __wake_up_common+0x114/0x160 ? __wake_up_common+0x114/0x160 __wake_up_common+0x114/0x160 ? __pfx__raw_spin_lock_irqsave+0x10/0x10 __wake_up+0x36/0x60 virtio_pmem_host_ack+0x11d/0x3b0 ? sched_balance_domains+0x29f/0xb00 ? __pfx_virtio_pmem_host_ack+0x10/0x10 ? _raw_spin_lock_irqsave+0x98/0x100 ? __pfx__raw_spin_lock_irqsave+0x10/0x10 vring_interrupt+0x1c9/0x5e0 ? __pfx_vp_interrupt+0x10/0x10 vp_vring_interrupt+0x87/0x100 ? __pfx_vp_interrupt+0x10/0x10 __handle_irq_event_percpu+0x17f/0x550 ? __pfx__raw_spin_lock+0x10/0x10 handle_irq_event+0xab/0x1c0 handle_fasteoi_irq+0x276/0xae0 __common_interrupt+0x65/0x130 common_interrupt+0x78/0xa0 virtio_pmem_host_ack() wakes a request that has already been freed by the submitter. This happens when the request token is still reachable via the virtqueue, but virtio_pmem_flush() returns and frees it. Fix the token lifetime by refcounting struct virtio_pmem_request. virtio_pmem_flush() holds a submitter reference, and the virtqueue holds an extra reference once the request is queued. The completion path drops the virtqueue reference, and the submitter drops its reference before returning. Fixes: 6e84200c0a29 ("virtio-pmem: Add virtio pmem driver") Cc: stable@vger.kernel.org Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-9-me@linux.beauty> --- drivers/nvdimm/nd_virtio.c | 34 +++++++++++++++++++++++++++++----- drivers/nvdimm/virtio_pmem.h | 2 ++ 2 files changed, 31 insertions(+), 5 deletions(-) diff --git a/drivers/nvdimm/nd_virtio.c b/drivers/nvdimm/nd_virtio.c index 8c0d4347938a1c..1cf53f75b12810 100644 --- a/drivers/nvdimm/nd_virtio.c +++ b/drivers/nvdimm/nd_virtio.c @@ -15,6 +15,14 @@ struct virtio_pmem_flush_work { struct bio *bio; }; +static void virtio_pmem_req_release(struct kref *kref) +{ + struct virtio_pmem_request *req; + + req = container_of(kref, struct virtio_pmem_request, kref); + kfree(req); +} + static void virtio_pmem_wake_one_waiter(struct virtio_pmem *vpmem) { struct virtio_pmem_request *req_buf; @@ -42,6 +50,7 @@ void virtio_pmem_host_ack(struct virtqueue *vq) virtio_pmem_wake_one_waiter(vpmem); WRITE_ONCE(req_data->done, true); wake_up(&req_data->host_acked); + kref_put(&req_data->kref, virtio_pmem_req_release); } spin_unlock_irqrestore(&vpmem->pmem_lock, flags); } @@ -72,6 +81,7 @@ static int virtio_pmem_flush(struct nd_region *nd_region) if (!req_data) return -ENOMEM; + kref_init(&req_data->kref); WRITE_ONCE(req_data->done, false); init_waitqueue_head(&req_data->host_acked); init_waitqueue_head(&req_data->wq_buf); @@ -89,10 +99,23 @@ static int virtio_pmem_flush(struct nd_region *nd_region) * to req_list and wait for host_ack to wake us up when free * slots are available. */ - while ((err = virtqueue_add_sgs(vpmem->req_vq, sgs, 1, 1, req_data, - GFP_ATOMIC)) == -ENOSPC) { - - dev_info(&vdev->dev, "failed to send command to virtio pmem device, no free slots in the virtqueue\n"); + for (;;) { + err = virtqueue_add_sgs(vpmem->req_vq, sgs, 1, 1, req_data, + GFP_ATOMIC); + if (!err) { + /* + * Take the virtqueue reference while @pmem_lock is + * held so completion cannot run concurrently. + */ + kref_get(&req_data->kref); + break; + } + + if (err != -ENOSPC) + break; + + dev_info_ratelimited(&vdev->dev, + "failed to send command to virtio pmem device, no free slots in the virtqueue\n"); WRITE_ONCE(req_data->wq_buf_avail, false); list_add_tail(&req_data->list, &vpmem->req_list); spin_unlock_irqrestore(&vpmem->pmem_lock, flags); @@ -101,6 +124,7 @@ static int virtio_pmem_flush(struct nd_region *nd_region) wait_event(req_data->wq_buf, READ_ONCE(req_data->wq_buf_avail)); spin_lock_irqsave(&vpmem->pmem_lock, flags); } + err1 = virtqueue_kick(vpmem->req_vq); spin_unlock_irqrestore(&vpmem->pmem_lock, flags); /* @@ -116,7 +140,7 @@ static int virtio_pmem_flush(struct nd_region *nd_region) err = le32_to_cpu(req_data->resp.ret); } - kfree(req_data); + kref_put(&req_data->kref, virtio_pmem_req_release); return err; }; diff --git a/drivers/nvdimm/virtio_pmem.h b/drivers/nvdimm/virtio_pmem.h index e6dfc10ce0762a..3af92588bd9d18 100644 --- a/drivers/nvdimm/virtio_pmem.h +++ b/drivers/nvdimm/virtio_pmem.h @@ -12,12 +12,14 @@ #include #include +#include #include #include #include #include struct virtio_pmem_request { + struct kref kref; struct virtio_pmem_req req; struct virtio_pmem_resp resp; From 47c6063b6e3225d2165e4bcfffeac49ba68a3fb6 Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:34 +0800 Subject: [PATCH 070/857] nvdimm: virtio_pmem: publish done with release/acquire virtio_pmem_host_ack() publishes the device response by setting done and waking the submitter. The submitter reads resp.ret after wait_event() observes done. Use smp_store_release() on done and smp_load_acquire() in the wait condition so the response read is ordered after completion. Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-10-me@linux.beauty> --- drivers/nvdimm/nd_virtio.c | 19 ++++++++++++++++--- 1 file changed, 16 insertions(+), 3 deletions(-) diff --git a/drivers/nvdimm/nd_virtio.c b/drivers/nvdimm/nd_virtio.c index 1cf53f75b12810..e4e4284ae19e51 100644 --- a/drivers/nvdimm/nd_virtio.c +++ b/drivers/nvdimm/nd_virtio.c @@ -23,6 +23,19 @@ static void virtio_pmem_req_release(struct kref *kref) kfree(req); } +static void virtio_pmem_signal_done(struct virtio_pmem_request *req) +{ + /* Pairs with smp_load_acquire() in virtio_pmem_req_done(). */ + smp_store_release(&req->done, true); + wake_up(&req->host_acked); +} + +static bool virtio_pmem_req_done(struct virtio_pmem_request *req) +{ + /* Pairs with smp_store_release() in virtio_pmem_signal_done(). */ + return smp_load_acquire(&req->done); +} + static void virtio_pmem_wake_one_waiter(struct virtio_pmem *vpmem) { struct virtio_pmem_request *req_buf; @@ -48,8 +61,7 @@ void virtio_pmem_host_ack(struct virtqueue *vq) spin_lock_irqsave(&vpmem->pmem_lock, flags); while ((req_data = virtqueue_get_buf(vq, &len)) != NULL) { virtio_pmem_wake_one_waiter(vpmem); - WRITE_ONCE(req_data->done, true); - wake_up(&req_data->host_acked); + virtio_pmem_signal_done(req_data); kref_put(&req_data->kref, virtio_pmem_req_release); } spin_unlock_irqrestore(&vpmem->pmem_lock, flags); @@ -136,7 +148,8 @@ static int virtio_pmem_flush(struct nd_region *nd_region) err = -EIO; } else { /* A host response results in "host_ack" getting called */ - wait_event(req_data->host_acked, READ_ONCE(req_data->done)); + wait_event(req_data->host_acked, + virtio_pmem_req_done(req_data)); err = le32_to_cpu(req_data->resp.ret); } From 926e727970a9e95d31cb5ad83d55de5754d2be84 Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:35 +0800 Subject: [PATCH 071/857] nvdimm: virtio_pmem: isolate DMA request buffers The virtio-pmem request object stores wait queues, flags, and list pointers next to buffers mapped for virtqueue DMA. The response buffer is mapped DMA_FROM_DEVICE, so non-coherent DMA invalidation must not share a cache line with CPU-owned fields. Keep the request buffer outside the DMA-from-device group and wrap only the response buffer with __dma_from_device_group_begin/end. Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-11-me@linux.beauty> --- drivers/nvdimm/virtio_pmem.h | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/drivers/nvdimm/virtio_pmem.h b/drivers/nvdimm/virtio_pmem.h index 3af92588bd9d18..8843a8b9658741 100644 --- a/drivers/nvdimm/virtio_pmem.h +++ b/drivers/nvdimm/virtio_pmem.h @@ -10,6 +10,7 @@ #ifndef _LINUX_VIRTIO_PMEM_H #define _LINUX_VIRTIO_PMEM_H +#include #include #include #include @@ -20,8 +21,6 @@ struct virtio_pmem_request { struct kref kref; - struct virtio_pmem_req req; - struct virtio_pmem_resp resp; /* Wait queue to process deferred work after ack from host */ wait_queue_head_t host_acked; @@ -31,6 +30,11 @@ struct virtio_pmem_request { wait_queue_head_t wq_buf; bool wq_buf_avail; struct list_head list; + + struct virtio_pmem_req req; + __dma_from_device_group_begin(resp); + struct virtio_pmem_resp resp; + __dma_from_device_group_end(resp); }; struct virtio_pmem { From 34a57abfd949e8e25800270321a8ebc1d8eb397e Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:36 +0800 Subject: [PATCH 072/857] nvdimm: virtio_pmem: converge broken virtqueue to -EIO dmesg reports virtqueue failure and device reset: virtio_pmem virtio2: failed to send command to virtio pmem device, no free slots in the virtqueue virtio_pmem virtio2: virtio pmem device needs a reset virtio_pmem_flush() can wait for a free virtqueue descriptor (-ENOSPC). It can also wait for host completion. If the request virtqueue breaks, those waiters may never make progress. One example is notify failure from virtqueue_kick(). Track a device-level broken state and converge the failure to -EIO. New requests fail fast, -ENOSPC waiters are unlinked and woken, and the currently submitted request is woken so its host_acked waiter can return without waiting forever for host completion. Completed requests are forced to report an error after the queue is marked broken. Also serialize async parent-bio flush work against the broken state with pmem_lock. That way remove and freeze either drain work queued before virtio_pmem_mark_broken(), or later callers see nvdimm_flush() complete the parent bio synchronously with -EIO instead of queuing work after the drain point. Do not detach unused buffers from an active virtqueue. Runtime broken-queue handling only stops new submissions and wakes local waiters. Removal resets the device first. It then drains request tokens. After that, the device no longer owns the buffers when the virtqueue reference is dropped. Closes: https://lore.kernel.org/r/202512250116.ewtzlD0g-lkp@intel.com/ Signed-off-by: Li Chen Link: https://lore.kernel.org/r/202512250116.ewtzlD0g-lkp@intel.com/ Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-12-me@linux.beauty> --- drivers/nvdimm/nd_virtio.c | 126 +++++++++++++++++++++++++++++++---- drivers/nvdimm/virtio_pmem.c | 16 ++++- drivers/nvdimm/virtio_pmem.h | 8 +++ 3 files changed, 136 insertions(+), 14 deletions(-) diff --git a/drivers/nvdimm/nd_virtio.c b/drivers/nvdimm/nd_virtio.c index e4e4284ae19e51..a6820300cbe8fb 100644 --- a/drivers/nvdimm/nd_virtio.c +++ b/drivers/nvdimm/nd_virtio.c @@ -36,6 +36,12 @@ static bool virtio_pmem_req_done(struct virtio_pmem_request *req) return smp_load_acquire(&req->done); } +static void virtio_pmem_complete_err(struct virtio_pmem_request *req) +{ + req->resp.ret = cpu_to_le32(1); + virtio_pmem_signal_done(req); +} + static void virtio_pmem_wake_one_waiter(struct virtio_pmem *vpmem) { struct virtio_pmem_request *req_buf; @@ -50,6 +56,63 @@ static void virtio_pmem_wake_one_waiter(struct virtio_pmem *vpmem) wake_up(&req_buf->wq_buf); } +static void virtio_pmem_wake_all_waiters(struct virtio_pmem *vpmem) +{ + struct virtio_pmem_request *req, *tmp; + + list_for_each_entry_safe(req, tmp, &vpmem->req_list, list) { + list_del_init(&req->list); + WRITE_ONCE(req->wq_buf_avail, true); + wake_up(&req->wq_buf); + } +} + +static void virtio_pmem_clear_inflight(struct virtio_pmem *vpmem, + struct virtio_pmem_request *req) +{ + if (vpmem->req_inflight == req) + vpmem->req_inflight = NULL; +} + +static void virtio_pmem_wake_inflight(struct virtio_pmem *vpmem) +{ + struct virtio_pmem_request *req = vpmem->req_inflight; + + if (req) + wake_up(&req->host_acked); +} + +void virtio_pmem_mark_broken(struct virtio_pmem *vpmem) +{ + if (!READ_ONCE(vpmem->broken)) { + WRITE_ONCE(vpmem->broken, true); + dev_err_once(&vpmem->vdev->dev, "virtqueue is broken\n"); + } + + virtio_pmem_wake_inflight(vpmem); + virtio_pmem_wake_all_waiters(vpmem); +} +EXPORT_SYMBOL_GPL(virtio_pmem_mark_broken); + +void virtio_pmem_drain(struct virtio_pmem *vpmem) +{ + struct virtio_pmem_request *req; + unsigned int len; + + while ((req = virtqueue_get_buf(vpmem->req_vq, &len)) != NULL) { + virtio_pmem_clear_inflight(vpmem, req); + virtio_pmem_complete_err(req); + kref_put(&req->kref, virtio_pmem_req_release); + } + + while ((req = virtqueue_detach_unused_buf(vpmem->req_vq)) != NULL) { + virtio_pmem_clear_inflight(vpmem, req); + virtio_pmem_complete_err(req); + kref_put(&req->kref, virtio_pmem_req_release); + } +} +EXPORT_SYMBOL_GPL(virtio_pmem_drain); + /* The interrupt handler */ void virtio_pmem_host_ack(struct virtqueue *vq) { @@ -60,8 +123,12 @@ void virtio_pmem_host_ack(struct virtqueue *vq) spin_lock_irqsave(&vpmem->pmem_lock, flags); while ((req_data = virtqueue_get_buf(vq, &len)) != NULL) { + virtio_pmem_clear_inflight(vpmem, req_data); virtio_pmem_wake_one_waiter(vpmem); - virtio_pmem_signal_done(req_data); + if (READ_ONCE(vpmem->broken)) + virtio_pmem_complete_err(req_data); + else + virtio_pmem_signal_done(req_data); kref_put(&req_data->kref, virtio_pmem_req_release); } spin_unlock_irqrestore(&vpmem->pmem_lock, flags); @@ -89,6 +156,9 @@ static int virtio_pmem_flush(struct nd_region *nd_region) return -EIO; } + if (READ_ONCE(vpmem->broken)) + return -EIO; + req_data = kmalloc_obj(*req_data, GFP_NOIO); if (!req_data) return -ENOMEM; @@ -105,13 +175,18 @@ static int virtio_pmem_flush(struct nd_region *nd_region) sgs[1] = &ret; spin_lock_irqsave(&vpmem->pmem_lock, flags); - /* - * If virtqueue_add_sgs returns -ENOSPC then req_vq virtual - * queue does not have free descriptor. We add the request - * to req_list and wait for host_ack to wake us up when free - * slots are available. - */ + /* + * If virtqueue_add_sgs returns -ENOSPC then req_vq virtual + * queue does not have free descriptor. We add the request + * to req_list and wait for host_ack to wake us up when free + * slots are available. + */ for (;;) { + if (READ_ONCE(vpmem->broken)) { + err = -EIO; + break; + } + err = virtqueue_add_sgs(vpmem->req_vq, sgs, 1, 1, req_data, GFP_ATOMIC); if (!err) { @@ -120,6 +195,7 @@ static int virtio_pmem_flush(struct nd_region *nd_region) * held so completion cannot run concurrently. */ kref_get(&req_data->kref); + vpmem->req_inflight = req_data; break; } @@ -133,24 +209,41 @@ static int virtio_pmem_flush(struct nd_region *nd_region) spin_unlock_irqrestore(&vpmem->pmem_lock, flags); /* A host response results in "host_ack" getting called */ - wait_event(req_data->wq_buf, READ_ONCE(req_data->wq_buf_avail)); + wait_event(req_data->wq_buf, + READ_ONCE(req_data->wq_buf_avail) || + READ_ONCE(vpmem->broken)); spin_lock_irqsave(&vpmem->pmem_lock, flags); + + if (READ_ONCE(vpmem->broken)) + break; } - err1 = virtqueue_kick(vpmem->req_vq); + if (err == -EIO || virtqueue_is_broken(vpmem->req_vq)) + virtio_pmem_mark_broken(vpmem); + + err1 = true; + if (!err && !READ_ONCE(vpmem->broken)) { + err1 = virtqueue_kick(vpmem->req_vq); + if (!err1) + virtio_pmem_mark_broken(vpmem); + } spin_unlock_irqrestore(&vpmem->pmem_lock, flags); /* * virtqueue_add_sgs failed with error different than -ENOSPC, we can't * do anything about that. */ - if (err || !err1) { + if (READ_ONCE(vpmem->broken) || err || !err1) { dev_info(&vdev->dev, "failed to send command to virtio pmem device\n"); err = -EIO; } else { /* A host response results in "host_ack" getting called */ wait_event(req_data->host_acked, - virtio_pmem_req_done(req_data)); - err = le32_to_cpu(req_data->resp.ret); + virtio_pmem_req_done(req_data) || + READ_ONCE(vpmem->broken)); + if (virtio_pmem_req_done(req_data)) + err = le32_to_cpu(req_data->resp.ret); + else + err = -EIO; } kref_put(&req_data->kref, virtio_pmem_req_release); @@ -178,6 +271,7 @@ int async_pmem_flush(struct nd_region *nd_region, struct bio *bio) struct virtio_device *vdev = nd_region->provider_data; struct virtio_pmem *vpmem = vdev->priv; struct virtio_pmem_flush_work *flush; + unsigned long flags; int err; if (bio && bio->bi_iter.bi_sector != -1) { @@ -188,7 +282,15 @@ int async_pmem_flush(struct nd_region *nd_region, struct bio *bio) INIT_WORK(&flush->work, virtio_pmem_flush_work); flush->nd_region = nd_region; flush->bio = bio; + + spin_lock_irqsave(&vpmem->pmem_lock, flags); + if (READ_ONCE(vpmem->broken)) { + spin_unlock_irqrestore(&vpmem->pmem_lock, flags); + kfree(flush); + return -EIO; + } queue_work(vpmem->flush_wq, &flush->work); + spin_unlock_irqrestore(&vpmem->pmem_lock, flags); return NVDIMM_FLUSH_ASYNC; } diff --git a/drivers/nvdimm/virtio_pmem.c b/drivers/nvdimm/virtio_pmem.c index 9cf822a6c0c38e..36664a5ea25e3c 100644 --- a/drivers/nvdimm/virtio_pmem.c +++ b/drivers/nvdimm/virtio_pmem.c @@ -25,6 +25,8 @@ static int init_vq(struct virtio_pmem *vpmem) spin_lock_init(&vpmem->pmem_lock); INIT_LIST_HEAD(&vpmem->req_list); + vpmem->req_inflight = NULL; + WRITE_ONCE(vpmem->broken, false); return 0; }; @@ -148,11 +150,21 @@ static void virtio_pmem_remove(struct virtio_device *vdev) { struct nvdimm_bus *nvdimm_bus = dev_get_drvdata(&vdev->dev); struct virtio_pmem *vpmem = vdev->priv; + unsigned long flags; + + spin_lock_irqsave(&vpmem->pmem_lock, flags); + virtio_pmem_mark_broken(vpmem); + spin_unlock_irqrestore(&vpmem->pmem_lock, flags); - nvdimm_bus_unregister(nvdimm_bus); drain_workqueue(vpmem->flush_wq); - vdev->config->del_vqs(vdev); virtio_reset_device(vdev); + + spin_lock_irqsave(&vpmem->pmem_lock, flags); + virtio_pmem_drain(vpmem); + spin_unlock_irqrestore(&vpmem->pmem_lock, flags); + + nvdimm_bus_unregister(nvdimm_bus); + vdev->config->del_vqs(vdev); destroy_workqueue(vpmem->flush_wq); } diff --git a/drivers/nvdimm/virtio_pmem.h b/drivers/nvdimm/virtio_pmem.h index 8843a8b9658741..0b90777d7658b5 100644 --- a/drivers/nvdimm/virtio_pmem.h +++ b/drivers/nvdimm/virtio_pmem.h @@ -56,6 +56,12 @@ struct virtio_pmem { /* List to store deferred work if virtqueue is full */ struct list_head req_list; + /* Request currently owned by the virtqueue. */ + struct virtio_pmem_request *req_inflight; + + /* Fail fast and wake waiters if the request virtqueue is broken. */ + bool broken; + /* Synchronize virtqueue data */ spinlock_t pmem_lock; @@ -65,5 +71,7 @@ struct virtio_pmem { }; void virtio_pmem_host_ack(struct virtqueue *vq); +void virtio_pmem_mark_broken(struct virtio_pmem *vpmem); +void virtio_pmem_drain(struct virtio_pmem *vpmem); int async_pmem_flush(struct nd_region *nd_region, struct bio *bio); #endif From 7150f59e1c90d676b46b8633c73e3a4a8778c733 Mon Sep 17 00:00:00 2001 From: Li Chen Date: Tue, 30 Jun 2026 17:23:37 +0800 Subject: [PATCH 073/857] nvdimm: virtio_pmem: drain requests in freeze virtio_pmem_freeze() currently deletes virtqueues and resets the device without waking threads waiting for a virtqueue descriptor or a host completion. Mark the request virtqueue broken before reset. This makes new submissions fail fast and lets -ENOSPC waiters leave the wait list. Reset the device before draining used and unused request tokens, then delete the virtqueues. This wakes waiters with -EIO. It also keeps the detach call on a quiesced device. Clear req_vq after del_vqs(). Make drain tolerate a NULL queue so remove after freeze does not dereference a stale virtqueue pointer. Also make virtio_pmem_flush() stop checking req_vq once the broken state is visible. A waiter woken by freeze/remove can resume after del_vqs() has cleared req_vq. Signed-off-by: Li Chen Signed-off-by: Michael S. Tsirkin Message-ID: <20260630092338.2094628-13-me@linux.beauty> --- drivers/nvdimm/nd_virtio.c | 5 +++++ drivers/nvdimm/virtio_pmem.c | 34 +++++++++++++++++++++++++++++----- 2 files changed, 34 insertions(+), 5 deletions(-) diff --git a/drivers/nvdimm/nd_virtio.c b/drivers/nvdimm/nd_virtio.c index a6820300cbe8fb..3b8be79a20a0f7 100644 --- a/drivers/nvdimm/nd_virtio.c +++ b/drivers/nvdimm/nd_virtio.c @@ -99,6 +99,9 @@ void virtio_pmem_drain(struct virtio_pmem *vpmem) struct virtio_pmem_request *req; unsigned int len; + if (!vpmem->req_vq) + return; + while ((req = virtqueue_get_buf(vpmem->req_vq, &len)) != NULL) { virtio_pmem_clear_inflight(vpmem, req); virtio_pmem_complete_err(req); @@ -218,6 +221,8 @@ static int virtio_pmem_flush(struct nd_region *nd_region) break; } + if (READ_ONCE(vpmem->broken)) + err = -EIO; if (err == -EIO || virtqueue_is_broken(vpmem->req_vq)) virtio_pmem_mark_broken(vpmem); diff --git a/drivers/nvdimm/virtio_pmem.c b/drivers/nvdimm/virtio_pmem.c index 36664a5ea25e3c..7ee3fb1779f733 100644 --- a/drivers/nvdimm/virtio_pmem.c +++ b/drivers/nvdimm/virtio_pmem.c @@ -17,11 +17,16 @@ static struct virtio_device_id id_table[] = { /* Initialize virt queue */ static int init_vq(struct virtio_pmem *vpmem) { + int err; + /* single vq */ vpmem->req_vq = virtio_find_single_vq(vpmem->vdev, virtio_pmem_host_ack, "flush_queue"); - if (IS_ERR(vpmem->req_vq)) - return PTR_ERR(vpmem->req_vq); + if (IS_ERR(vpmem->req_vq)) { + err = PTR_ERR(vpmem->req_vq); + vpmem->req_vq = NULL; + return err; + } spin_lock_init(&vpmem->pmem_lock); INIT_LIST_HEAD(&vpmem->req_list); @@ -31,6 +36,15 @@ static int init_vq(struct virtio_pmem *vpmem) return 0; }; +static void virtio_pmem_del_vqs(struct virtio_pmem *vpmem) +{ + if (!vpmem->req_vq) + return; + + vpmem->vdev->config->del_vqs(vpmem->vdev); + vpmem->req_vq = NULL; +} + static int virtio_pmem_validate(struct virtio_device *vdev) { struct virtio_shm_region shm_reg; @@ -139,7 +153,7 @@ static int virtio_pmem_probe(struct virtio_device *vdev) virtio_reset_device(vdev); nvdimm_bus_unregister(vpmem->nvdimm_bus); out_vq: - vdev->config->del_vqs(vdev); + virtio_pmem_del_vqs(vpmem); out_wq: destroy_workqueue(vpmem->flush_wq); out_err: @@ -164,18 +178,28 @@ static void virtio_pmem_remove(struct virtio_device *vdev) spin_unlock_irqrestore(&vpmem->pmem_lock, flags); nvdimm_bus_unregister(nvdimm_bus); - vdev->config->del_vqs(vdev); + virtio_pmem_del_vqs(vpmem); destroy_workqueue(vpmem->flush_wq); } static int virtio_pmem_freeze(struct virtio_device *vdev) { struct virtio_pmem *vpmem = vdev->priv; + unsigned long flags; + + spin_lock_irqsave(&vpmem->pmem_lock, flags); + virtio_pmem_mark_broken(vpmem); + spin_unlock_irqrestore(&vpmem->pmem_lock, flags); drain_workqueue(vpmem->flush_wq); - vdev->config->del_vqs(vdev); virtio_reset_device(vdev); + spin_lock_irqsave(&vpmem->pmem_lock, flags); + virtio_pmem_drain(vpmem); + spin_unlock_irqrestore(&vpmem->pmem_lock, flags); + + virtio_pmem_del_vqs(vpmem); + return 0; } From c4f1daf41d1ef159c87e6ba75e814a2dfb1c0b4a Mon Sep 17 00:00:00 2001 From: Li RongQing Date: Wed, 1 Jul 2026 19:36:08 +0800 Subject: [PATCH 074/857] vdpa/mlx5: fix wrong list iterated in add_direct_chain error path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit In add_direct_chain(), newly allocated direct MR entries are added to the local list 'tmp', which is spliced into mr->head only on success. On the error path, the cleanup loop was incorrectly iterating over mr->head instead of tmp. Fix by iterating over 'tmp' in the err_alloc cleanup path. Fixes: 94abbccdf291 ("vdpa/mlx5: Add shared memory registration code") Signed-off-by: Li RongQing Acked-by: Eugenio Pérez Reviewed-by: Dragos Tatulea Signed-off-by: Michael S. Tsirkin Message-ID: <20260701113608.1972-1-lirongqing@baidu.com> --- drivers/vdpa/mlx5/core/mr.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/vdpa/mlx5/core/mr.c b/drivers/vdpa/mlx5/core/mr.c index 77a479aeaa85ae..b0c5ff23d022b8 100644 --- a/drivers/vdpa/mlx5/core/mr.c +++ b/drivers/vdpa/mlx5/core/mr.c @@ -481,7 +481,7 @@ static int add_direct_chain(struct mlx5_vdpa_dev *mvdev, return 0; err_alloc: - list_for_each_entry_safe(dmr, n, &mr->head, list) { + list_for_each_entry_safe(dmr, n, &tmp, list) { list_del_init(&dmr->list); unmap_direct_mr(mvdev, dmr); kfree(dmr); From 8e856acae997bcdae62f68acd45388cc676efb5c Mon Sep 17 00:00:00 2001 From: Pengpeng Hou Date: Sat, 4 Jul 2026 23:27:32 +0800 Subject: [PATCH 075/857] vdpa: alibaba: add missing MODULE_DEVICE_TABLE() The driver has a match table for the pci bus wired into its driver structure, but the table is not exported with MODULE_DEVICE_TABLE(). Add the missing MODULE_DEVICE_TABLE() entry so module alias information is generated for automatic module loading. This is a source-level fix. It does not claim dynamic hardware reproduction; the evidence is the driver-owned match table, its use by the driver registration structure, and the missing module alias publication. Signed-off-by: Pengpeng Hou Signed-off-by: Michael S. Tsirkin Message-ID: <20260704152732.55338-1-pengpeng@iscas.ac.cn> --- drivers/vdpa/alibaba/eni_vdpa.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/vdpa/alibaba/eni_vdpa.c b/drivers/vdpa/alibaba/eni_vdpa.c index e476504db0c82d..fd6fdba460945e 100644 --- a/drivers/vdpa/alibaba/eni_vdpa.c +++ b/drivers/vdpa/alibaba/eni_vdpa.c @@ -545,6 +545,7 @@ static struct pci_device_id eni_pci_ids[] = { VIRTIO_ID_NET) }, { 0 }, }; +MODULE_DEVICE_TABLE(pci, eni_pci_ids); static struct pci_driver eni_vdpa_driver = { .name = "alibaba-eni-vdpa", From 89eca683984bbebde2516db452b3f1dac6dabc2b Mon Sep 17 00:00:00 2001 From: Pengpeng Hou Date: Sun, 5 Jul 2026 08:25:46 +0800 Subject: [PATCH 076/857] vdpa: octeon_ep: add missing MODULE_DEVICE_TABLE() The driver has a match table for the pci bus wired into its driver structure, but the table is not exported with MODULE_DEVICE_TABLE(). Add the missing MODULE_DEVICE_TABLE() entry so module alias information is generated for automatic module loading. This is a source-level fix. It does not claim dynamic hardware reproduction; the evidence is the driver-owned match table, its use by the driver registration structure, and the missing module alias publication. Signed-off-by: Pengpeng Hou Signed-off-by: Michael S. Tsirkin Message-ID: <20260705002546.85004-1-pengpeng@iscas.ac.cn> --- drivers/vdpa/octeon_ep/octep_vdpa_main.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/vdpa/octeon_ep/octep_vdpa_main.c b/drivers/vdpa/octeon_ep/octep_vdpa_main.c index 5b35993750f57f..6d8ccdc14fb660 100644 --- a/drivers/vdpa/octeon_ep/octep_vdpa_main.c +++ b/drivers/vdpa/octeon_ep/octep_vdpa_main.c @@ -979,6 +979,7 @@ static struct pci_device_id octep_pci_vdpa_map[] = { { PCI_DEVICE(PCI_VENDOR_ID_CAVIUM, OCTEP_VDPA_DEVID_CN103K_VF) }, { 0 }, }; +MODULE_DEVICE_TABLE(pci, octep_pci_vdpa_map); static struct pci_driver octep_pci_vdpa = { .name = OCTEP_VDPA_DRIVER_NAME, From 77de17065aba501fba5e36ede6e6e3e4cba9a264 Mon Sep 17 00:00:00 2001 From: Li RongQing Date: Mon, 6 Jul 2026 14:09:02 +0800 Subject: [PATCH 077/857] vdpa/mlx5: fix wrong MLX5_ADDR_OF struct type in alloc_inout() In alloc_inout(), the qpc field offset was computed using MLX5_ADDR_OF(rst2init_qp_in, ...) in both the INIT2RTR_QP and RTR2RTS_QP cases. This is a copy-paste error: each case should use its own input structure type to get the correct qpc offset. Fix the INIT2RTR_QP case to use MLX5_ADDR_OF(init2rtr_qp_in, ...) and the RTR2RTS_QP case to use MLX5_ADDR_OF(rtr2rts_qp_in, ...). Signed-off-by: Li RongQing Reviewed-by: Dragos Tatulea Signed-off-by: Michael S. Tsirkin Message-ID: <20260706060902.2341-1-lirongqing@baidu.com> --- drivers/vdpa/mlx5/net/mlx5_vnet.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/vdpa/mlx5/net/mlx5_vnet.c b/drivers/vdpa/mlx5/net/mlx5_vnet.c index ad0d5fbbbca848..eb431a1471a04f 100644 --- a/drivers/vdpa/mlx5/net/mlx5_vnet.c +++ b/drivers/vdpa/mlx5/net/mlx5_vnet.c @@ -1080,7 +1080,7 @@ static void alloc_inout(struct mlx5_vdpa_net *ndev, int cmd, void **in, int *inl MLX5_SET(init2rtr_qp_in, *in, opcode, cmd); MLX5_SET(init2rtr_qp_in, *in, uid, ndev->mvdev.res.uid); MLX5_SET(init2rtr_qp_in, *in, qpn, qpn); - qpc = MLX5_ADDR_OF(rst2init_qp_in, *in, qpc); + qpc = MLX5_ADDR_OF(init2rtr_qp_in, *in, qpc); MLX5_SET(qpc, qpc, mtu, MLX5_QPC_MTU_256_BYTES); MLX5_SET(qpc, qpc, log_msg_max, 30); MLX5_SET(qpc, qpc, remote_qpn, rqpn); @@ -1098,7 +1098,7 @@ static void alloc_inout(struct mlx5_vdpa_net *ndev, int cmd, void **in, int *inl MLX5_SET(rtr2rts_qp_in, *in, opcode, cmd); MLX5_SET(rtr2rts_qp_in, *in, uid, ndev->mvdev.res.uid); MLX5_SET(rtr2rts_qp_in, *in, qpn, qpn); - qpc = MLX5_ADDR_OF(rst2init_qp_in, *in, qpc); + qpc = MLX5_ADDR_OF(rtr2rts_qp_in, *in, qpc); pp = MLX5_ADDR_OF(qpc, qpc, primary_address_path); MLX5_SET(ads, pp, ack_timeout, 14); MLX5_SET(qpc, qpc, retry_count, 7); From dffbb5120764be8812689d51d0d6f28048e08235 Mon Sep 17 00:00:00 2001 From: GuoHan Zhao Date: Tue, 14 Jul 2026 10:43:52 +0800 Subject: [PATCH 078/857] virtio: rtc: time out alarm requests RTC class operations run with rtc_device.ops_lock held. The virtio RTC alarm requests currently wait without a timeout for the device to return their requestq buffers. On surprise removal, virtio-pci marks the virtqueues broken before unregistering the virtio device. If an alarm request is waiting when the device stops responding, viortc_remove() blocks in viortc_class_stop() while trying to acquire ops_lock. The request cannot complete and device removal hangs until the waiting task is signalled. Use the same 60-second timeout as clock read requests for alarm reads, alarm programming, and alarm interrupt enable requests. The existing message reference counting keeps a timed-out request alive until a late response or device teardown. Fixes: 9d4f22fd563e ("virtio_rtc: Add RTC class driver") Assisted-by: Codex:gpt-5.6-sol Signed-off-by: GuoHan Zhao Reviewed-by: Peter Hilber Signed-off-by: Michael S. Tsirkin Message-ID: <20260714024352.71307-1-zhaoguohan@kylinos.cn> --- drivers/virtio/virtio_rtc_driver.c | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/drivers/virtio/virtio_rtc_driver.c b/drivers/virtio/virtio_rtc_driver.c index 4419735b0f0dcf..74616ba5be119c 100644 --- a/drivers/virtio/virtio_rtc_driver.c +++ b/drivers/virtio/virtio_rtc_driver.c @@ -574,8 +574,8 @@ static int viortc_msg_xfer(struct viortc_vq *vq, struct viortc_msg *msg, * read requests */ -/** timeout for clock readings, where timeouts are considered non-fatal */ -#define VIORTC_MSG_READ_TIMEOUT secs_to_jiffies(60) +/** timeout for runtime requests, where timeouts are considered non-fatal */ +#define VIORTC_MSG_TIMEOUT secs_to_jiffies(60) /** * viortc_read() - VIRTIO_RTC_REQ_READ wrapper @@ -600,7 +600,7 @@ int viortc_read(struct viortc_dev *viortc, u16 vio_clk_id, u64 *reading) VIORTC_MSG_WRITE(hdl, clock_id, &vio_clk_id); ret = viortc_msg_xfer(&viortc->vqs[VIORTC_REQUESTQ], VIORTC_MSG(hdl), - VIORTC_MSG_READ_TIMEOUT); + VIORTC_MSG_TIMEOUT); if (ret) { dev_dbg(&viortc->vdev->dev, "%s: xfer returned %d\n", __func__, ret); @@ -642,7 +642,7 @@ int viortc_read_cross(struct viortc_dev *viortc, u16 vio_clk_id, u8 hw_counter, VIORTC_MSG_WRITE(hdl, hw_counter, &hw_counter); ret = viortc_msg_xfer(&viortc->vqs[VIORTC_REQUESTQ], VIORTC_MSG(hdl), - VIORTC_MSG_READ_TIMEOUT); + VIORTC_MSG_TIMEOUT); if (ret) { dev_dbg(&viortc->vdev->dev, "%s: xfer returned %d\n", __func__, ret); @@ -809,7 +809,7 @@ int viortc_read_alarm(struct viortc_dev *viortc, u16 vio_clk_id, VIORTC_MSG_WRITE(hdl, clock_id, &vio_clk_id); ret = viortc_msg_xfer(&viortc->vqs[VIORTC_REQUESTQ], VIORTC_MSG(hdl), - 0); + VIORTC_MSG_TIMEOUT); if (ret) { dev_dbg(&viortc->vdev->dev, "%s: xfer returned %d\n", __func__, ret); @@ -858,7 +858,7 @@ int viortc_set_alarm(struct viortc_dev *viortc, u16 vio_clk_id, u64 alarm_time, VIORTC_MSG_WRITE(hdl, flags, &flags); ret = viortc_msg_xfer(&viortc->vqs[VIORTC_REQUESTQ], VIORTC_MSG(hdl), - 0); + VIORTC_MSG_TIMEOUT); if (ret) { dev_dbg(&viortc->vdev->dev, "%s: xfer returned %d\n", __func__, ret); @@ -900,7 +900,7 @@ int viortc_set_alarm_enabled(struct viortc_dev *viortc, u16 vio_clk_id, VIORTC_MSG_WRITE(hdl, flags, &flags); ret = viortc_msg_xfer(&viortc->vqs[VIORTC_REQUESTQ], VIORTC_MSG(hdl), - 0); + VIORTC_MSG_TIMEOUT); if (ret) { dev_dbg(&viortc->vdev->dev, "%s: xfer returned %d\n", __func__, ret); From 863b92f1c0b0a3a35040a5a8030df28c64f98d85 Mon Sep 17 00:00:00 2001 From: xiongweimin Date: Tue, 14 Jul 2026 10:44:34 +0800 Subject: [PATCH 079/857] vhost: fix inaccurate kdoc in iotlb helpers Correct missing "if" in the add_range_ctx return description, and align vhost_iotlb_alloc documentation with its NULL return on allocation failure. Signed-off-by: xiongweimin Signed-off-by: Michael S. Tsirkin Message-ID: <20260714024434.188302-1-15927021679@163.com> --- drivers/vhost/iotlb.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/drivers/vhost/iotlb.c b/drivers/vhost/iotlb.c index a1d4376a5b8722..3e5748e4d5fd39 100644 --- a/drivers/vhost/iotlb.c +++ b/drivers/vhost/iotlb.c @@ -50,7 +50,7 @@ EXPORT_SYMBOL_GPL(vhost_iotlb_map_free); * @perm: access permission of this range * @opaque: the opaque pointer for the new mapping * - * Returns an error last is smaller than start or memory allocation + * Returns an error if last is smaller than start or memory allocation * fails */ int vhost_iotlb_add_range_ctx(struct vhost_iotlb *iotlb, @@ -162,11 +162,11 @@ void vhost_iotlb_init(struct vhost_iotlb *iotlb, unsigned int limit, EXPORT_SYMBOL_GPL(vhost_iotlb_init); /** - * vhost_iotlb_alloc - add a new vhost IOTLB + * vhost_iotlb_alloc - allocate a new vhost IOTLB * @limit: maximum number of IOTLB entries * @flags: VHOST_IOTLB_FLAG_XXX * - * Returns an error is memory allocation fails + * Returns NULL if memory allocation fails */ struct vhost_iotlb *vhost_iotlb_alloc(unsigned int limit, unsigned int flags) { From dd0365d2b73f7b2dd63a739819fe330ff33ab8e7 Mon Sep 17 00:00:00 2001 From: xiongweimin Date: Tue, 14 Jul 2026 10:45:13 +0800 Subject: [PATCH 080/857] virtio: fix article before virtio in dma-buf comment Use "a virtio" rather than "an virtio". Signed-off-by: xiongweimin Signed-off-by: Michael S. Tsirkin Message-ID: <20260714024513.188571-1-15927021679@163.com> --- drivers/virtio/virtio_dma_buf.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/virtio/virtio_dma_buf.c b/drivers/virtio/virtio_dma_buf.c index 95c10632f84a7d..901282d82f0f2d 100644 --- a/drivers/virtio/virtio_dma_buf.c +++ b/drivers/virtio/virtio_dma_buf.c @@ -14,7 +14,7 @@ * struct embedded in a virtio_dma_buf_ops. * * This wraps dma_buf_export() to allow virtio drivers to create a dma-buf - * for an virtio exported object that can be queried by other virtio drivers + * for a virtio exported object that can be queried by other virtio drivers * for the object's UUID. */ struct dma_buf *virtio_dma_buf_export From 4b33968eaf4372706cc52228ccae2fd3542e702f Mon Sep 17 00:00:00 2001 From: xiongweimin Date: Tue, 14 Jul 2026 10:45:27 +0800 Subject: [PATCH 081/857] vdpa/solidrun: fix typos in snet_ctrl comments Correct "readind" and "the an error" in the DPU control path comments. Signed-off-by: xiongweimin Signed-off-by: Michael S. Tsirkin Message-ID: <20260714024527.188645-1-15927021679@163.com> --- drivers/vdpa/solidrun/snet_ctrl.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/vdpa/solidrun/snet_ctrl.c b/drivers/vdpa/solidrun/snet_ctrl.c index 3cef2571d15d3d..e284c3a06717c9 100644 --- a/drivers/vdpa/solidrun/snet_ctrl.c +++ b/drivers/vdpa/solidrun/snet_ctrl.c @@ -124,10 +124,10 @@ static int snet_wait_for_dpu_completion(struct snet_ctrl_regs __iomem *ctrl_regs * reading the in_process and error bits in the control register. * (2) Write the request opcode and the VQ idx in the opcode register * and write the buffer size in the control register. - * (3) Start readind chunks of data, chunk_ready bit indicates that a + * (3) Start reading chunks of data, chunk_ready bit indicates that a * data chunk is available, we signal that we read the data by clearing the bit. * (4) Detect that the transfer is completed when the in_process bit - * in the control register is cleared or when the an error appears. + * in the control register is cleared or when an error appears. */ static int snet_ctrl_read_from_dpu(struct snet *snet, u16 opcode, u16 vq_idx, void *buffer, u32 buf_size) From ffa8083cf37c6142afe5511f439a928468f59554 Mon Sep 17 00:00:00 2001 From: xiongweimin Date: Tue, 14 Jul 2026 11:24:17 +0800 Subject: [PATCH 082/857] virtio_mem: fix typo in comment Correct "actipn" to "action". Signed-off-by: xiongweimin Reviewed-by: Parav Pandit Acked-by: David Hildenbrand (Arm) Signed-off-by: Michael S. Tsirkin Message-ID: <20260714032417.201353-1-xiongwm2026@163.com> --- drivers/virtio/virtio_mem.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/virtio/virtio_mem.c b/drivers/virtio/virtio_mem.c index 82a285c0926de8..e18dd736f2eceb 100644 --- a/drivers/virtio/virtio_mem.c +++ b/drivers/virtio/virtio_mem.c @@ -1080,7 +1080,7 @@ static int virtio_mem_memory_notifier_cb(struct notifier_block *nb, atomic64_sub(size, &vm->offline_size); /* * Start adding more memory once we onlined half of our - * threshold. Don't trigger if it's possibly due to our actipn + * threshold. Don't trigger if it's possibly due to our action * (e.g., us adding memory which gets onlined immediately from * the core). */ From b1efd619a751850c9ed972786c82b59121edd826 Mon Sep 17 00:00:00 2001 From: Gabriel Somlo Date: Wed, 15 Jul 2026 11:19:08 -0400 Subject: [PATCH 083/857] MAINTAINERS: remove Gabriel from LiteX and fw-cfg drivers I no longer have the bandwidth to look after these drivers, so I'm leaving them in the able hands of my co-maintainers. Signed-off-by: Gabriel Somlo Signed-off-by: Michael S. Tsirkin Message-ID: <20260715151908.1534002-1-gsomlo@gmail.com> --- MAINTAINERS | 2 -- 1 file changed, 2 deletions(-) diff --git a/MAINTAINERS b/MAINTAINERS index 5114e6db7307d2..3ea61eb118eb84 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -15008,7 +15008,6 @@ F: lib/tests/list-test.c LITEX PLATFORM M: Karol Gugala M: Mateusz Holenko -M: Gabriel Somlo M: Joel Stanley S: Maintained F: Documentation/devicetree/bindings/*/litex,*.yaml @@ -21919,7 +21918,6 @@ S: Maintained F: drivers/net/ipa/ QEMU MACHINE EMULATOR AND VIRTUALIZER SUPPORT -M: Gabriel Somlo M: "Michael S. Tsirkin" L: qemu-devel@nongnu.org S: Maintained From 5c445c2cc50292c2e3924061b2cbfc0a83b1c339 Mon Sep 17 00:00:00 2001 From: Weimin Xiong Date: Thu, 16 Jul 2026 13:43:53 +0800 Subject: [PATCH 084/857] vdpa/mlx5: roll back MR update after VQ setup failure mlx5_vdpa_change_map() must install the new MR before rebuilding or resuming virtqueues, because both paths read the MR keys from mvdev->mres.mr[]. If rebuilding the virtqueue resources fails, the new MR must not remain installed after its reference is released. Keep an extra reference to the old MR before replacing it. On setup failure, restore the old MR; the saved reference then becomes the map reference, while replacing the new MR drops its map reference. Make mlx5_vdpa_change_map() consume new_mr on all error paths so that set_map_data() does not release an MR already released during rollback. v2: - Keep the new MR installed while virtqueues are rebuilt. - Restore the old MR only after setup_vq_resources() fails. Signed-off-by: Weimin Xiong Signed-off-by: Michael S. Tsirkin Message-ID: <20260716054353.155805-1-xiongwm2026@163.com> --- drivers/vdpa/mlx5/net/mlx5_vnet.c | 23 +++++++++++++++-------- 1 file changed, 15 insertions(+), 8 deletions(-) diff --git a/drivers/vdpa/mlx5/net/mlx5_vnet.c b/drivers/vdpa/mlx5/net/mlx5_vnet.c index eb431a1471a04f..8563fec2855d50 100644 --- a/drivers/vdpa/mlx5/net/mlx5_vnet.c +++ b/drivers/vdpa/mlx5/net/mlx5_vnet.c @@ -3055,18 +3055,24 @@ static int mlx5_vdpa_change_map(struct mlx5_vdpa_dev *mvdev, unsigned int asid) { struct mlx5_vdpa_net *ndev = to_mlx5_vdpa_ndev(mvdev); + struct mlx5_vdpa_mr *old_mr; bool teardown = !is_resumable(ndev); int err; suspend_vqs(ndev, 0, ndev->cur_num_vqs); if (teardown) { err = save_channels_info(ndev); - if (err) + if (err) { + mlx5_vdpa_put_mr(mvdev, new_mr); return err; + } teardown_vq_resources(ndev); } + /* Keep the old MR alive in case rebuilding the VQs fails. */ + old_mr = mvdev->mres.mr[asid]; + mlx5_vdpa_get_mr(mvdev, old_mr); mlx5_vdpa_update_mr(mvdev, new_mr, asid); for (int i = 0; i < mvdev->max_vqs; i++) @@ -3074,17 +3080,22 @@ static int mlx5_vdpa_change_map(struct mlx5_vdpa_dev *mvdev, MLX5_VIRTQ_MODIFY_MASK_DESC_GROUP_MKEY; if (!(mvdev->status & VIRTIO_CONFIG_S_DRIVER_OK) || mvdev->suspended) - return 0; + goto out; if (teardown) { restore_channels_info(ndev); err = setup_vq_resources(ndev, true); - if (err) + if (err) { + /* The saved reference becomes the restored map reference. */ + mlx5_vdpa_update_mr(mvdev, old_mr, asid); return err; + } } resume_vqs(ndev, 0, ndev->cur_num_vqs); +out: + mlx5_vdpa_put_mr(mvdev, old_mr); return 0; } @@ -3368,15 +3379,11 @@ static int set_map_data(struct mlx5_vdpa_dev *mvdev, struct vhost_iotlb *iotlb, err = mlx5_vdpa_change_map(mvdev, new_mr, asid); if (err) { mlx5_vdpa_err(mvdev, "change map failed(%d)\n", err); - goto out_err; + return err; } } return mlx5_vdpa_update_cvq_iotlb(mvdev, iotlb, asid); - -out_err: - mlx5_vdpa_put_mr(mvdev, new_mr); - return err; } static int mlx5_vdpa_set_map(struct vdpa_device *vdev, unsigned int asid, From e22849120e8698a1bbb5d44e95581dfe7c1f1527 Mon Sep 17 00:00:00 2001 From: Jinqian Yang Date: Thu, 16 Jul 2026 19:59:40 +0800 Subject: [PATCH 085/857] virtio_ring: fix infinite loop in virtnet_poll_cleantx when device is broken virtnet_poll_cleantx() contains a do-while loop that cleans up transmitted TX buffers and calls virtqueue_enable_cb_delayed() to check whether more buffers need processing. When the virtio backend stops responding during guest reboot, used->idx is never updated, so virtqueue_enable_cb_delayed() always returns false and the loop never terminates. Then it will block reboot process, and the guest will hang. The problem occurs during guest reboot under network traffic: 1. kernel_restart() -> device_shutdown() traverses the device list 2. virtio_dev_shutdown() calls virtio_break_device() which sets vq->broken = true 3. virtio_dev_shutdown() then calls virtio_synchronize_cbs() to wait for in-flight callbacks to complete 4. A virtio interrupt fires, softirq is deferred to ksoftirqd which calls net_rx_action() -> virtnet_poll() -> virtnet_poll_cleantx() 5. virtnet_poll_cleantx() enters the do-while loop and never exits because the QEMU backend has stopped updating used->idx, despite vq->broken having been set to true in step 2. Since the loop runs inside ksoftirqd (a SCHED_OTHER kthread), it is visible to the scheduler and does not trigger a hard lockup. However, the kthread never leaves the loop, so RCU detects it as a CPU stall and reports it periodically. Meanwhile, the reboot process remains blocked in device_shutdown() because virtio_dev_shutdown() cannot complete its synchronization step, and the guest hangs permanently. This can be reproduced on a guest with a virtio-net device: run iperf3 traffic in the guest, then trigger reboot. The reboot occasionally hangs permanently with RCU stall on ksoftirqd. Observed on ARM64 KVM guest: CPU#1 RCU stall (ksoftirqd/1), repeated periodically: virtqueue_enable_cb_delayed_split <- virtnet_poll <- __napi_poll <- net_rx_action <- handle_softirqs <- run_ksoftirqd <- smpboot_thread_fn <- kthread Fix by adding a vq->broken check in virtqueue_enable_cb_delayed(), so that the loop exits immediately when the device is broken, allowing the device shutdown to proceed. Signed-off-by: Jinqian Yang Reviewed-by: Xuan Zhuo Signed-off-by: Michael S. Tsirkin Message-ID: <20260716115940.394832-1-yangjinqian1@huawei.com> --- drivers/virtio/virtio_ring.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/drivers/virtio/virtio_ring.c b/drivers/virtio/virtio_ring.c index b438dc2ce1b80a..5c169fbb418ad0 100644 --- a/drivers/virtio/virtio_ring.c +++ b/drivers/virtio/virtio_ring.c @@ -3233,6 +3233,14 @@ bool virtqueue_enable_cb_delayed(struct virtqueue *_vq) { struct vring_virtqueue *vq = to_vvq(_vq); + /* + * When the device is broken there is no point in polling used->idx, + * the backend will never update it. Return true to let callers + * exit their cleanup loops instead of spinning forever. + */ + if (unlikely(vq->broken)) + return true; + if (vq->event_triggered) data_race(vq->event_triggered = false); From e9660ae5f436f4af1b25ac4f6d72607f17173282 Mon Sep 17 00:00:00 2001 From: Pan Chuang Date: Thu, 16 Jul 2026 22:13:43 +0800 Subject: [PATCH 086/857] vdpa: Remove redundant dev_err() Since commit 55b48e23f5c4 ("genirq/devres: Add error handling in devm_request_*_irq()"), devm_request_irq() automatically logs detailed error messages on failure. Remove the now-redundant driver-specific dev_err() calls. Signed-off-by: Pan Chuang Signed-off-by: Michael S. Tsirkin Message-ID: <20260716141349.158824-1-panchuang@vivo.com> --- drivers/vdpa/octeon_ep/octep_vdpa_main.c | 4 +--- drivers/vdpa/virtio_pci/vp_vdpa.c | 12 +++--------- 2 files changed, 4 insertions(+), 12 deletions(-) diff --git a/drivers/vdpa/octeon_ep/octep_vdpa_main.c b/drivers/vdpa/octeon_ep/octep_vdpa_main.c index 6d8ccdc14fb660..23e280a29209ba 100644 --- a/drivers/vdpa/octeon_ep/octep_vdpa_main.c +++ b/drivers/vdpa/octeon_ep/octep_vdpa_main.c @@ -170,10 +170,8 @@ static int octep_request_irqs(struct octep_hw *oct_hw, irqreturn_t (*irq_handler irq = pci_irq_vector(pdev, idx); ret = devm_request_irq(&pdev->dev, irq, irq_handler, 0, dev_name(&pdev->dev), oct_hw); - if (ret) { - dev_err(&pdev->dev, "Failed to register interrupt handler\n"); + if (ret) goto free_irqs; - } oct_hw->irqs[idx] = irq; } oct_hw->requested_irqs = nb_irqs; diff --git a/drivers/vdpa/virtio_pci/vp_vdpa.c b/drivers/vdpa/virtio_pci/vp_vdpa.c index 51ffc245a03855..f2eb654b1665a1 100644 --- a/drivers/vdpa/virtio_pci/vp_vdpa.c +++ b/drivers/vdpa/virtio_pci/vp_vdpa.c @@ -189,11 +189,8 @@ static int vp_vdpa_request_irq(struct vp_vdpa *vp_vdpa) vp_vdpa_vq_handler, 0, vp_vdpa->vring[i].msix_name, &vp_vdpa->vring[i]); - if (ret) { - dev_err(&pdev->dev, - "vp_vdpa: fail to request irq for vq %d\n", i); + if (ret) goto err; - } vp_modern_queue_vector(mdev, i, msix_vec); vp_vdpa->vring[i].irq = irq; msix_vec++; @@ -204,11 +201,8 @@ static int vp_vdpa_request_irq(struct vp_vdpa *vp_vdpa) irq = pci_irq_vector(pdev, msix_vec); ret = devm_request_irq(&pdev->dev, irq, vp_vdpa_config_handler, 0, vp_vdpa->msix_name, vp_vdpa); - if (ret) { - dev_err(&pdev->dev, - "vp_vdpa: fail to request irq for config: %d\n", ret); - goto err; - } + if (ret) + goto err; vp_modern_config_vector(mdev, msix_vec); vp_vdpa->config_irq = irq; From 57c53140db212722945c2e37b1dd94d73fa22c26 Mon Sep 17 00:00:00 2001 From: xiongweimin Date: Thu, 16 Jul 2026 11:02:36 +0800 Subject: [PATCH 087/857] vhost: reject zero-size IOTLB INVALIDATE Reject VHOST_IOTLB_INVALIDATE messages with size == 0 to prevent iova + size - 1 from underflowing to U64_MAX, which would incorrectly delete the entire IOTLB. Signed-off-by: xiongweimin Signed-off-by: Michael S. Tsirkin Message-ID: <20260716030236.124322-1-xiongwm2026@163.com> --- drivers/vhost/vhost.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/drivers/vhost/vhost.c b/drivers/vhost/vhost.c index 1be9c0c53bedbb..a0c1d54019aac5 100644 --- a/drivers/vhost/vhost.c +++ b/drivers/vhost/vhost.c @@ -1671,6 +1671,10 @@ static int vhost_process_iotlb_msg(struct vhost_dev *dev, u32 asid, ret = -EFAULT; break; } + if (!msg->size) { + ret = -EINVAL; + break; + } vhost_vq_meta_reset(dev); vhost_iotlb_del_range(dev->iotlb, msg->iova, msg->iova + msg->size - 1); From 53f9db4a4920d810058a226664c2b0c9f1280e56 Mon Sep 17 00:00:00 2001 From: GuoHan Zhao Date: Mon, 20 Jul 2026 09:44:21 +0800 Subject: [PATCH 088/857] tools/virtio: Fix userspace typo in vringh test comment Fix a misspelling of "userspace" in the vringh test description. Signed-off-by: GuoHan Zhao Signed-off-by: Michael S. Tsirkin Message-ID: <20260720014421.89345-1-zhaoguohan@kylinos.cn> --- tools/virtio/vringh_test.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/virtio/vringh_test.c b/tools/virtio/vringh_test.c index 5ea6d29bc992d2..84961b9ab5ffd5 100644 --- a/tools/virtio/vringh_test.c +++ b/tools/virtio/vringh_test.c @@ -1,5 +1,5 @@ // SPDX-License-Identifier: GPL-2.0 -/* Simple test of virtio code, entirely in userpsace. */ +/* Simple test of virtio code, entirely in userspace. */ #define _GNU_SOURCE #include #include From a68cef7ac226e40f493d4f1031cf0b755c46ea6b Mon Sep 17 00:00:00 2001 From: GuoHan Zhao Date: Mon, 20 Jul 2026 09:45:06 +0800 Subject: [PATCH 089/857] tools/virtio: Fix control typo in trace agent comment Fix a misspelling of "control" in the trace agent controller description. Signed-off-by: GuoHan Zhao Signed-off-by: Michael S. Tsirkin Message-ID: <20260720014506.90012-1-zhaoguohan@kylinos.cn> --- tools/virtio/virtio-trace/trace-agent-ctl.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/virtio/virtio-trace/trace-agent-ctl.c b/tools/virtio/virtio-trace/trace-agent-ctl.c index 39860be6e2d86f..9577579e86da35 100644 --- a/tools/virtio/virtio-trace/trace-agent-ctl.c +++ b/tools/virtio/virtio-trace/trace-agent-ctl.c @@ -84,7 +84,7 @@ static int wait_order(int ctl_fd) } /* - * contol read/write threads by handling global_run_operation + * control read/write threads by handling global_run_operation */ void *rw_ctl_loop(int ctl_fd) { From 9bbe30207f300997e73fc16b4ee4b25020fd0457 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Eugenio=20P=C3=A9rez?= Date: Tue, 7 Jul 2026 14:24:59 +0200 Subject: [PATCH 090/857] vduse: store control device pointer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit This helps log the errors in next patches. The alternative is to perform a linear search for it with class_find_device_by_devt(class, devt), as device_destroy do for cleaning. Signed-off-by: Eugenio Pérez Signed-off-by: Michael S. Tsirkin Message-ID: <20260707122502.239022-2-eperezma@redhat.com> --- drivers/vdpa/vdpa_user/vduse_dev.c | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c index 10dcf016bfb061..861a8093daa07d 100644 --- a/drivers/vdpa/vdpa_user/vduse_dev.c +++ b/drivers/vdpa/vdpa_user/vduse_dev.c @@ -163,6 +163,7 @@ static DEFINE_IDR(vduse_idr); static dev_t vduse_major; static struct cdev vduse_ctrl_cdev; +static const struct device *vduse_ctrl_dev; static struct cdev vduse_cdev; static struct workqueue_struct *vduse_irq_wq; static struct workqueue_struct *vduse_irq_bound_wq; @@ -2531,7 +2532,6 @@ static void vduse_mgmtdev_exit(void) static int vduse_init(void) { int ret; - struct device *dev; ret = class_register(&vduse_class); if (ret) @@ -2548,9 +2548,10 @@ static int vduse_init(void) if (ret) goto err_ctrl_cdev; - dev = device_create(&vduse_class, NULL, vduse_major, NULL, "control"); - if (IS_ERR(dev)) { - ret = PTR_ERR(dev); + vduse_ctrl_dev = device_create(&vduse_class, NULL, vduse_major, NULL, "control"); + if (IS_ERR(vduse_ctrl_dev)) { + ret = PTR_ERR(vduse_ctrl_dev); + vduse_ctrl_dev = NULL; goto err_device; } From a503a9174044b4ace6863db5262079c50c341efd Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Eugenio=20P=C3=A9rez?= Date: Tue, 7 Jul 2026 14:25:00 +0200 Subject: [PATCH 091/857] vduse: add VDUSE_GET_FEATURES ioctl MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add an ioctl to allow VDUSE instances to query the available features supported by the kernel module. Signed-off-by: Eugenio Pérez Signed-off-by: Michael S. Tsirkin Message-ID: <20260707122502.239022-3-eperezma@redhat.com> --- drivers/vdpa/vdpa_user/vduse_dev.c | 7 +++++++ include/uapi/linux/vduse.h | 3 +++ 2 files changed, 10 insertions(+) diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c index 861a8093daa07d..f3f24cc59eb02f 100644 --- a/drivers/vdpa/vdpa_user/vduse_dev.c +++ b/drivers/vdpa/vdpa_user/vduse_dev.c @@ -51,6 +51,9 @@ #define IRQ_UNBOUND -1 +/* Supported VDUSE features */ +static const uint64_t vduse_features; + /* * VDUSE instance have not asked the vduse API version, so assume 0. * @@ -2342,6 +2345,10 @@ static long vduse_ioctl(struct file *file, unsigned int cmd, ret = vduse_destroy_dev(name); break; } + case VDUSE_GET_FEATURES: + ret = put_user(vduse_features, (u64 __user *)argp); + break; + default: ret = -EINVAL; break; diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h index 361eea511c2182..89aa3b448c0a0b 100644 --- a/include/uapi/linux/vduse.h +++ b/include/uapi/linux/vduse.h @@ -63,6 +63,9 @@ struct vduse_dev_config { */ #define VDUSE_DESTROY_DEV _IOW(VDUSE_BASE, 0x03, char[VDUSE_NAME_MAX]) +/* Get the VDUSE supported features */ +#define VDUSE_GET_FEATURES _IOR(VDUSE_BASE, 0x04, __u64) + /* The ioctls for VDUSE device (/dev/vduse/$NAME) */ /** From eefeab03e49c5183a22b30e3cf205e0d83f8071e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Eugenio=20P=C3=A9rez?= Date: Tue, 7 Jul 2026 14:25:01 +0200 Subject: [PATCH 092/857] vduse: add VDUSE_SET_FEATURES ioctl MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add an ioctl to allow VDUSE instances to set the VDUSE features supported by the userland VDUSE instance. Signed-off-by: Eugenio Pérez Signed-off-by: Michael S. Tsirkin Message-ID: <20260707122502.239022-4-eperezma@redhat.com> --- drivers/vdpa/vdpa_user/vduse_dev.c | 24 ++++++++++++++++++++++++ include/uapi/linux/vduse.h | 3 +++ 2 files changed, 27 insertions(+) diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c index f3f24cc59eb02f..2292cabe4270f7 100644 --- a/drivers/vdpa/vdpa_user/vduse_dev.c +++ b/drivers/vdpa/vdpa_user/vduse_dev.c @@ -159,6 +159,7 @@ struct vduse_dev_msg { struct vduse_control { u64 api_version; + u64 vduse_features; }; static DEFINE_MUTEX(vduse_lock); @@ -2348,7 +2349,29 @@ static long vduse_ioctl(struct file *file, unsigned int cmd, case VDUSE_GET_FEATURES: ret = put_user(vduse_features, (u64 __user *)argp); break; + case VDUSE_SET_FEATURES: { + u64 features; + ret = -EFAULT; + if (get_user(features, (u64 __user *)argp)) { + dev_dbg(vduse_ctrl_dev, "Could not get vduse features"); + break; + } + + ret = -EINVAL; + if (features & ~vduse_features) { + dev_dbg(vduse_ctrl_dev, + "Invalid features in %llx, expected %llx", + features, vduse_features); + break; + } + + ret = 0; + control->vduse_features = features; + dev_dbg(vduse_ctrl_dev, "Set features %llx", features); + + break; + } default: ret = -EINVAL; break; @@ -2375,6 +2398,7 @@ static int vduse_open(struct inode *inode, struct file *file) return -ENOMEM; control->api_version = VDUSE_API_VERSION_NOT_ASKED; + control->vduse_features = 0; file->private_data = control; return 0; diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h index 89aa3b448c0a0b..f14c965bb7f68d 100644 --- a/include/uapi/linux/vduse.h +++ b/include/uapi/linux/vduse.h @@ -66,6 +66,9 @@ struct vduse_dev_config { /* Get the VDUSE supported features */ #define VDUSE_GET_FEATURES _IOR(VDUSE_BASE, 0x04, __u64) +/* Set the VDUSE features */ +#define VDUSE_SET_FEATURES _IOW(VDUSE_BASE, 0x05, __u64) + /* The ioctls for VDUSE device (/dev/vduse/$NAME) */ /** From f66f0b2d4486973b847451368142eb9a07bd03dc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Eugenio=20P=C3=A9rez?= Date: Tue, 7 Jul 2026 14:25:02 +0200 Subject: [PATCH 093/857] vduse: add F_QUEUE_READY feature MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add the VDUSE_F_QUEUE_READY feature flag. This allows the kernel module to explicitly signal userspace when a specific virtqueue has been enabled. In scenarios like Live Migration of VirtIO net devices, the dataplane starts after the control virtqueue allowing QEMU to apply configuration in the destination device. Signed-off-by: Eugenio Pérez Signed-off-by: Michael S. Tsirkin Message-ID: <20260707122502.239022-5-eperezma@redhat.com> --- drivers/vdpa/vdpa_user/vduse_dev.c | 73 +++++++++++++++++++++++------- include/uapi/linux/vduse.h | 18 ++++++++ 2 files changed, 74 insertions(+), 17 deletions(-) diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c index 2292cabe4270f7..87d6748b50cc6d 100644 --- a/drivers/vdpa/vdpa_user/vduse_dev.c +++ b/drivers/vdpa/vdpa_user/vduse_dev.c @@ -9,6 +9,7 @@ */ #include "linux/virtio_net.h" +#include #include #include #include @@ -52,7 +53,7 @@ #define IRQ_UNBOUND -1 /* Supported VDUSE features */ -static const uint64_t vduse_features; +static const uint64_t vduse_features = BIT_U64(VDUSE_F_QUEUE_READY); /* * VDUSE instance have not asked the vduse API version, so assume 0. @@ -76,6 +77,7 @@ struct vduse_virtqueue { u32 group; spinlock_t kick_lock; spinlock_t irq_lock; + spinlock_t ready_lock; struct eventfd_ctx *kickfd; struct vdpa_callback cb; struct work_struct inject; @@ -119,6 +121,7 @@ struct vduse_dev { char *name; struct mutex lock; spinlock_t msg_lock; + u64 vduse_features; u64 msg_unique; u32 msg_timeout; wait_queue_head_t waitq; @@ -513,7 +516,9 @@ static void vduse_dev_reset(struct vduse_dev *dev) for (i = 0; i < dev->vq_num; i++) { struct vduse_virtqueue *vq = dev->vqs[i]; - vq->ready = false; + scoped_guard(spinlock_bh, &vq->ready_lock) { + vq->ready = false; + } vq->desc_addr = 0; vq->driver_addr = 0; vq->device_addr = 0; @@ -555,16 +560,15 @@ static int vduse_vdpa_set_vq_address(struct vdpa_device *vdpa, u16 idx, static void vduse_vq_kick(struct vduse_virtqueue *vq) { - spin_lock(&vq->kick_lock); - if (!vq->ready) - goto unlock; + guard(spinlock)(&vq->kick_lock); + scoped_guard(spinlock_bh, &vq->ready_lock) + if (!vq->ready) + return; if (vq->kickfd) eventfd_signal(vq->kickfd); else vq->kicked = true; -unlock: - spin_unlock(&vq->kick_lock); } static void vduse_vq_kick_work(struct work_struct *work) @@ -624,7 +628,30 @@ static void vduse_vdpa_set_vq_ready(struct vdpa_device *vdpa, { struct vduse_dev *dev = vdpa_to_vduse(vdpa); struct vduse_virtqueue *vq = dev->vqs[idx]; + struct vduse_dev_msg msg = { 0 }; + int r; + + if (dev->vduse_features & BIT_U64(VDUSE_F_QUEUE_READY)) { + msg.req.type = VDUSE_SET_VQ_READY; + msg.req.vq_ready.num = idx; + msg.req.vq_ready.ready = !!ready; + + r = vduse_dev_msg_sync(dev, &msg); + + if (r < 0) { + dev_dbg(&vdpa->dev, "device refuses to set vq %u ready %u", + idx, ready); + /* We can't do better than break the device in this case */ + spin_lock(&dev->msg_lock); + vduse_dev_broken(dev); + spin_unlock(&dev->msg_lock); + + return; + } + } + + guard(spinlock_bh)(&vq->ready_lock); vq->ready = ready; } @@ -633,6 +660,7 @@ static bool vduse_vdpa_get_vq_ready(struct vdpa_device *vdpa, u16 idx) struct vduse_dev *dev = vdpa_to_vduse(vdpa); struct vduse_virtqueue *vq = dev->vqs[idx]; + guard(spinlock_bh)(&vq->ready_lock); return vq->ready; } @@ -1120,15 +1148,16 @@ static int vduse_kickfd_setup(struct vduse_dev *dev, } else if (eventfd->fd != VDUSE_EVENTFD_DEASSIGN) return 0; - spin_lock(&vq->kick_lock); + guard(spinlock)(&vq->kick_lock); if (vq->kickfd) eventfd_ctx_put(vq->kickfd); vq->kickfd = ctx; + + guard(spinlock_bh)(&vq->ready_lock); if (vq->ready && vq->kicked && vq->kickfd) { eventfd_signal(vq->kickfd); vq->kicked = false; } - spin_unlock(&vq->kick_lock); return 0; } @@ -1159,10 +1188,10 @@ static void vduse_vq_irq_inject(struct work_struct *work) struct vduse_virtqueue *vq = container_of(work, struct vduse_virtqueue, inject); - spin_lock_bh(&vq->irq_lock); + guard(spinlock_bh)(&vq->irq_lock); + guard(spinlock_bh)(&vq->ready_lock); if (vq->ready && vq->cb.callback) vq->cb.callback(vq->cb.private); - spin_unlock_bh(&vq->irq_lock); } static bool vduse_vq_signal_irqfd(struct vduse_virtqueue *vq) @@ -1172,12 +1201,12 @@ static bool vduse_vq_signal_irqfd(struct vduse_virtqueue *vq) if (!vq->cb.trigger) return false; - spin_lock_irq(&vq->irq_lock); + guard(spinlock_irq)(&vq->irq_lock); + guard(spinlock_irq)(&vq->ready_lock); if (vq->ready && vq->cb.trigger) { eventfd_signal(vq->cb.trigger); signal = true; } - spin_unlock_irq(&vq->irq_lock); return signal; } @@ -1515,7 +1544,9 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd, vq_info.split.avail_index = vq->state.split.avail_index; - vq_info.ready = vq->ready; + scoped_guard(spinlock_bh, &vq->ready_lock) { + vq_info.ready = vq->ready; + } ret = -EFAULT; if (copy_to_user(argp, &vq_info, sizeof(vq_info))) @@ -1745,7 +1776,9 @@ static long vduse_dev_compat_ioctl(struct file *file, unsigned int cmd, vq_info.split.avail_index = vq->state.split.avail_index; - vq_info.ready = vq->ready; + scoped_guard(spinlock_bh, &vq->ready_lock) { + vq_info.ready = vq->ready; + } ret = -EFAULT; if (copy_to_user(argp, &vq_info, @@ -1958,6 +1991,7 @@ static int vduse_dev_init_vqs(struct vduse_dev *dev, u32 vq_align, u32 vq_num) INIT_WORK(&dev->vqs[i]->kick, vduse_vq_kick_work); spin_lock_init(&dev->vqs[i]->kick_lock); spin_lock_init(&dev->vqs[i]->irq_lock); + spin_lock_init(&dev->vqs[i]->ready_lock); cpumask_setall(&dev->vqs[i]->irq_affinity); kobject_init(&dev->vqs[i]->kobj, &vq_type); @@ -2193,7 +2227,8 @@ static struct attribute *vduse_dev_attrs[] = { ATTRIBUTE_GROUPS(vduse_dev); static int vduse_create_dev(struct vduse_dev_config *config, - void *config_buf, u64 api_version) + void *config_buf, u64 api_version, + uint64_t vduse_features) { int ret; struct vduse_dev *dev; @@ -2215,6 +2250,9 @@ static int vduse_create_dev(struct vduse_dev_config *config, dev->device_features = config->features; dev->device_id = config->device_id; dev->vendor_id = config->vendor_id; + dev->vduse_features = vduse_features; + dev_dbg(vduse_ctrl_dev, "Creating device %s with features 0x%llx", + config->name, vduse_features); dev->nas = (dev->api_version < VDUSE_API_VERSION_1) ? 1 : config->nas; dev->as = kzalloc_objs(dev->as[0], dev->nas); @@ -2330,7 +2368,8 @@ static long vduse_ioctl(struct file *file, unsigned int cmd, break; } config.name[VDUSE_NAME_MAX - 1] = '\0'; - ret = vduse_create_dev(&config, buf, control->api_version); + ret = vduse_create_dev(&config, buf, control->api_version, + control->vduse_features); if (ret) kvfree(buf); break; diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h index f14c965bb7f68d..7285f8570237bf 100644 --- a/include/uapi/linux/vduse.h +++ b/include/uapi/linux/vduse.h @@ -14,6 +14,9 @@ #define VDUSE_API_VERSION_1 1 +/* The VDUSE instance expects a request for vq ready */ +#define VDUSE_F_QUEUE_READY 0 + /* * Get the version of VDUSE API that kernel supported (VDUSE_API_VERSION). * This is used for future extension. @@ -331,6 +334,7 @@ enum vduse_req_type { VDUSE_SET_STATUS, VDUSE_UPDATE_IOTLB, VDUSE_SET_VQ_GROUP_ASID, + VDUSE_SET_VQ_READY, }; /** @@ -378,6 +382,15 @@ struct vduse_iova_range_v2 { __u32 padding; }; +/** + * struct vduse_vq_ready - Virtqueue ready request message + * @num: Virtqueue number + */ +struct vduse_vq_ready { + __u32 num; + __u32 ready; +}; + /** * struct vduse_dev_request - control request * @type: request type @@ -388,6 +401,7 @@ struct vduse_iova_range_v2 { * @iova: IOVA range for updating * @iova_v2: IOVA range for updating if API_VERSION >= 1 * @vq_group_asid: ASID of a virtqueue group + * @vq_ready: Virtqueue ready request * @padding: padding * * Structure used by read(2) on /dev/vduse/$NAME. @@ -405,6 +419,10 @@ struct vduse_dev_request { */ struct vduse_iova_range_v2 iova_v2; struct vduse_vq_group_asid vq_group_asid; + + /* Only if VDUSE_F_QUEUE_READY is negotiated */ + struct vduse_vq_ready vq_ready; + __u32 padding[32]; }; }; From 78bbdaecb2d8ac460a69175dd61c27bf77e7e2f8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Eugenio=20P=C3=A9rez?= Date: Tue, 7 Jul 2026 14:33:43 +0200 Subject: [PATCH 094/857] vduse: do not take rwsem at reset work flush MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Next patches need to check suspend flag at this work item, and the rwlock is used to protect the suspend flag update. If the work takes the rwlock too it will produce a deadlock. Make flushing work do nothing when called by de-initializing everything: vq->ready, vq->kickfd, vq->cb.callback. Signed-off-by: Eugenio Pérez Signed-off-by: Michael S. Tsirkin Message-ID: <20260707123344.244575-2-eperezma@redhat.com> --- drivers/vdpa/vdpa_user/vduse_dev.c | 67 ++++++++++++++++-------------- 1 file changed, 35 insertions(+), 32 deletions(-) diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c index 87d6748b50cc6d..9aff26fbb583b0 100644 --- a/drivers/vdpa/vdpa_user/vduse_dev.c +++ b/drivers/vdpa/vdpa_user/vduse_dev.c @@ -502,46 +502,49 @@ static void vduse_dev_reset(struct vduse_dev *dev) vduse_domain_reset_bounce_map(domain); } - down_write(&dev->rwsem); + scoped_guard(rwsem_write, &dev->rwsem) { + dev->status = 0; + dev->driver_features = 0; + dev->generation++; + spin_lock(&dev->irq_lock); + dev->config_cb.callback = NULL; + dev->config_cb.private = NULL; + spin_unlock(&dev->irq_lock); + + for (i = 0; i < dev->vq_num; i++) { + struct vduse_virtqueue *vq = dev->vqs[i]; + + scoped_guard(spinlock_bh, &vq->ready_lock) { + vq->ready = false; + } + vq->desc_addr = 0; + vq->driver_addr = 0; + vq->device_addr = 0; + vq->num = 0; + memset(&vq->state, 0, sizeof(vq->state)); + + spin_lock(&vq->kick_lock); + vq->kicked = false; + if (vq->kickfd) + eventfd_ctx_put(vq->kickfd); + vq->kickfd = NULL; + spin_unlock(&vq->kick_lock); + + spin_lock(&vq->irq_lock); + vq->cb.callback = NULL; + vq->cb.private = NULL; + vq->cb.trigger = NULL; + spin_unlock(&vq->irq_lock); + } + } - dev->status = 0; - dev->driver_features = 0; - dev->generation++; - spin_lock(&dev->irq_lock); - dev->config_cb.callback = NULL; - dev->config_cb.private = NULL; - spin_unlock(&dev->irq_lock); flush_work(&dev->inject); - for (i = 0; i < dev->vq_num; i++) { struct vduse_virtqueue *vq = dev->vqs[i]; - scoped_guard(spinlock_bh, &vq->ready_lock) { - vq->ready = false; - } - vq->desc_addr = 0; - vq->driver_addr = 0; - vq->device_addr = 0; - vq->num = 0; - memset(&vq->state, 0, sizeof(vq->state)); - - spin_lock(&vq->kick_lock); - vq->kicked = false; - if (vq->kickfd) - eventfd_ctx_put(vq->kickfd); - vq->kickfd = NULL; - spin_unlock(&vq->kick_lock); - - spin_lock(&vq->irq_lock); - vq->cb.callback = NULL; - vq->cb.private = NULL; - vq->cb.trigger = NULL; - spin_unlock(&vq->irq_lock); flush_work(&vq->inject); flush_work(&vq->kick); } - - up_write(&dev->rwsem); } static int vduse_vdpa_set_vq_address(struct vdpa_device *vdpa, u16 idx, From ae864da7d762b4c692e2cdf83cad43110d5d46ee Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Eugenio=20P=C3=A9rez?= Date: Tue, 7 Jul 2026 14:33:44 +0200 Subject: [PATCH 095/857] vduse: Add suspend MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Implement suspend operation for vduse devices, so vhost-vdpa will offer that backend feature and userspace can effectively suspend the device. This is a must before get virtqueue indexes (base) for live migration, since the device could modify them after userland gets them. This patch does not implement resume, so VMM resets the whole device to recover from a live migration failure. Resume optimization can be implemented on top of these patches, as other vDPA devices have done in the past. Signed-off-by: Eugenio Pérez Signed-off-by: Michael S. Tsirkin Message-ID: <20260707123344.244575-3-eperezma@redhat.com> --- drivers/vdpa/vdpa_user/vduse_dev.c | 95 +++++++++++++++++++++++++++--- include/uapi/linux/vduse.h | 4 ++ 2 files changed, 92 insertions(+), 7 deletions(-) diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c index 9aff26fbb583b0..9891cd2cf7121b 100644 --- a/drivers/vdpa/vdpa_user/vduse_dev.c +++ b/drivers/vdpa/vdpa_user/vduse_dev.c @@ -53,7 +53,8 @@ #define IRQ_UNBOUND -1 /* Supported VDUSE features */ -static const uint64_t vduse_features = BIT_U64(VDUSE_F_QUEUE_READY); +static const uint64_t vduse_features = BIT_U64(VDUSE_F_QUEUE_READY) | + BIT_U64(VDUSE_F_SUSPEND); /* * VDUSE instance have not asked the vduse API version, so assume 0. @@ -85,6 +86,7 @@ struct vduse_virtqueue { int irq_effective_cpu; struct cpumask irq_affinity; struct kobject kobj; + struct vduse_dev *dev; }; struct vduse_dev; @@ -134,6 +136,7 @@ struct vduse_dev { int minor; bool broken; bool connected; + bool suspended; u64 api_version; u64 device_features; u64 driver_features; @@ -503,6 +506,7 @@ static void vduse_dev_reset(struct vduse_dev *dev) } scoped_guard(rwsem_write, &dev->rwsem) { + dev->suspended = false; dev->status = 0; dev->driver_features = 0; dev->generation++; @@ -563,6 +567,10 @@ static int vduse_vdpa_set_vq_address(struct vdpa_device *vdpa, u16 idx, static void vduse_vq_kick(struct vduse_virtqueue *vq) { + guard(rwsem_read)(&vq->dev->rwsem); + if (vq->dev->suspended) + return; + guard(spinlock)(&vq->kick_lock); scoped_guard(spinlock_bh, &vq->ready_lock) if (!vq->ready) @@ -927,6 +935,27 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa, return 0; } +static int vduse_vdpa_suspend(struct vdpa_device *vdpa) +{ + struct vduse_dev *dev = vdpa_to_vduse(vdpa); + struct vduse_dev_msg msg = { 0 }; + int ret; + + msg.req.type = VDUSE_SUSPEND; + + ret = vduse_dev_msg_sync(dev, &msg); + if (ret == 0) { + scoped_guard(rwsem_write, &dev->rwsem) + dev->suspended = true; + + cancel_work_sync(&dev->inject); + for (u32 i = 0; i < dev->vq_num; i++) + cancel_work_sync(&dev->vqs[i]->inject); + } + + return ret; +} + static void vduse_vdpa_free(struct vdpa_device *vdpa) { struct vduse_dev *dev = vdpa_to_vduse(vdpa); @@ -968,6 +997,41 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = { .free = vduse_vdpa_free, }; +static const struct vdpa_config_ops vduse_vdpa_config_ops_with_suspend = { + .set_vq_address = vduse_vdpa_set_vq_address, + .kick_vq = vduse_vdpa_kick_vq, + .set_vq_cb = vduse_vdpa_set_vq_cb, + .set_vq_num = vduse_vdpa_set_vq_num, + .get_vq_size = vduse_vdpa_get_vq_size, + .get_vq_group = vduse_get_vq_group, + .set_vq_ready = vduse_vdpa_set_vq_ready, + .get_vq_ready = vduse_vdpa_get_vq_ready, + .set_vq_state = vduse_vdpa_set_vq_state, + .get_vq_state = vduse_vdpa_get_vq_state, + .get_vq_align = vduse_vdpa_get_vq_align, + .get_device_features = vduse_vdpa_get_device_features, + .set_driver_features = vduse_vdpa_set_driver_features, + .get_driver_features = vduse_vdpa_get_driver_features, + .set_config_cb = vduse_vdpa_set_config_cb, + .get_vq_num_max = vduse_vdpa_get_vq_num_max, + .get_device_id = vduse_vdpa_get_device_id, + .get_vendor_id = vduse_vdpa_get_vendor_id, + .get_status = vduse_vdpa_get_status, + .set_status = vduse_vdpa_set_status, + .get_config_size = vduse_vdpa_get_config_size, + .get_config = vduse_vdpa_get_config, + .set_config = vduse_vdpa_set_config, + .get_generation = vduse_vdpa_get_generation, + .set_vq_affinity = vduse_vdpa_set_vq_affinity, + .get_vq_affinity = vduse_vdpa_get_vq_affinity, + .reset = vduse_vdpa_reset, + .set_map = vduse_vdpa_set_map, + .set_group_asid = vduse_set_group_asid, + .get_vq_map = vduse_get_vq_map, + .suspend = vduse_vdpa_suspend, + .free = vduse_vdpa_free, +}; + static void vduse_dev_sync_single_for_device(union virtio_map token, dma_addr_t dma_addr, size_t size, enum dma_data_direction dir) @@ -1180,6 +1244,10 @@ static void vduse_dev_irq_inject(struct work_struct *work) { struct vduse_dev *dev = container_of(work, struct vduse_dev, inject); + guard(rwsem_read)(&dev->rwsem); + if (dev->suspended) + return; + spin_lock_bh(&dev->irq_lock); if (dev->config_cb.callback) dev->config_cb.callback(dev->config_cb.private); @@ -1191,6 +1259,10 @@ static void vduse_vq_irq_inject(struct work_struct *work) struct vduse_virtqueue *vq = container_of(work, struct vduse_virtqueue, inject); + guard(rwsem_read)(&vq->dev->rwsem); + if (vq->dev->suspended) + return; + guard(spinlock_bh)(&vq->irq_lock); guard(spinlock_bh)(&vq->ready_lock); if (vq->ready && vq->cb.callback) @@ -1201,6 +1273,10 @@ static bool vduse_vq_signal_irqfd(struct vduse_virtqueue *vq) { bool signal = false; + guard(rwsem_read)(&vq->dev->rwsem); + if (vq->dev->suspended) + return false; + if (!vq->cb.trigger) return false; @@ -1220,9 +1296,9 @@ static int vduse_dev_queue_irq_work(struct vduse_dev *dev, { int ret = -EINVAL; - down_read(&dev->rwsem); - if (!(dev->status & VIRTIO_CONFIG_S_DRIVER_OK)) - goto unlock; + guard(rwsem_read)(&dev->rwsem); + if (dev->suspended || !(dev->status & VIRTIO_CONFIG_S_DRIVER_OK)) + return ret; ret = 0; if (irq_effective_cpu == IRQ_UNBOUND) @@ -1230,8 +1306,6 @@ static int vduse_dev_queue_irq_work(struct vduse_dev *dev, else queue_work_on(irq_effective_cpu, vduse_irq_bound_wq, irq_work); -unlock: - up_read(&dev->rwsem); return ret; } @@ -1989,6 +2063,7 @@ static int vduse_dev_init_vqs(struct vduse_dev *dev, u32 vq_align, u32 vq_num) } dev->vqs[i]->index = i; + dev->vqs[i]->dev = dev; dev->vqs[i]->irq_effective_cpu = IRQ_UNBOUND; INIT_WORK(&dev->vqs[i]->inject, vduse_vq_irq_inject); INIT_WORK(&dev->vqs[i]->kick, vduse_vq_kick_work); @@ -2465,12 +2540,18 @@ static struct vduse_mgmt_dev *vduse_mgmt; static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name) { struct vduse_vdpa *vdev; + const struct vdpa_config_ops *ops; if (dev->vdev) return -EEXIST; + if (dev->vduse_features & BIT_U64(VDUSE_F_SUSPEND)) + ops = &vduse_vdpa_config_ops_with_suspend; + else + ops = &vduse_vdpa_config_ops; + vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev, - &vduse_vdpa_config_ops, &vduse_map_ops, + ops, &vduse_map_ops, dev->ngroups, dev->nas, name, true); if (IS_ERR(vdev)) return PTR_ERR(vdev); diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h index 7285f8570237bf..b7f8c04a0a4443 100644 --- a/include/uapi/linux/vduse.h +++ b/include/uapi/linux/vduse.h @@ -17,6 +17,9 @@ /* The VDUSE instance expects a request for vq ready */ #define VDUSE_F_QUEUE_READY 0 +/* The VDUSE instance expects a request for suspend */ +#define VDUSE_F_SUSPEND 1 + /* * Get the version of VDUSE API that kernel supported (VDUSE_API_VERSION). * This is used for future extension. @@ -335,6 +338,7 @@ enum vduse_req_type { VDUSE_UPDATE_IOTLB, VDUSE_SET_VQ_GROUP_ASID, VDUSE_SET_VQ_READY, + VDUSE_SUSPEND, }; /** From 81aa917982c10b9f2cd0dd767d7af5af6057d93d Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Tue, 4 Aug 2026 22:12:39 +0200 Subject: [PATCH 096/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 627e782b02af54..43d861f1c427c0 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -16,6 +16,8 @@ soc/arm https://github.com/vzapolskiy/linux-lpc32xx tags/lpc32xx-arm-for-7.3 imx/soc https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux tags/imx-soc-7.3 + tegra/arm + git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.3-arm-core soc/dt patch @@ -61,10 +63,26 @@ soc/drivers https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux tags/samsung-drivers-7.3 mediatek/soc https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-soc-for-v7.3 + amlogic/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux tags/amlogic-drivers-for-v7.3 + tegra/soc-drivers + git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.3-soc + omap/drivers + git://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap tags/omap-for-v7.3/drivers-signed + aspeed/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/aspeed-7.3-drivers-0 + qcom/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-drivers-for-7.3 soc/defconfig riscv/defconfig https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux tags/riscv-config-for-v7.3 + qcom/defconfig + https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm64-defconfig-for-7.3 + omap/defconfig + git://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap tags/omap-for-v7.3/defconfig-signed + allwinner/defconfig + https://git.kernel.org/pub/scm/linux/kernel/git/sunxi/linux tags/sunxi-config-for-7.3 soc/late @@ -75,4 +93,12 @@ arm/fixes https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux tags/imx-maintainers-7.3 qcom/dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm64-fixes-for-7.2 + broadcom/fixes-2 + https://github.com/Broadcom/stblinux tags/arm-soc/for-7.2/devicetree-fixes-v2 + aspeed/fixes + https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/aspeed-7.2-driver-fixes-0 + nuvoton/fixes + https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/nuvoton-7.2-arm-fixes-0 + nuvoton/maintainers + https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/aspeed-7.3-maintainers-0 From b83637163279a2a69a559c8789cd2197d9ae0104 Mon Sep 17 00:00:00 2001 From: Francesco Dolcini Date: Thu, 23 Jul 2026 12:52:17 +0200 Subject: [PATCH 097/857] ARM: dts: imx7d-colibri-emmc: Add Toradex Capacitive Touch Display 7" Parallel Add a device tree overlay for the Capacitive Touch Display 7" Parallel on the Colibri iMX7 parallel RGB LCD interface. The panel is a LogicTechno LT161010-2NHC 7" WVGA TFT Transmissive LCD and the touch input is provided by an Atmel MaxTouch capacitive touch controller. The touch controller is connected to the Toradex 10-way Capacitive Touch Interface connector available on Iris v2 and Aster carrier boards. The overlay is also combined with the Iris v2 and Aster carrier board device tree to provide a ready-to-use DTB. Link: https://developer.toradex.com/hardware/accessories/displays/capacitive-touch-display-7inch-parallel/ Signed-off-by: Francesco Dolcini Signed-off-by: Frank Li --- arch/arm/boot/dts/nxp/imx/Makefile | 12 ++++++ ...i-emmc-panel-cap-touch-7inch-parallel.dtso | 43 +++++++++++++++++++ 2 files changed, 55 insertions(+) create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtso diff --git a/arch/arm/boot/dts/nxp/imx/Makefile b/arch/arm/boot/dts/nxp/imx/Makefile index 1a2539fa19b447..4d273000020256 100644 --- a/arch/arm/boot/dts/nxp/imx/Makefile +++ b/arch/arm/boot/dts/nxp/imx/Makefile @@ -419,13 +419,25 @@ dtb-$(CONFIG_SOC_IMX6UL) += \ imx6ull-var-som-concerto-full.dtb \ imx6ulz-14x14-evk.dtb \ imx6ulz-bsh-smm-m2.dtb + +imx7d-colibri-emmc-aster-panel-cap-touch-7inch-parallel-dtbs := \ + imx7d-colibri-emmc-aster.dtb \ + imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtbo + +imx7d-colibri-emmc-iris-v2-panel-cap-touch-7inch-parallel-dtbs := \ + imx7d-colibri-emmc-iris-v2.dtb \ + imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtbo + dtb-$(CONFIG_SOC_IMX7D) += \ imx7d-cl-som-imx7.dtb \ imx7d-colibri-aster.dtb \ imx7d-colibri-emmc-aster.dtb \ + imx7d-colibri-emmc-aster-panel-cap-touch-7inch-parallel.dtb \ imx7d-colibri-emmc-iris.dtb \ imx7d-colibri-emmc-iris-v2.dtb \ + imx7d-colibri-emmc-iris-v2-panel-cap-touch-7inch-parallel.dtb \ imx7d-colibri-emmc-eval-v3.dtb \ + imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtbo \ imx7d-colibri-eval-v3.dtb \ imx7d-colibri-iris.dtb \ imx7d-colibri-iris-v2.dtb \ diff --git a/arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtso b/arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtso new file mode 100644 index 00000000000000..f336f7b05a2b71 --- /dev/null +++ b/arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtso @@ -0,0 +1,43 @@ +// SPDX-License-Identifier: GPL-2.0-or-later OR MIT +/* + * Copyright (c) Toradex + * + * Toradex Capacitive Touch Display 7" Parallel connected through the 40-way + * Unified Interface Display and 10-way Capacitive Touch Interface connectors + * (as featured on Iris v2 and Aster v1.1 carrier boards). + * + * https://docs.toradex.com/104497-7-inch-parallel-capacitive-touch-display-800x480-datasheet.pdf + * https://developer.toradex.com/hardware/accessories/displays/capacitive-touch-display-7inch-parallel/ + * https://www.toradex.com/accessories/capacitive-touch-display-7-inch-parallel + */ + +/dts-v1/; +/plugin/; + +#include +#include + +&atmel_mxt_ts { + pinctrl-0 = <&pinctrl_atmel_connector>; + /* SODIMM 107 / TOUCH_INT# */ + interrupt-parent = <&gpio2>; + interrupts = <15 IRQ_TYPE_EDGE_FALLING>; + /* SODIMM 106 / TOUCH_RST# */ + reset-gpios = <&gpio2 28 GPIO_ACTIVE_LOW>; + + status = "okay"; +}; + +&backlight { + status = "okay"; +}; + +&lcdif { + status = "okay"; +}; + +&panel_dpi { + compatible = "logictechno,lt161010-2nhc"; + + status = "okay"; +}; From ce580e453d27327c79ac8e95dd3533a325cd4202 Mon Sep 17 00:00:00 2001 From: Francesco Dolcini Date: Thu, 23 Jul 2026 12:52:18 +0200 Subject: [PATCH 098/857] ARM: dts: imx7d-colibri-emmc: Add Toradex Capacitive Touch Display 7" Parallel with Touch Adapter Add a device tree overlay for the Capacitive Touch Display 7" Parallel on the Colibri iMX7 parallel RGB LCD interface. The panel is a LogicTechno LT161010-2NHC 7" WVGA TFT Transmissive LCD and the touch input is provided by an Atmel MaxTouch capacitive touch controller. The touch controller is connected to the Toradex Capacitive Touch Adapter, the connection to the various carrier boards is documented in the datasheet [1]. The overlay is also combined with the Eval carrier board device tree to provide a ready-to-use DTB. Link: https://developer.toradex.com/hardware/accessories/displays/capacitive-touch-display-7inch-parallel/ Link: https://developer.toradex.com/hardware/accessories/add-ons/capacitive-touch-adapter/ Link: https://docs.toradex.com/104615-capacitive-touch-adapter-datasheet.pdf [1] Signed-off-by: Francesco Dolcini Signed-off-by: Frank Li --- arch/arm/boot/dts/nxp/imx/Makefile | 5 ++ ...ap-touch-7inch-parallel-touch-adapter.dtso | 58 +++++++++++++++++++ 2 files changed, 63 insertions(+) create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel-touch-adapter.dtso diff --git a/arch/arm/boot/dts/nxp/imx/Makefile b/arch/arm/boot/dts/nxp/imx/Makefile index 4d273000020256..3c16843c307149 100644 --- a/arch/arm/boot/dts/nxp/imx/Makefile +++ b/arch/arm/boot/dts/nxp/imx/Makefile @@ -428,6 +428,10 @@ imx7d-colibri-emmc-iris-v2-panel-cap-touch-7inch-parallel-dtbs := \ imx7d-colibri-emmc-iris-v2.dtb \ imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtbo +imx7d-colibri-emmc-eval-v3-panel-cap-touch-7inch-parallel-touch-adapter-dtbs := \ + imx7d-colibri-emmc-eval-v3.dtb \ + imx7d-colibri-emmc-panel-cap-touch-7inch-parallel-touch-adapter.dtbo + dtb-$(CONFIG_SOC_IMX7D) += \ imx7d-cl-som-imx7.dtb \ imx7d-colibri-aster.dtb \ @@ -437,6 +441,7 @@ dtb-$(CONFIG_SOC_IMX7D) += \ imx7d-colibri-emmc-iris-v2.dtb \ imx7d-colibri-emmc-iris-v2-panel-cap-touch-7inch-parallel.dtb \ imx7d-colibri-emmc-eval-v3.dtb \ + imx7d-colibri-emmc-eval-v3-panel-cap-touch-7inch-parallel-touch-adapter.dtb \ imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtbo \ imx7d-colibri-eval-v3.dtb \ imx7d-colibri-iris.dtb \ diff --git a/arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel-touch-adapter.dtso b/arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel-touch-adapter.dtso new file mode 100644 index 00000000000000..890461f92442e5 --- /dev/null +++ b/arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel-touch-adapter.dtso @@ -0,0 +1,58 @@ +// SPDX-License-Identifier: GPL-2.0-or-later OR MIT +/* + * Copyright (c) Toradex + * + * Toradex Capacitive Touch Display 7" Parallel connected through the 40-way + * Unified Interface Display connector and Toradex Capacitive Touch Adapter. + * + * Toradex Capacitive Touch Adapter connection with the various carrier boards + * is documented in the datasheet [1]. + * + * https://developer.toradex.com/hardware/accessories/displays/capacitive-touch-display-7inch-parallel/ + * https://www.toradex.com/accessories/capacitive-touch-display-7-inch-parallel + * https://docs.toradex.com/104497-7-inch-parallel-capacitive-touch-display-800x480-datasheet.pdf + * https://developer.toradex.com/hardware/accessories/add-ons/capacitive-touch-adapter/ + * https://www.toradex.com/accessories/capacitive-touch-adapter + * https://docs.toradex.com/104615-capacitive-touch-adapter-datasheet.pdf [1] + */ + +/dts-v1/; +/plugin/; + +#include +#include + +&atmel_mxt_ts { + pinctrl-0 = <&pinctrl_atmel_adapter>; + /* SODIMM 28 / TOUCH_INT# */ + interrupt-parent = <&gpio1>; + interrupts = <9 IRQ_TYPE_EDGE_FALLING>; + /* SODIMM 30 / TOUCH_RST# */ + reset-gpios = <&gpio1 10 GPIO_ACTIVE_LOW>; + + status = "okay"; +}; + +&backlight { + status = "okay"; +}; + +&lcdif { + status = "okay"; +}; + +&panel_dpi { + compatible = "logictechno,lt161010-2nhc"; + + status = "okay"; +}; + +/* Conflict with SODIMM 28 / TOUCH_INT# */ +&pwm2 { + status = "disabled"; +}; + +/* Conflict with SODIMM 30 / TOUCH_RST# */ +&pwm3 { + status = "disabled"; +}; From f37a9b80fad87bf39b66a3f5b4dcf0d0d91c9994 Mon Sep 17 00:00:00 2001 From: Francesco Dolcini Date: Thu, 23 Jul 2026 12:52:19 +0200 Subject: [PATCH 099/857] ARM: dts: imx7d-colibri-emmc: Add Toradex Resistive Touch Display 7" Parallel Add a device tree overlay for the Resistive Touch Display 7" Parallel on the Colibri iMX7 parallel RGB LCD interface. The panel is a LogicTechno LT161010-2NHR 7" WVGA TFT Transmissive LCD with a resistive touch. The overlay is also combined with the Eval carrier board device tree to provide a ready-to-use DTB. Link: https://developer.toradex.com/hardware/accessories/displays/resistive-touch-display-7inch-parallel/ Signed-off-by: Francesco Dolcini Signed-off-by: Frank Li --- arch/arm/boot/dts/nxp/imx/Makefile | 5 +++ ...i-emmc-panel-res-touch-7inch-parallel.dtso | 32 +++++++++++++++++++ 2 files changed, 37 insertions(+) create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-res-touch-7inch-parallel.dtso diff --git a/arch/arm/boot/dts/nxp/imx/Makefile b/arch/arm/boot/dts/nxp/imx/Makefile index 3c16843c307149..e9f95c3b83534c 100644 --- a/arch/arm/boot/dts/nxp/imx/Makefile +++ b/arch/arm/boot/dts/nxp/imx/Makefile @@ -432,6 +432,10 @@ imx7d-colibri-emmc-eval-v3-panel-cap-touch-7inch-parallel-touch-adapter-dtbs := imx7d-colibri-emmc-eval-v3.dtb \ imx7d-colibri-emmc-panel-cap-touch-7inch-parallel-touch-adapter.dtbo +imx7d-colibri-emmc-eval-v3-panel-res-touch-7inch-parallel-dtbs := \ + imx7d-colibri-emmc-eval-v3.dtb \ + imx7d-colibri-emmc-panel-res-touch-7inch-parallel.dtbo + dtb-$(CONFIG_SOC_IMX7D) += \ imx7d-cl-som-imx7.dtb \ imx7d-colibri-aster.dtb \ @@ -442,6 +446,7 @@ dtb-$(CONFIG_SOC_IMX7D) += \ imx7d-colibri-emmc-iris-v2-panel-cap-touch-7inch-parallel.dtb \ imx7d-colibri-emmc-eval-v3.dtb \ imx7d-colibri-emmc-eval-v3-panel-cap-touch-7inch-parallel-touch-adapter.dtb \ + imx7d-colibri-emmc-eval-v3-panel-res-touch-7inch-parallel.dtb \ imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtbo \ imx7d-colibri-eval-v3.dtb \ imx7d-colibri-iris.dtb \ diff --git a/arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-res-touch-7inch-parallel.dtso b/arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-res-touch-7inch-parallel.dtso new file mode 100644 index 00000000000000..b926cc020e6268 --- /dev/null +++ b/arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-res-touch-7inch-parallel.dtso @@ -0,0 +1,32 @@ +// SPDX-License-Identifier: GPL-2.0-or-later OR MIT +/* + * Copyright (c) Toradex + * + * Toradex Resistive Touch Display 7" Parallel connected through the 40-way + * Unified Interface Display connector. + * + * https://docs.toradex.com/104498-7-inch-parallel-resistive-touch-display-800x480.pdf + * https://developer.toradex.com/hardware/accessories/displays/resistive-touch-display-7inch-parallel/ + * https://www.toradex.com/accessories/resistive-touch-display + */ + +/dts-v1/; +/plugin/; + +&ad7879_ts { + status = "okay"; +}; + +&backlight { + status = "okay"; +}; + +&lcdif { + status = "okay"; +}; + +&panel_dpi { + compatible = "logictechno,lt161010-2nhr"; + + status = "okay"; +}; From e297c55e6d89100b32e1b7ea6a9ab0cd2a36893d Mon Sep 17 00:00:00 2001 From: Krzysztof Kozlowski Date: Sat, 1 Aug 2026 23:04:21 +0200 Subject: [PATCH 100/857] ARM: dts: nxp: Correct white-space style Correct a few white-space issues, like missing space before bracket '{' character or spurious space, which will be flagged by dt-check-style ("redundant-whitespace" warning). No functional changes. Signed-off-by: Krzysztof Kozlowski Reviewed-by: Alexander Stein Signed-off-by: Frank Li --- arch/arm/boot/dts/nxp/imx/e60k02.dtsi | 8 ++++---- arch/arm/boot/dts/nxp/imx/e70k02.dtsi | 8 ++++---- arch/arm/boot/dts/nxp/imx/imx6dl-mamoj.dts | 2 +- arch/arm/boot/dts/nxp/imx/imx6q-bosch-acc.dts | 4 ++-- arch/arm/boot/dts/nxp/imx/imx6sl-kobo-aura2.dts | 8 ++++---- arch/arm/boot/dts/nxp/imx/imx6sl-tolino-shine2hd.dts | 8 ++++---- arch/arm/boot/dts/nxp/imx/imx6sll-evk.dts | 2 +- arch/arm/boot/dts/nxp/imx/imx7s-warp.dts | 2 +- arch/arm/boot/dts/nxp/ls/ls1021a.dtsi | 10 +++++----- 9 files changed, 26 insertions(+), 26 deletions(-) diff --git a/arch/arm/boot/dts/nxp/imx/e60k02.dtsi b/arch/arm/boot/dts/nxp/imx/e60k02.dtsi index aac7b9ef762753..ed59b27834a6d3 100644 --- a/arch/arm/boot/dts/nxp/imx/e60k02.dtsi +++ b/arch/arm/boot/dts/nxp/imx/e60k02.dtsi @@ -240,13 +240,13 @@ }; /* IR_3V3 */ - ldo1_reg: LDO1 { + ldo1_reg: LDO1 { regulator-name = "LDO1"; regulator-boot-on; }; /* Core1_3V3 */ - ldo2_reg: LDO2 { + ldo2_reg: LDO2 { regulator-name = "LDO2"; regulator-always-on; regulator-boot-on; @@ -259,7 +259,7 @@ }; /* Core5_1V2 */ - ldo3_reg: LDO3 { + ldo3_reg: LDO3 { regulator-name = "LDO3"; regulator-always-on; regulator-boot-on; @@ -310,7 +310,7 @@ regulator-boot-on; }; - ldortc1_reg: LDORTC1 { + ldortc1_reg: LDORTC1 { regulator-name = "LDORTC1"; regulator-boot-on; }; diff --git a/arch/arm/boot/dts/nxp/imx/e70k02.dtsi b/arch/arm/boot/dts/nxp/imx/e70k02.dtsi index 3bb11c5a63536d..dc7ff108aa8534 100644 --- a/arch/arm/boot/dts/nxp/imx/e70k02.dtsi +++ b/arch/arm/boot/dts/nxp/imx/e70k02.dtsi @@ -243,13 +243,13 @@ }; }; - ldo1_reg: LDO1 { + ldo1_reg: LDO1 { regulator-name = "LDO1"; regulator-boot-on; }; /* Core1_3V3 */ - ldo2_reg: LDO2 { + ldo2_reg: LDO2 { regulator-name = "LDO2"; regulator-always-on; regulator-boot-on; @@ -262,7 +262,7 @@ }; /* Core5_1V2 */ - ldo3_reg: LDO3 { + ldo3_reg: LDO3 { regulator-name = "LDO3"; regulator-always-on; regulator-boot-on; @@ -311,7 +311,7 @@ regulator-boot-on; }; - ldortc1_reg: LDORTC1 { + ldortc1_reg: LDORTC1 { regulator-name = "LDORTC1"; regulator-boot-on; }; diff --git a/arch/arm/boot/dts/nxp/imx/imx6dl-mamoj.dts b/arch/arm/boot/dts/nxp/imx/imx6dl-mamoj.dts index ec5a9bf40677df..30275319902ec4 100644 --- a/arch/arm/boot/dts/nxp/imx/imx6dl-mamoj.dts +++ b/arch/arm/boot/dts/nxp/imx/imx6dl-mamoj.dts @@ -151,7 +151,7 @@ enable-active-high; }; - reg_wl18xx_vmmc: regulator-wl18xx-vmcc { + reg_wl18xx_vmmc: regulator-wl18xx-vmcc { compatible = "regulator-fixed"; regulator-name = "vwl1807"; pinctrl-names = "default"; diff --git a/arch/arm/boot/dts/nxp/imx/imx6q-bosch-acc.dts b/arch/arm/boot/dts/nxp/imx/imx6q-bosch-acc.dts index 929def2bb35ebb..20b6cc74f622ad 100644 --- a/arch/arm/boot/dts/nxp/imx/imx6q-bosch-acc.dts +++ b/arch/arm/boot/dts/nxp/imx/imx6q-bosch-acc.dts @@ -177,7 +177,7 @@ regulator-name = "usb_h2_vbus"; regulator-min-microvolt = <5000000>; regulator-max-microvolt = <5000000>; - vin-supply = <®_5p0> ; + vin-supply = <®_5p0>; regulator-always-on; }; @@ -207,7 +207,7 @@ regulator-name = "vref_dac"; regulator-min-microvolt = <20000>; regulator-max-microvolt = <20000>; - vin-supply = <®_5p0> ; + vin-supply = <®_5p0>; regulator-boot-on; }; diff --git a/arch/arm/boot/dts/nxp/imx/imx6sl-kobo-aura2.dts b/arch/arm/boot/dts/nxp/imx/imx6sl-kobo-aura2.dts index 657d0f1b6115f0..47b05207574e06 100644 --- a/arch/arm/boot/dts/nxp/imx/imx6sl-kobo-aura2.dts +++ b/arch/arm/boot/dts/nxp/imx/imx6sl-kobo-aura2.dts @@ -211,14 +211,14 @@ }; /* IR_3V3 */ - ldo1_reg: LDO1 { + ldo1_reg: LDO1 { regulator-name = "LDO1"; regulator-always-on; regulator-boot-on; }; /* Core1_3V3 */ - ldo2_reg: LDO2 { + ldo2_reg: LDO2 { regulator-name = "LDO2"; regulator-always-on; regulator-boot-on; @@ -231,7 +231,7 @@ }; /* Core5_1V2 */ - ldo3_reg: LDO3 { + ldo3_reg: LDO3 { regulator-name = "LDO3"; regulator-always-on; regulator-boot-on; @@ -282,7 +282,7 @@ regulator-boot-on; }; - ldortc1_reg: LDORTC1 { + ldortc1_reg: LDORTC1 { regulator-name = "LDORTC1"; regulator-always-on; regulator-boot-on; diff --git a/arch/arm/boot/dts/nxp/imx/imx6sl-tolino-shine2hd.dts b/arch/arm/boot/dts/nxp/imx/imx6sl-tolino-shine2hd.dts index 4c655579f43efb..64a74719c216f6 100644 --- a/arch/arm/boot/dts/nxp/imx/imx6sl-tolino-shine2hd.dts +++ b/arch/arm/boot/dts/nxp/imx/imx6sl-tolino-shine2hd.dts @@ -276,13 +276,13 @@ }; /* IR_3V3 */ - ldo1_reg: LDO1 { + ldo1_reg: LDO1 { regulator-name = "LDO1"; regulator-boot-on; }; /* Core1_3V3 */ - ldo2_reg: LDO2 { + ldo2_reg: LDO2 { regulator-name = "LDO2"; regulator-always-on; regulator-boot-on; @@ -295,7 +295,7 @@ }; /* Core5_1V2 */ - ldo3_reg: LDO3 { + ldo3_reg: LDO3 { regulator-name = "LDO3"; regulator-always-on; regulator-boot-on; @@ -346,7 +346,7 @@ regulator-boot-on; }; - ldortc1_reg: LDORTC1 { + ldortc1_reg: LDORTC1 { regulator-name = "LDORTC1"; regulator-always-on; regulator-boot-on; diff --git a/arch/arm/boot/dts/nxp/imx/imx6sll-evk.dts b/arch/arm/boot/dts/nxp/imx/imx6sll-evk.dts index 814401486792ca..d20d2735cab524 100644 --- a/arch/arm/boot/dts/nxp/imx/imx6sll-evk.dts +++ b/arch/arm/boot/dts/nxp/imx/imx6sll-evk.dts @@ -633,7 +633,7 @@ >; }; - pinctrl_wdog1: wdog1grp { + pinctrl_wdog1: wdog1grp { fsl,pins = < MX6SLL_PAD_WDOG_B__WDOG1_B 0x170b0 >; diff --git a/arch/arm/boot/dts/nxp/imx/imx7s-warp.dts b/arch/arm/boot/dts/nxp/imx/imx7s-warp.dts index 25f38acc53501b..b1989013c1a4a5 100644 --- a/arch/arm/boot/dts/nxp/imx/imx7s-warp.dts +++ b/arch/arm/boot/dts/nxp/imx/imx7s-warp.dts @@ -272,7 +272,7 @@ status = "okay"; }; -&uart3 { +&uart3 { pinctrl-names = "default"; pinctrl-0 = <&pinctrl_uart3>; assigned-clocks = <&clks IMX7D_UART3_ROOT_SRC>; diff --git a/arch/arm/boot/dts/nxp/ls/ls1021a.dtsi b/arch/arm/boot/dts/nxp/ls/ls1021a.dtsi index e0b9ea6dd51005..af518fc08809f1 100644 --- a/arch/arm/boot/dts/nxp/ls/ls1021a.dtsi +++ b/arch/arm/boot/dts/nxp/ls/ls1021a.dtsi @@ -721,7 +721,7 @@ ; }; - queue-group@2d14000 { + queue-group@2d14000 { reg = <0x0 0x2d14000 0x0 0x1000>; interrupts = , , @@ -740,14 +740,14 @@ ranges; dma-coherent; - queue-group@2d50000 { + queue-group@2d50000 { reg = <0x0 0x2d50000 0x0 0x1000>; interrupts = , , ; }; - queue-group@2d54000 { + queue-group@2d54000 { reg = <0x0 0x2d54000 0x0 0x1000>; interrupts = , , @@ -766,14 +766,14 @@ ranges; dma-coherent; - queue-group@2d90000 { + queue-group@2d90000 { reg = <0x0 0x2d90000 0x0 0x1000>; interrupts = , , ; }; - queue-group@2d94000 { + queue-group@2d94000 { reg = <0x0 0x2d94000 0x0 0x1000>; interrupts = , , From e361f5b6d08f90eb58c71577aaf94638e1db9288 Mon Sep 17 00:00:00 2001 From: Alexandre Belloni Date: Sat, 8 Aug 2026 22:18:54 +0200 Subject: [PATCH 101/857] soc: document merges Signed-off-by: Alexandre Belloni --- arch/arm/arm-soc-for-next-contents.txt | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 43d861f1c427c0..3390a58eb95015 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -47,6 +47,28 @@ soc/dt https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-dts32-for-v7.3 mediatek/dt64 https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-dts64-for-v7.3 + hisilicon/dts + https://github.com/hisilicon/linux-hisi tags/hisi-arm64-dt-for-7.3 + imx/dt64 + https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux tags/imx-dt64-7.3 + qcom/dt64 + https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm64-for-7.3 + amlogic/dt64 + https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux tags/amlogic-arm64-dt-for-v7.3 + rockchip/dt64 + https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip tags/v7.3-rockchip-dts64-1 + spacemit/dt + https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux tags/spacemit-dt-for-7.3-1 + omap/dt + git://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap tags/omap-for-v7.3/dt-signed + sunxi/dt + https://git.kernel.org/pub/scm/linux/kernel/git/sunxi/linux tags/sunxi-dt-for-7.3 + nuvoton/dt + https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/nuvoton-arm-7.3-devicetree-0 + nuvoton/dt64 + https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/nuvoton-arm64-7.3-devicetree-0 + aspeed/dt32 + https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/aspeed-arm-7.3-devicetree-0 soc/drivers ixp4xx/soc-drivers From b9a43aeabca0162b4a405c1c093e0764a549ba77 Mon Sep 17 00:00:00 2001 From: Alexandre Belloni Date: Sun, 9 Aug 2026 04:02:39 +0200 Subject: [PATCH 102/857] soc: document merges Signed-off-by: Alexandre Belloni --- arch/arm/arm-soc-for-next-contents.txt | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 3390a58eb95015..8cb20f599bd3cb 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -69,6 +69,10 @@ soc/dt https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/nuvoton-arm64-7.3-devicetree-0 aspeed/dt32 https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/aspeed-arm-7.3-devicetree-0 + tegra/dt-bindings + git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.3-dt-bindings + tegra/dt64 + git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.3-arm64-dt soc/drivers ixp4xx/soc-drivers From a31a88d83935868bd3ab19149b493d78fd29ddc5 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Mon, 10 Aug 2026 09:32:10 +0200 Subject: [PATCH 103/857] soc: update fixes branch Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 18 ++++-------------- 1 file changed, 4 insertions(+), 14 deletions(-) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 8cb20f599bd3cb..1fa452947b5fbe 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -113,18 +113,8 @@ soc/defconfig soc/late arm/fixes - broadcom/fixes - https://github.com/Broadcom/stblinux tags/arm-soc/for-7.2/devicetree-arm64-fixes - imx/maintainers - https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux tags/imx-maintainers-7.3 - qcom/dt-fixes - https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm64-fixes-for-7.2 - broadcom/fixes-2 - https://github.com/Broadcom/stblinux tags/arm-soc/for-7.2/devicetree-fixes-v2 - aspeed/fixes - https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/aspeed-7.2-driver-fixes-0 - nuvoton/fixes - https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/nuvoton-7.2-arm-fixes-0 - nuvoton/maintainers - https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/aspeed-7.3-maintainers-0 + optee-fix + git://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee tags/optee-fix-for-v7.2 + apple/fixes + https://git.kernel.org/pub/scm/linux/kernel/git/sven/linux tags/apple-soc-fixes-7.2 From 274481df02f0ce6d164db2b81cde14c9c1efd93d Mon Sep 17 00:00:00 2001 From: Amelie Delaunay Date: Fri, 12 Jun 2026 14:56:02 +0200 Subject: [PATCH 104/857] arm64: dts: st: reorder ommanager node in stm32mp257f-ev1.dts In the ST board DTS files, the &label entries must be ordered alphanumerically. The nodes became misordered when &ommanager and &lptimer3 were added simultaneously. After that, <dc and &lvds used the &lptimers position as a reference. Move ommanager at the right place to avoid future misordering. Signed-off-by: Amelie Delaunay Link: https://lore.kernel.org/r/20260612-node_reordering-v2-1-f68032ca3088@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp257f-ev1.dts | 56 +++++++++++----------- 1 file changed, 28 insertions(+), 28 deletions(-) diff --git a/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts b/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts index 14e033f365e398..f044331b8b553a 100644 --- a/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts +++ b/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts @@ -307,34 +307,6 @@ /delete-property/dma-names; }; -&ommanager { - memory-region = <&mm_ospi1>; - memory-region-names = "ospi1"; - pinctrl-0 = <&ospi_port1_clk_pins_a - &ospi_port1_io03_pins_a - &ospi_port1_cs0_pins_a>; - pinctrl-1 = <&ospi_port1_clk_sleep_pins_a - &ospi_port1_io03_sleep_pins_a - &ospi_port1_cs0_sleep_pins_a>; - pinctrl-names = "default", "sleep"; - status = "okay"; - - spi@0 { - #address-cells = <1>; - #size-cells = <0>; - memory-region = <&mm_ospi1>; - status = "okay"; - - flash0: flash@0 { - compatible = "jedec,spi-nor"; - reg = <0>; - spi-rx-bus-width = <4>; - spi-tx-bus-width = <4>; - spi-max-frequency = <50000000>; - }; - }; -}; - /* use LPTIMER with tick broadcast for suspend mode */ &lptimer3 { status = "okay"; @@ -374,6 +346,34 @@ }; }; +&ommanager { + memory-region = <&mm_ospi1>; + memory-region-names = "ospi1"; + pinctrl-0 = <&ospi_port1_clk_pins_a + &ospi_port1_io03_pins_a + &ospi_port1_cs0_pins_a>; + pinctrl-1 = <&ospi_port1_clk_sleep_pins_a + &ospi_port1_io03_sleep_pins_a + &ospi_port1_cs0_sleep_pins_a>; + pinctrl-names = "default", "sleep"; + status = "okay"; + + spi@0 { + #address-cells = <1>; + #size-cells = <0>; + memory-region = <&mm_ospi1>; + status = "okay"; + + flash0: flash@0 { + compatible = "jedec,spi-nor"; + reg = <0>; + spi-rx-bus-width = <4>; + spi-tx-bus-width = <4>; + spi-max-frequency = <50000000>; + }; + }; +}; + &pcie_ep { pinctrl-names = "default", "init"; pinctrl-0 = <&pcie_pins_a>; From 954e8aad2a1ee334e360cdf911c70848d870941f Mon Sep 17 00:00:00 2001 From: Amelie Delaunay Date: Fri, 12 Jun 2026 14:56:03 +0200 Subject: [PATCH 105/857] ARM: dts: stm32: reorder cs_cti_trace node in stm32mp135f-dk.dts In the ST board DTS files, the &label entries must be ordered alphanumerically. The nodes became misordered when Coresight support was added. Move cs_cti_trace to the right place to avoid future misordering. Signed-off-by: Amelie Delaunay Link: https://lore.kernel.org/r/20260612-node_reordering-v2-2-f68032ca3088@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp135f-dk.dts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/arch/arm/boot/dts/st/stm32mp135f-dk.dts b/arch/arm/boot/dts/st/stm32mp135f-dk.dts index 6022e73f58afd6..bc3050a9bec542 100644 --- a/arch/arm/boot/dts/st/stm32mp135f-dk.dts +++ b/arch/arm/boot/dts/st/stm32mp135f-dk.dts @@ -190,11 +190,11 @@ status = "okay"; }; -&cs_cti_trace { +&cs_cti_cpu0 { status = "okay"; }; -&cs_cti_cpu0 { +&cs_cti_trace { status = "okay"; }; From c39ef0a87f2cabafd7056dadb829398c58befb6e Mon Sep 17 00:00:00 2001 From: Amelie Delaunay Date: Fri, 12 Jun 2026 14:56:04 +0200 Subject: [PATCH 106/857] ARM: dts: stm32: reorder cs_cti_trace node in stm32mp15xx-dkx.dtsi In the ST board DTS files, the &label entries must be ordered alphanumerically. The nodes became misordered when Coresight support was added. Move cs_cti_trace to the right place to avoid future misordering. Signed-off-by: Amelie Delaunay Link: https://lore.kernel.org/r/20260612-node_reordering-v2-3-f68032ca3088@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi b/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi index 599ea07bdb19ca..956509cef32141 100644 --- a/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi +++ b/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi @@ -155,15 +155,15 @@ status = "okay"; }; -&cs_cti_trace { +&cs_cti_cpu0 { status = "okay"; }; -&cs_cti_cpu0 { +&cs_cti_cpu1 { status = "okay"; }; -&cs_cti_cpu1 { +&cs_cti_trace { status = "okay"; }; From b2bc6021a37f44ebf83e74a60262c757a2057e1d Mon Sep 17 00:00:00 2001 From: Amelie Delaunay Date: Fri, 12 Jun 2026 14:56:05 +0200 Subject: [PATCH 107/857] ARM: dts: stm32: reorder cs_cti_trace node in stm32mp157c-ev1.dts In the ST board DTS files, the &label entries must be ordered alphanumerically. The nodes became misordered when Coresight support was added. Move cs_cti_trace to the right place to avoid future misordering. Signed-off-by: Amelie Delaunay Link: https://lore.kernel.org/r/20260612-node_reordering-v2-4-f68032ca3088@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp157c-ev1.dts | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/arm/boot/dts/st/stm32mp157c-ev1.dts b/arch/arm/boot/dts/st/stm32mp157c-ev1.dts index 0e65a1862eb533..eaab09e1755fcb 100644 --- a/arch/arm/boot/dts/st/stm32mp157c-ev1.dts +++ b/arch/arm/boot/dts/st/stm32mp157c-ev1.dts @@ -81,15 +81,15 @@ status = "okay"; }; -&cs_cti_trace { +&cs_cti_cpu0 { status = "okay"; }; -&cs_cti_cpu0 { +&cs_cti_cpu1 { status = "okay"; }; -&cs_cti_cpu1 { +&cs_cti_trace { status = "okay"; }; From 3a886e93febd468e3c59487c3af91b8388ad71a3 Mon Sep 17 00:00:00 2001 From: Amelie Delaunay Date: Fri, 12 Jun 2026 14:56:06 +0200 Subject: [PATCH 108/857] ARM: dts: stm32: reorder mdma1 node in stm32mp15*-scmi.dts In the ST board DTS files, the &label entries must be ordered alphanumerically. The nodes became misordered when mlahb was replaced by m4_rproc. Move mdma1 to the right place to avoid future misordering. Signed-off-by: Amelie Delaunay Link: https://lore.kernel.org/r/20260612-node_reordering-v2-5-f68032ca3088@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp157a-dk1-scmi.dts | 8 ++++---- arch/arm/boot/dts/st/stm32mp157c-dk2-scmi.dts | 8 ++++---- arch/arm/boot/dts/st/stm32mp157c-ed1-scmi.dts | 8 ++++---- arch/arm/boot/dts/st/stm32mp157c-ev1-scmi.dts | 8 ++++---- 4 files changed, 16 insertions(+), 16 deletions(-) diff --git a/arch/arm/boot/dts/st/stm32mp157a-dk1-scmi.dts b/arch/arm/boot/dts/st/stm32mp157a-dk1-scmi.dts index 847b360f02fccf..53e40e2f776b7e 100644 --- a/arch/arm/boot/dts/st/stm32mp157a-dk1-scmi.dts +++ b/arch/arm/boot/dts/st/stm32mp157a-dk1-scmi.dts @@ -51,10 +51,6 @@ clocks = <&rcc IWDG2>, <&scmi_clk CK_SCMI_LSI>; }; -&mdma1 { - resets = <&scmi_reset RST_SCMI_MDMA>; -}; - &m4_rproc { /delete-property/ st,syscfg-holdboot; resets = <&scmi_reset RST_SCMI_MCU>, @@ -62,6 +58,10 @@ reset-names = "mcu_rst", "hold_boot"; }; +&mdma1 { + resets = <&scmi_reset RST_SCMI_MDMA>; +}; + &optee { interrupt-parent = <&intc>; interrupts = ; diff --git a/arch/arm/boot/dts/st/stm32mp157c-dk2-scmi.dts b/arch/arm/boot/dts/st/stm32mp157c-dk2-scmi.dts index 43280289759d02..0790ed426ebc02 100644 --- a/arch/arm/boot/dts/st/stm32mp157c-dk2-scmi.dts +++ b/arch/arm/boot/dts/st/stm32mp157c-dk2-scmi.dts @@ -57,10 +57,6 @@ clocks = <&rcc IWDG2>, <&scmi_clk CK_SCMI_LSI>; }; -&mdma1 { - resets = <&scmi_reset RST_SCMI_MDMA>; -}; - &m4_rproc { /delete-property/ st,syscfg-holdboot; resets = <&scmi_reset RST_SCMI_MCU>, @@ -68,6 +64,10 @@ reset-names = "mcu_rst", "hold_boot"; }; +&mdma1 { + resets = <&scmi_reset RST_SCMI_MDMA>; +}; + &optee { interrupt-parent = <&intc>; interrupts = ; diff --git a/arch/arm/boot/dts/st/stm32mp157c-ed1-scmi.dts b/arch/arm/boot/dts/st/stm32mp157c-ed1-scmi.dts index 6f27d794d2702c..0a3894aff4aead 100644 --- a/arch/arm/boot/dts/st/stm32mp157c-ed1-scmi.dts +++ b/arch/arm/boot/dts/st/stm32mp157c-ed1-scmi.dts @@ -56,10 +56,6 @@ clocks = <&rcc IWDG2>, <&scmi_clk CK_SCMI_LSI>; }; -&mdma1 { - resets = <&scmi_reset RST_SCMI_MDMA>; -}; - &m4_rproc { /delete-property/ st,syscfg-holdboot; resets = <&scmi_reset RST_SCMI_MCU>, @@ -67,6 +63,10 @@ reset-names = "mcu_rst", "hold_boot"; }; +&mdma1 { + resets = <&scmi_reset RST_SCMI_MDMA>; +}; + &optee { interrupt-parent = <&intc>; interrupts = ; diff --git a/arch/arm/boot/dts/st/stm32mp157c-ev1-scmi.dts b/arch/arm/boot/dts/st/stm32mp157c-ev1-scmi.dts index 6ae391bffee53a..c2b6efb1cbb74e 100644 --- a/arch/arm/boot/dts/st/stm32mp157c-ev1-scmi.dts +++ b/arch/arm/boot/dts/st/stm32mp157c-ev1-scmi.dts @@ -61,10 +61,6 @@ clocks = <&scmi_clk CK_SCMI_HSE>, <&rcc FDCAN_K>; }; -&mdma1 { - resets = <&scmi_reset RST_SCMI_MDMA>; -}; - &m4_rproc { /delete-property/ st,syscfg-holdboot; resets = <&scmi_reset RST_SCMI_MCU>, @@ -72,6 +68,10 @@ reset-names = "mcu_rst", "hold_boot"; }; +&mdma1 { + resets = <&scmi_reset RST_SCMI_MDMA>; +}; + &optee { interrupt-parent = <&intc>; interrupts = ; From 40afa1b1a56c1e3d7713c0f1f37b95174dfa25c8 Mon Sep 17 00:00:00 2001 From: Ahmad Fatoum Date: Thu, 11 Jun 2026 20:12:33 +0200 Subject: [PATCH 109/857] ARM: dts: stm32: lxa-mc1: change stdout-path baud rate from 9600 to 115200 The default baud rate when none is specified is up to the DT consumer. In the case of the Linux STM32 serial driver, it defaults to 9600 baud, which differs from the 115200 baud that this board's barebox bootloader configured. This went unnoticed, because barebox automatically fixes up a console= command-line option that looks like this on the LXA boards: console=ttySTM0,115200n8 This had precedence over the 9600 fallback baud rate. But when EFI booting a kernel via GRUB, we run into this issue, because the barebox-provided command-line is disregarded by GRUB. Fix this by explicitly setting the baud rate to the correct 115200. Fixes: 666b5ca85cd3 ("ARM: dts: stm32: add STM32MP1-based Linux Automation MC-1 board") Signed-off-by: Ahmad Fatoum Link: https://lore.kernel.org/r/20260611-lxa-stdout-path-baudrate-v1-1-59b60a5069ff@pengutronix.de Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp157c-lxa-mc1.dts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm/boot/dts/st/stm32mp157c-lxa-mc1.dts b/arch/arm/boot/dts/st/stm32mp157c-lxa-mc1.dts index eada9cf257be9c..b3d8eb57aa24eb 100644 --- a/arch/arm/boot/dts/st/stm32mp157c-lxa-mc1.dts +++ b/arch/arm/boot/dts/st/stm32mp157c-lxa-mc1.dts @@ -33,7 +33,7 @@ }; chosen { - stdout-path = &uart4; + stdout-path = "serial0:115200n8"; }; led-controller-0 { From e78fba2ce1682ed52058fba358840194463ba9a9 Mon Sep 17 00:00:00 2001 From: Ahmad Fatoum Date: Thu, 11 Jun 2026 20:12:34 +0200 Subject: [PATCH 110/857] ARM: dts: stm32: lxa-tac: change stdout-path baud rate from 9600 to 115200 The default baud rate when none is specified is up to the DT consumer. In the case of the Linux STM32 serial driver, it defaults to 9600 baud, which differs from the 115200 baud that this board's barebox bootloader configured. This went unnoticed, because barebox automatically fixes up a console= command-line option that looks like this on the LXA boards: console=ttySTM0,115200n8 This had precedence over the 9600 fallback baud rate. But when EFI booting a kernel via GRUB, we run into this issue, because the barebox-provided command-line is disregarded by GRUB. Fix this by explicitly setting the baud rate to the correct 115200. Fixes: 518272af37b2 ("ARM: dts: stm32: lxa-tac: add Linux Automation GmbH TAC") Signed-off-by: Ahmad Fatoum Link: https://lore.kernel.org/r/20260611-lxa-stdout-path-baudrate-v1-2-59b60a5069ff@pengutronix.de Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp15xc-lxa-tac.dtsi | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm/boot/dts/st/stm32mp15xc-lxa-tac.dtsi b/arch/arm/boot/dts/st/stm32mp15xc-lxa-tac.dtsi index ab13f0c39892f8..ddb1657cd78575 100644 --- a/arch/arm/boot/dts/st/stm32mp15xc-lxa-tac.dtsi +++ b/arch/arm/boot/dts/st/stm32mp15xc-lxa-tac.dtsi @@ -33,7 +33,7 @@ }; chosen { - stdout-path = &uart4; + stdout-path = "serial0:115200n8"; }; led-controller-0 { From dd4c4b1231b4b3ce97f94daa6f3874040b8b8302 Mon Sep 17 00:00:00 2001 From: Ahmad Fatoum Date: Thu, 11 Jun 2026 20:12:35 +0200 Subject: [PATCH 111/857] ARM: dts: stm32: fairytux2: change stdout-path baud rate from 9600 to 115200 The default baud rate when none is specified is up to the DT consumer. In the case of the Linux STM32 serial driver, it defaults to 9600 baud, which differs from the 115200 baud that this board's barebox bootloader configured. This went unnoticed, because barebox automatically fixes up a console= command-line option that looks like this on the LXA boards: console=ttySTM0,115200n8 This had precedence over the 9600 fallback baud rate. But when EFI booting a kernel via GRUB, we run into this issue, because the barebox-provided command-line is disregarded by GRUB. Fix this by explicitly setting the baud rate to the correct 115200. Fixes: 8c6d469f5249 ("ARM: dts: stm32: lxa-fairytux2: add Linux Automation GmbH FairyTux 2") Signed-off-by: Ahmad Fatoum Link: https://lore.kernel.org/r/20260611-lxa-stdout-path-baudrate-v1-3-59b60a5069ff@pengutronix.de Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp153c-lxa-fairytux2.dtsi | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm/boot/dts/st/stm32mp153c-lxa-fairytux2.dtsi b/arch/arm/boot/dts/st/stm32mp153c-lxa-fairytux2.dtsi index 7d3a6a3b5d09ea..d30b626a18c20e 100644 --- a/arch/arm/boot/dts/st/stm32mp153c-lxa-fairytux2.dtsi +++ b/arch/arm/boot/dts/st/stm32mp153c-lxa-fairytux2.dtsi @@ -28,7 +28,7 @@ }; chosen { - stdout-path = &uart4; + stdout-path = "serial0:115200n8"; }; backlight: backlight { From 308edd981accf51d0153199f97d5d87b514db224 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Mon, 10 Aug 2026 10:51:31 +0200 Subject: [PATCH 112/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 1fa452947b5fbe..f60ed2c162572e 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -18,6 +18,8 @@ soc/arm https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux tags/imx-soc-7.3 tegra/arm git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.3-arm-core + omap/soc + git://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap tags/omap-for-v7.3/soc-signed soc/dt patch @@ -99,6 +101,18 @@ soc/drivers https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux tags/aspeed-7.3-drivers-0 qcom/drivers https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-drivers-for-7.3 + ti/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/ti/linux tags/ti-driver-soc-for-v7.3 + qcomtee/driver + git://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee tags/qcomtee-for-v7.3 + qcom/drivers-2 + https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-drivers-for-7.3-2 + apple/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/sven/linux tags/apple-soc-drivers-7.3 + fsl/soc-drivers + https://git.kernel.org/pub/scm/linux/kernel/git/chleroy/linux tags/soc_fsl-7.3-1 + zynx/firmware + https://github.com/Xilinx/linux-xlnx tags/zynqmp-soc-for-7.3 soc/defconfig riscv/defconfig @@ -109,6 +123,8 @@ soc/defconfig git://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap tags/omap-for-v7.3/defconfig-signed allwinner/defconfig https://git.kernel.org/pub/scm/linux/kernel/git/sunxi/linux tags/sunxi-config-for-7.3 + k3/defconfig + https://git.kernel.org/pub/scm/linux/kernel/git/ti/linux tags/ti-k3-config-for-v7.3 soc/late @@ -117,4 +133,6 @@ arm/fixes git://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee tags/optee-fix-for-v7.2 apple/fixes https://git.kernel.org/pub/scm/linux/kernel/git/sven/linux tags/apple-soc-fixes-7.2 + tegra/fix + git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.2-arm64-dt-fixes-v2 From 395890b455d2e1cf698358c73741707bcdb7ac1d Mon Sep 17 00:00:00 2001 From: Alain Volmat Date: Tue, 21 Jul 2026 19:00:27 +0200 Subject: [PATCH 113/857] ARM: dts: stm32: add LTDC / DCMIPP access controller for STM32MP135 Add ETZPC as an access controller for the LTDC and DCMIPP found on the stm32mp135. Fixes: a06b9560eb6c ("ARM: dts: stm32: add ETZPC as a system bus for STM32MP13x boards") Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260721-etzpc-mp13-ltdc-dcmipp-fix-v1-1-a969f2ab8da1@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp135.dtsi | 42 ++++++++++++++-------------- 1 file changed, 21 insertions(+), 21 deletions(-) diff --git a/arch/arm/boot/dts/st/stm32mp135.dtsi b/arch/arm/boot/dts/st/stm32mp135.dtsi index 834a4d545fe448..f573335edf6b44 100644 --- a/arch/arm/boot/dts/st/stm32mp135.dtsi +++ b/arch/arm/boot/dts/st/stm32mp135.dtsi @@ -6,29 +6,29 @@ #include "stm32mp133.dtsi" -/ { - soc { - dcmipp: dcmipp@5a000000 { - compatible = "st,stm32mp13-dcmipp"; - reg = <0x5a000000 0x400>; - interrupts = ; - resets = <&rcc DCMIPP_R>; - clocks = <&rcc DCMIPP_K>; - status = "disabled"; +&etzpc { + dcmipp: dcmipp@5a000000 { + compatible = "st,stm32mp13-dcmipp"; + reg = <0x5a000000 0x400>; + interrupts = ; + resets = <&rcc DCMIPP_R>; + clocks = <&rcc DCMIPP_K>; + access-controllers = <&etzpc 4>; + status = "disabled"; - port { - }; + port { }; + }; - ltdc: display-controller@5a001000 { - compatible = "st,stm32-ltdc"; - reg = <0x5a001000 0x400>; - interrupts = , - ; - clocks = <&rcc LTDC_PX>; - clock-names = "lcd"; - resets = <&scmi_reset RST_SCMI_LTDC>; - status = "disabled"; - }; + ltdc: display-controller@5a001000 { + compatible = "st,stm32-ltdc"; + reg = <0x5a001000 0x400>; + interrupts = , + ; + clocks = <&rcc LTDC_PX>; + clock-names = "lcd"; + resets = <&scmi_reset RST_SCMI_LTDC>; + access-controllers = <&etzpc 3>; + status = "disabled"; }; }; From 759d2d79a5ca4b7146c658f6375d6c4fa2f7e050 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Cl=C3=A9ment=20Le=20Goffic?= Date: Wed, 22 Jul 2026 11:16:11 +0200 Subject: [PATCH 114/857] ARM: dts: stm32: add sram[123] nodes on stm32mp131 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add SRAM nodes with "mmio-sram" compatible in STM32MP131 SoC device tree. Those nodes describe the SRAM memory area (SRAM1 16kB, SRAM2 8kB, SRAM3 8kB). Signed-off-by: Clément Le Goffic Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260722-spi_arm_stm32_dt-v1-1-6ef611523232@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp131.dtsi | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/arch/arm/boot/dts/st/stm32mp131.dtsi b/arch/arm/boot/dts/st/stm32mp131.dtsi index 83ae59b73dd099..5ae33b26e313fa 100644 --- a/arch/arm/boot/dts/st/stm32mp131.dtsi +++ b/arch/arm/boot/dts/st/stm32mp131.dtsi @@ -139,6 +139,30 @@ interrupt-parent = <&intc>; ranges; + sram1: sram@30000000 { + compatible = "mmio-sram"; + reg = <0x30000000 0x4000>; + #address-cells = <1>; + #size-cells = <1>; + ranges = <0 0x30000000 0x4000>; + }; + + sram2: sram@30004000 { + compatible = "mmio-sram"; + reg = <0x30004000 0x2000>; + #address-cells = <1>; + #size-cells = <1>; + ranges = <0 0x30004000 0x2000>; + }; + + sram3: sram@30006000 { + compatible = "mmio-sram"; + reg = <0x30006000 0x2000>; + #address-cells = <1>; + #size-cells = <1>; + ranges = <0 0x30006000 0x2000>; + }; + timers2: timer@40000000 { #address-cells = <1>; #size-cells = <0>; From 0b6dda3c968d4ba7840866e2afbb4fddc30c4141 Mon Sep 17 00:00:00 2001 From: Valentin Caron Date: Wed, 22 Jul 2026 11:16:12 +0200 Subject: [PATCH 115/857] ARM: dts: stm32: add pins for spi4 and spi5 in stm32mp15-pinctrl Add pins for spi4 and spi5 in stm32mp15-pinctrl.dtsi Signed-off-by: Valentin Caron Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260722-spi_arm_stm32_dt-v1-2-6ef611523232@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp15-pinctrl.dtsi | 34 +++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/arch/arm/boot/dts/st/stm32mp15-pinctrl.dtsi b/arch/arm/boot/dts/st/stm32mp15-pinctrl.dtsi index aaa91b634c12ff..4480bc52927027 100644 --- a/arch/arm/boot/dts/st/stm32mp15-pinctrl.dtsi +++ b/arch/arm/boot/dts/st/stm32mp15-pinctrl.dtsi @@ -2873,6 +2873,31 @@ }; }; + /omit-if-no-ref/ + spi4_pins_b: spi4-1 { + pins1 { + pinmux = , /* SPI4_SCK */ + ; /* SPI4_MOSI */ + bias-disable; + drive-push-pull; + slew-rate = <1>; + }; + + pins2 { + pinmux = ; /* SPI4_MISO */ + bias-disable; + }; + }; + + /omit-if-no-ref/ + spi4_sleep_pins_b: spi4-sleep-1 { + pins { + pinmux = , /* SPI4_SCK */ + , /* SPI4_MISO */ + ; /* SPI4_MOSI */ + }; + }; + /omit-if-no-ref/ spi5_pins_a: spi5-0 { pins1 { @@ -2889,6 +2914,15 @@ }; }; + /omit-if-no-ref/ + spi5_sleep_pins_a: spi5-sleep-0 { + pins { + pinmux = , /* SPI5_SCK */ + , /* SPI5_MISO */ + ; /* SPI5_MOSI */ + }; + }; + /omit-if-no-ref/ stusb1600_pins_a: stusb1600-0 { pins { From 0bd09d86484387da514623f89d4dba412f0c3d2a Mon Sep 17 00:00:00 2001 From: Alain Volmat Date: Wed, 22 Jul 2026 11:16:13 +0200 Subject: [PATCH 116/857] ARM: dts: stm32: Use DMA FIFO mode for all spi in stm32mp151 When used, configure the DMA in FIFO mode (instead of Direct) for all SPI instances of stm32mp151.dtsi Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260722-spi_arm_stm32_dt-v1-3-6ef611523232@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp151.dtsi | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/arch/arm/boot/dts/st/stm32mp151.dtsi b/arch/arm/boot/dts/st/stm32mp151.dtsi index 84f68e8563d853..e7fec8b49ae165 100644 --- a/arch/arm/boot/dts/st/stm32mp151.dtsi +++ b/arch/arm/boot/dts/st/stm32mp151.dtsi @@ -943,8 +943,8 @@ interrupts = ; clocks = <&rcc SPI2_K>; resets = <&rcc SPI2_R>; - dmas = <&dmamux1 39 0x400 0x05>, - <&dmamux1 40 0x400 0x05>; + dmas = <&dmamux1 39 0x400 0x01>, + <&dmamux1 40 0x400 0x01>; dma-names = "rx", "tx"; access-controllers = <&etzpc 27>; status = "disabled"; @@ -970,8 +970,8 @@ interrupts = ; clocks = <&rcc SPI3_K>; resets = <&rcc SPI3_R>; - dmas = <&dmamux1 61 0x400 0x05>, - <&dmamux1 62 0x400 0x05>; + dmas = <&dmamux1 61 0x400 0x01>, + <&dmamux1 62 0x400 0x01>; dma-names = "rx", "tx"; access-controllers = <&etzpc 28>; status = "disabled"; @@ -1301,8 +1301,8 @@ interrupts = ; clocks = <&rcc SPI1_K>; resets = <&rcc SPI1_R>; - dmas = <&dmamux1 37 0x400 0x05>, - <&dmamux1 38 0x400 0x05>; + dmas = <&dmamux1 37 0x400 0x01>, + <&dmamux1 38 0x400 0x01>; dma-names = "rx", "tx"; access-controllers = <&etzpc 52>; status = "disabled"; @@ -1316,8 +1316,8 @@ interrupts = ; clocks = <&rcc SPI4_K>; resets = <&rcc SPI4_R>; - dmas = <&dmamux1 83 0x400 0x05>, - <&dmamux1 84 0x400 0x05>; + dmas = <&dmamux1 83 0x400 0x01>, + <&dmamux1 84 0x400 0x01>; dma-names = "rx", "tx"; access-controllers = <&etzpc 53>; status = "disabled"; @@ -1432,8 +1432,8 @@ interrupts = ; clocks = <&rcc SPI5_K>; resets = <&rcc SPI5_R>; - dmas = <&dmamux1 85 0x400 0x05>, - <&dmamux1 86 0x400 0x05>; + dmas = <&dmamux1 85 0x400 0x01>, + <&dmamux1 86 0x400 0x01>; dma-names = "rx", "tx"; access-controllers = <&etzpc 57>; status = "disabled"; From cb72caaf12cbbd968e048a3249c57ef01cbc85fb Mon Sep 17 00:00:00 2001 From: Alain Volmat Date: Wed, 22 Jul 2026 11:16:14 +0200 Subject: [PATCH 117/857] ARM: dts: stm32: Add disabled spi4 and spi5 in stm32mp15xx-dkx Add disabled spi4 and spi5 nodes within stm32mp15xx-dkx. SPI4 can be accessed via the Arduino connectors SPI5 can be accessed via the GPIO expansion connectors. Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260722-spi_arm_stm32_dt-v1-4-6ef611523232@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi b/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi index 956509cef32141..db379de1d71fed 100644 --- a/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi +++ b/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi @@ -623,6 +623,20 @@ status = "disabled"; }; +&spi4 { + pinctrl-names = "default", "sleep"; + pinctrl-0 = <&spi4_pins_b>; + pinctrl-1 = <&spi4_sleep_pins_b>; + status = "disabled"; +}; + +&spi5 { + pinctrl-names = "default", "sleep"; + pinctrl-0 = <&spi5_pins_a>; + pinctrl-1 = <&spi5_sleep_pins_a>; + status = "disabled"; +}; + &timers1 { /* spare dmas for other usage */ /delete-property/dmas; From 6aa9dda968ed437e0f220126a91f31176262957c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Cl=C3=A9ment=20Le=20Goffic?= Date: Wed, 22 Jul 2026 11:16:15 +0200 Subject: [PATCH 118/857] ARM: dts: stm32: add sram pool to spi4 for DMA-MDMA chaining on MP15 DK MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The sram pool is used as a buffer area during a spi transfer using DMA-MDMA chaining. The pool size depends on the SPI framework that creates sg up to 4096 bytes. Signed-off-by: Clément Le Goffic Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260722-spi_arm_stm32_dt-v1-5-6ef611523232@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi b/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi index db379de1d71fed..31071cb89b999c 100644 --- a/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi +++ b/arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi @@ -627,6 +627,11 @@ pinctrl-names = "default", "sleep"; pinctrl-0 = <&spi4_pins_b>; pinctrl-1 = <&spi4_sleep_pins_b>; + dmas = <&dmamux1 83 0x400 0x01>, + <&dmamux1 84 0x400 0x01>, + <&mdma1 0 0x3 0x1200000a 0 0>; + dma-names = "rx", "tx", "rxm2m"; + sram = <&spi4_dma_pool>; status = "disabled"; }; @@ -637,6 +642,13 @@ status = "disabled"; }; +&sram4 { + spi4_dma_pool: dma-sram@9000 { + reg = <0x9000 0x1000>; + pool; + }; +}; + &timers1 { /* spare dmas for other usage */ /delete-property/dmas; From 87bbe6f23fd791be31693098f7c85ac1dfe7a2c3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Cl=C3=A9ment=20Le=20Goffic?= Date: Wed, 22 Jul 2026 11:16:16 +0200 Subject: [PATCH 119/857] ARM: dts: stm32: add sram pool to spi5 for DMA-MDMA chaining on MP13 DK MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The sram pool is used as a buffer area during a spi transfer using DMA-MDMA chaining. The pool size depends on the SPI framework that creates sg up to 4096 bytes. Signed-off-by: Clément Le Goffic Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260722-spi_arm_stm32_dt-v1-6-6ef611523232@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm/boot/dts/st/stm32mp135f-dk.dts | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/arch/arm/boot/dts/st/stm32mp135f-dk.dts b/arch/arm/boot/dts/st/stm32mp135f-dk.dts index bc3050a9bec542..b7835ef96db7e1 100644 --- a/arch/arm/boot/dts/st/stm32mp135f-dk.dts +++ b/arch/arm/boot/dts/st/stm32mp135f-dk.dts @@ -479,9 +479,21 @@ pinctrl-names = "default", "sleep"; pinctrl-0 = <&spi5_pins_a>; pinctrl-1 = <&spi5_sleep_pins_a>; + dmas = <&dmamux1 85 0x400 0x01>, + <&dmamux1 86 0x400 0x01>, + <&mdma 0 0x3 0x1200000a 0 0>; + dma-names = "rx", "tx", "rxm2m"; + sram = <&spi5_dma_pool>; status = "disabled"; }; +&sram2 { + spi5_dma_pool: dma-sram@1000 { + reg = <0x1000 0x1000>; + pool; + }; +}; + &timers3 { /delete-property/dmas; /delete-property/dma-names; From 90d4d470d4d478ea011f2ebb6b4ecf199a6d0653 Mon Sep 17 00:00:00 2001 From: Christophe Roullier Date: Tue, 4 Aug 2026 13:40:19 +0200 Subject: [PATCH 120/857] arm64: dts: st: Add I/O sync to eth1 pinctrl in stm32mp25-pinctrl.dtsi On board stm32mp235f-dk, stm32mp257f-dk the propagation delay between eth1 and the external PHY requires a compensation to guarantee that no packet get lost in all the working conditions. Add I/O synchronization properties in pinctrl on all the RGMII data pins, activating re-sampling on both edges of the clock. Co-developed-by: Antonio Borneo Signed-off-by: Antonio Borneo Signed-off-by: Christophe Roullier Link: https://lore.kernel.org/r/20260804114019.201949-2-christophe.roullier@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi b/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi index 456ece7f8ebc31..4fcfc4528c5e84 100644 --- a/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi +++ b/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi @@ -95,6 +95,7 @@ bias-disable; drive-push-pull; slew-rate = <3>; + st,io-sync = "data on both edges"; }; pins2 { pinmux = , /* ETH_RGMII_CLK125 */ @@ -112,6 +113,7 @@ , /* ETH_RGMII_RXD3 */ ; /* ETH_RGMII_RX_CTL */ bias-disable; + st,io-sync = "data on both edges"; }; pins4 { pinmux = ; /* ETH_RGMII_RX_CLK */ From 8a6e7bbf8ba5294621defde97281d9a6ddabe24f Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Mon, 10 Aug 2026 18:40:05 +0200 Subject: [PATCH 121/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 29 ++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index f60ed2c162572e..40cfa472be788e 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -75,6 +75,35 @@ soc/dt git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.3-dt-bindings tegra/dt64 git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.3-arm64-dt + patch + Documentation/process: maintainer-soc: Mention expectation about dt-check-style + Revert "riscv: dts: spacemit: k3: add i2s0-i2s5 nodes" + sophgo/dt + https://github.com/sophgo/linux tags/riscv-sophgo-dt-for-v7.3 + sophgo/dt-arm + https://github.com/sophgo/linux tags/arm-sophgo-dt-for-v7.3 + rockchip/dt + https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip tags/v7.3-rockchip-dts32-2 + rockchip/dt64-2 + https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip tags/v7.3-rockchip-dts64-2 + samsung/dt-2 + https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux tags/samsung-dt64-7.3-2 + samsung/dt32-2 + https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux tags/samsung-dt-7.3 + arm64/dt-cleanup + https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt tags/dt64-cleanup-7.3 + arm/dt-cleanup + https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt tags/dt-cleanup-7.3 + zynq/dt64 + https://github.com/Xilinx/linux-xlnx tags/zynqmp-dt-for-7.3 + ti/dt64 + https://git.kernel.org/pub/scm/linux/kernel/git/ti/linux tags/ti-k3-dts-for-v7.3 + qcom/dt-2 + https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm64-for-7.3-2 + qcom/dt32-2 + https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm32-for-7.3 + apple/dt + https://git.kernel.org/pub/scm/linux/kernel/git/sven/linux tags/apple-soc-dt-7.3 soc/drivers ixp4xx/soc-drivers From af3564e6cd8b9c8295b65e68ad7d29cbb8b416e0 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Mon, 10 Aug 2026 22:21:18 +0200 Subject: [PATCH 122/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 40cfa472be788e..311eaaf2532da9 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -20,6 +20,8 @@ soc/arm git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux tags/tegra-for-7.3-arm-core omap/soc git://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap tags/omap-for-v7.3/soc-signed + samsung/soc + https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux tags/samsung-soc-7.3 soc/dt patch From ed516e3f1aa6112925a889515548c30d9f4517b7 Mon Sep 17 00:00:00 2001 From: Ibrahim Abdelkader Date: Tue, 11 Aug 2026 10:37:29 +0200 Subject: [PATCH 123/857] Bluetooth: hci_sync: Clear HCI_CMD_PENDING when dropping the last request A synchronous HCI command that never receives a response leaves HCI_CMD_PENDING set: hci_req_cmd_complete() is the only place that clears it, and it only runs when a response matching the last command sent arrives. hci_send_cmd_sync() populates hdev->req_skb only when the flag transitions from clear to set, while hci_dev_open_sync() and hci_dev_close_sync() drop req_skb without clearing the flag. After a timeout followed by either, the two disagree: the flag claims a request is outstanding while req_skb is NULL. Subsequent synchronous commands are then sent with no req_skb, so hci_event_packet() has nothing to match an arriving event against, and the caller times out even though the controller answered. Commands answered by Command Complete recover on their own, since hci_req_cmd_complete() clears the flag as a side effect. Drivers using __hci_cmd_sync_ev() with a custom event do not, because a vendor event never reaches that path. On a WCN3988 (hci_qca over UART) this makes a controller firmware hang unrecoverable: the driver injects a hardware error and re-runs qca_setup(), qca_read_soc_version() waits for HCI_EV_VENDOR, the reply arrives within 4 ms and is discarded, and every retry fails the same way. The adapter is left down until the driver is unbound and rebound, or power is removed. Clear the flag wherever the last request is dropped, restoring the invariant that req_skb is non-NULL exactly when HCI_CMD_PENDING is set. Verified on hardware by forcing a command timeout: without this change setup fails on every attempt, with it setup succeeds on the first. Fixes: 2615fd9a7c25 ("Bluetooth: hci_sync: Fix overwriting request callback") Cc: stable@vger.kernel.org Signed-off-by: Ibrahim Abdelkader Signed-off-by: Hans de Goede Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/hci_sync.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/net/bluetooth/hci_sync.c b/net/bluetooth/hci_sync.c index b5897545d79552..4d6ab5d39e94b0 100644 --- a/net/bluetooth/hci_sync.c +++ b/net/bluetooth/hci_sync.c @@ -5448,6 +5448,7 @@ int hci_dev_open_sync(struct hci_dev *hdev) if (hdev->req_skb) { kfree_skb(hdev->req_skb); hdev->req_skb = NULL; + hci_dev_clear_flag(hdev, HCI_CMD_PENDING); } clear_bit(HCI_RUNNING, &hdev->flags); @@ -5632,6 +5633,7 @@ int hci_dev_close_sync(struct hci_dev *hdev) if (hdev->req_skb) { kfree_skb(hdev->req_skb); hdev->req_skb = NULL; + hci_dev_clear_flag(hdev, HCI_CMD_PENDING); } clear_bit(HCI_RUNNING, &hdev->flags); From 2fe8aa99af5577e4fb24c599210475963cbdfb7c Mon Sep 17 00:00:00 2001 From: Hans de Goede Date: Tue, 11 Aug 2026 10:37:30 +0200 Subject: [PATCH 124/857] Bluetooth: hci_sync: Factor common cleanup code into a helper The hci_dev_init_sync() failure path in hci_dev_open_sync() and the cleanup code in hci_dev_close_sync() have a bunch of common code. Factor this duplicate code out into a hci_dev_drop_last_cmd_req_and_close() helper function. Signed-off-by: Hans de Goede Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/hci_sync.c | 64 +++++++++++++++++----------------------- 1 file changed, 27 insertions(+), 37 deletions(-) diff --git a/net/bluetooth/hci_sync.c b/net/bluetooth/hci_sync.c index 4d6ab5d39e94b0..ea8baf05178e90 100644 --- a/net/bluetooth/hci_sync.c +++ b/net/bluetooth/hci_sync.c @@ -5353,6 +5353,30 @@ static int hci_dev_init_sync(struct hci_dev *hdev) return ret; } +static void hci_dev_drop_last_cmd_req_and_close(struct hci_dev *hdev) +{ + /* Drop last sent command */ + if (hdev->sent_cmd) { + cancel_delayed_work_sync(&hdev->cmd_timer); + kfree_skb(hdev->sent_cmd); + hdev->sent_cmd = NULL; + } + + /* Drop last request */ + if (hdev->req_skb) { + kfree_skb(hdev->req_skb); + hdev->req_skb = NULL; + hci_dev_clear_flag(hdev, HCI_CMD_PENDING); + } + + clear_bit(HCI_RUNNING, &hdev->flags); + hci_sock_dev_event(hdev, HCI_DEV_CLOSE); + + /* After this point our queues are empty and no tasks are scheduled. */ + hdev->close(hdev); + hdev->flags &= BIT(HCI_RAW); +} + int hci_dev_open_sync(struct hci_dev *hdev) { int ret; @@ -5439,23 +5463,7 @@ int hci_dev_open_sync(struct hci_dev *hdev) if (hdev->flush) hdev->flush(hdev); - if (hdev->sent_cmd) { - cancel_delayed_work_sync(&hdev->cmd_timer); - kfree_skb(hdev->sent_cmd); - hdev->sent_cmd = NULL; - } - - if (hdev->req_skb) { - kfree_skb(hdev->req_skb); - hdev->req_skb = NULL; - hci_dev_clear_flag(hdev, HCI_CMD_PENDING); - } - - clear_bit(HCI_RUNNING, &hdev->flags); - hci_sock_dev_event(hdev, HCI_DEV_CLOSE); - - hdev->close(hdev); - hdev->flags &= BIT(HCI_RAW); + hci_dev_drop_last_cmd_req_and_close(hdev); } done: @@ -5622,28 +5630,10 @@ int hci_dev_close_sync(struct hci_dev *hdev) skb_queue_purge(&hdev->cmd_q); skb_queue_purge(&hdev->raw_q); - /* Drop last sent command */ - if (hdev->sent_cmd) { - cancel_delayed_work_sync(&hdev->cmd_timer); - kfree_skb(hdev->sent_cmd); - hdev->sent_cmd = NULL; - } - - /* Drop last request */ - if (hdev->req_skb) { - kfree_skb(hdev->req_skb); - hdev->req_skb = NULL; - hci_dev_clear_flag(hdev, HCI_CMD_PENDING); - } - - clear_bit(HCI_RUNNING, &hdev->flags); - hci_sock_dev_event(hdev, HCI_DEV_CLOSE); - - /* After this point our queues are empty and no tasks are scheduled. */ - hdev->close(hdev); + /* Drop last sent command, last request and close */ + hci_dev_drop_last_cmd_req_and_close(hdev); /* Clear flags */ - hdev->flags &= BIT(HCI_RAW); hci_dev_clear_volatile_flags(hdev); hci_dev_clear_flag(hdev, HCI_CMD_DRAIN_WORKQUEUE); From 6d5e3062ef688af0371fa5272cea3bde4c0e6138 Mon Sep 17 00:00:00 2001 From: Ajith P V Date: Tue, 11 Aug 2026 09:49:32 +0000 Subject: [PATCH 125/857] Bluetooth: bnep: refactor deprecated strcpy The strcpy() function is deprecated across the kernel tree and moving towards complete elimination. It provides no verification limits against buffer overflows and does not guarantee strict boundary restrictions [1][2]. Replace instances of strcpy() in `net/bluetooth/bnep/core.c` with the safer strscpy() alternative. Since both target destination blocks are statically allocated fixed-size arrays within their structure definitions, leverage the compile time sizeof() operator to explicitly pass the destination buffer capacities. Link: https://www.kernel.org/doc/html/latest/process/deprecated.html#strcpy [1] Link: https://github.com/KSPP/linux/issues/88 [2] Signed-off-by: Ajith P V Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/bnep/core.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/net/bluetooth/bnep/core.c b/net/bluetooth/bnep/core.c index f7d88c33e23e4f..0dde82e9a08597 100644 --- a/net/bluetooth/bnep/core.c +++ b/net/bluetooth/bnep/core.c @@ -670,7 +670,7 @@ int bnep_add_connection(struct bnep_connadd_req *req, struct socket *sock) goto failed; } - strcpy(req->device, dev->name); + strscpy(req->device, dev->name, sizeof(req->device)); up_write(&bnep_session_sem); return 0; @@ -712,7 +712,7 @@ static void __bnep_copy_ci(struct bnep_conninfo *ci, struct bnep_session *s) memset(ci, 0, sizeof(*ci)); memcpy(ci->dst, s->eh.h_source, ETH_ALEN); - strcpy(ci->device, s->dev->name); + strscpy(ci->device, s->dev->name, sizeof(ci->device)); ci->flags = s->flags & valid_flags; ci->state = s->state; ci->role = s->role; From b2d4577362602b2191bccfe84e469a676c574d21 Mon Sep 17 00:00:00 2001 From: Christophe JAILLET Date: Sun, 9 Aug 2026 14:59:59 +0200 Subject: [PATCH 126/857] Bluetooth: btmrvl: Slightly simplify btmrvl_process_event() In btmrvl_process_event(), all error handling paths except one do a direct return. Update the only one that makes a goto to be consistent. This does not change the behavior because ret is known to be != 0 when 'exit' is reached. This simplifies the code, saves 2 LoC and pleases one of my coccinelle script that tries to spot erroneously mixed goto and return statements. Signed-off-by: Christophe JAILLET Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmrvl_main.c | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/drivers/bluetooth/btmrvl_main.c b/drivers/bluetooth/btmrvl_main.c index e25930351f643c..d449fbdde46406 100644 --- a/drivers/bluetooth/btmrvl_main.c +++ b/drivers/bluetooth/btmrvl_main.c @@ -87,8 +87,7 @@ int btmrvl_process_event(struct btmrvl_private *priv, struct sk_buff *skb) event = (struct btmrvl_event *) skb->data; if (event->ec != 0xff) { BT_DBG("Not Marvell Event=%x", event->ec); - ret = -EINVAL; - goto exit; + return -EINVAL; } switch (event->data[0]) { @@ -165,7 +164,6 @@ int btmrvl_process_event(struct btmrvl_private *priv, struct sk_buff *skb) break; } -exit: if (!ret) kfree_skb(skb); From 809a66378e9e087481799de508990f606a79ccc4 Mon Sep 17 00:00:00 2001 From: Richard Nunley Date: Sun, 9 Aug 2026 15:30:50 -0500 Subject: [PATCH 127/857] Bluetooth: btusb: Add Realtek RTL8821CE device 13d3:3558 The IMC Networks Bluetooth controller with USB ID 13d3:3558 uses an RTL8821CE. Without BTUSB_REALTEK, btusb uses generic initialization and does not load the controller firmware. BLE connections then fail before pairing with HCI error 0x3e. Add the device ID to the RTL8821CE table. This enables loading rtl_bt/rtl8821c_fw.bin and rtl_bt/rtl8821c_config.bin, after which a BLE HID keyboard pairs successfully. Signed-off-by: Richard Nunley Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btusb.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/drivers/bluetooth/btusb.c b/drivers/bluetooth/btusb.c index be82bbbc1b5cf6..2bae85b0016cee 100644 --- a/drivers/bluetooth/btusb.c +++ b/drivers/bluetooth/btusb.c @@ -509,6 +509,8 @@ static const struct usb_device_id quirks_table[] = { BTUSB_WIDEBAND_SPEECH }, { USB_DEVICE(0x13d3, 0x3533), .driver_info = BTUSB_REALTEK | BTUSB_WIDEBAND_SPEECH }, + { USB_DEVICE(0x13d3, 0x3558), .driver_info = BTUSB_REALTEK | + BTUSB_WIDEBAND_SPEECH }, /* Realtek 8822CE Bluetooth devices */ { USB_DEVICE(0x0bda, 0xb00c), .driver_info = BTUSB_REALTEK | From 32b42c0704fddf56809d9107a8061a367ceb8aa0 Mon Sep 17 00:00:00 2001 From: Guangshuo Li Date: Sat, 8 Aug 2026 13:15:32 +0800 Subject: [PATCH 128/857] Bluetooth: hci_bcm: fix usage_count leak when autosuspend_delay is negative bcm_request_irq() calls pm_runtime_use_autosuspend(), but bcm_close() does not call the matching pm_runtime_dont_use_autosuspend() when tearing down runtime PM. If the autosuspend delay is set to a negative value while autosuspend is enabled, the runtime PM core increments usage_count to prevent runtime suspend. Without calling pm_runtime_dont_use_autosuspend() during driver teardown, this reference is not dropped and usage_count remains unbalanced. Add the missing pm_runtime_dont_use_autosuspend() call before disabling runtime PM. This issue was found by manual code inspection. Fixes: e88ab30d3669 ("Bluetooth: hci_bcm: Add suspend/resume runtime PM functions") Cc: stable@vger.kernel.org Signed-off-by: Guangshuo Li Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/hci_bcm.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/bluetooth/hci_bcm.c b/drivers/bluetooth/hci_bcm.c index 01da3fecb536d2..9a103db7e3558b 100644 --- a/drivers/bluetooth/hci_bcm.c +++ b/drivers/bluetooth/hci_bcm.c @@ -547,6 +547,7 @@ static int bcm_close(struct hci_uart *hu) if (IS_ENABLED(CONFIG_PM) && bdev->irq_acquired) { devm_free_irq(bdev->dev, bdev->irq, bdev); device_init_wakeup(bdev->dev, false); + pm_runtime_dont_use_autosuspend(bdev->dev); pm_runtime_disable(bdev->dev); } From 9e257e3586428be74ca545affc4f0b8d679b3a77 Mon Sep 17 00:00:00 2001 From: Guangshuo Li Date: Sat, 8 Aug 2026 13:26:54 +0800 Subject: [PATCH 129/857] Bluetooth: hci_h5: fix usage_count leak when autosuspend_delay is negative h5_btrtl_open() calls pm_runtime_use_autosuspend(), but h5_btrtl_close() does not call the matching pm_runtime_dont_use_autosuspend() when tearing down runtime PM. If the autosuspend delay is set to a negative value while autosuspend is enabled, the runtime PM core increments usage_count to prevent runtime suspend. Without calling pm_runtime_dont_use_autosuspend() during driver teardown, this reference is not dropped and usage_count remains unbalanced. Add the missing pm_runtime_dont_use_autosuspend() call before disabling runtime PM. This issue was found by manual code inspection. Fixes: d9dd833cf6d2 ("Bluetooth: hci_h5: Add runtime suspend") Cc: stable@vger.kernel.org Signed-off-by: Guangshuo Li Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/hci_h5.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/drivers/bluetooth/hci_h5.c b/drivers/bluetooth/hci_h5.c index 60b90f1e11fced..b1999e14aadef1 100644 --- a/drivers/bluetooth/hci_h5.c +++ b/drivers/bluetooth/hci_h5.c @@ -1023,8 +1023,10 @@ static void h5_btrtl_open(struct h5 *h5) static void h5_btrtl_close(struct h5 *h5) { - if (!test_bit(H5_WAKEUP_DISABLE, &h5->flags)) + if (!test_bit(H5_WAKEUP_DISABLE, &h5->flags)) { + pm_runtime_dont_use_autosuspend(&h5->hu->serdev->dev); pm_runtime_disable(&h5->hu->serdev->dev); + } gpiod_set_value_cansleep(h5->device_wake_gpio, 0); gpiod_set_value_cansleep(h5->enable_gpio, 0); From bc2791e60e0829e15a6f312006f4a667426deaf6 Mon Sep 17 00:00:00 2001 From: Guangshuo Li Date: Sat, 8 Aug 2026 13:30:57 +0800 Subject: [PATCH 130/857] Bluetooth: hci_intel: fix usage_count leak when autosuspend_delay is negative intel_set_power() calls pm_runtime_use_autosuspend() when powering on the device, but the power-off path does not call the matching pm_runtime_dont_use_autosuspend() before disabling runtime PM. If the autosuspend delay is set to a negative value while autosuspend is enabled, the runtime PM core increments usage_count to prevent runtime suspend. Without calling pm_runtime_dont_use_autosuspend() during teardown, this reference is not dropped and usage_count remains unbalanced. Add the missing pm_runtime_dont_use_autosuspend() call before disabling runtime PM. This issue was found by manual code inspection. Fixes: 74cdad37cd24 ("Bluetooth: hci_intel: Add runtime PM support") Cc: stable@vger.kernel.org Signed-off-by: Guangshuo Li Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/hci_intel.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/bluetooth/hci_intel.c b/drivers/bluetooth/hci_intel.c index ecf597f3e201e4..d10ce7a0ba3e77 100644 --- a/drivers/bluetooth/hci_intel.c +++ b/drivers/bluetooth/hci_intel.c @@ -345,6 +345,7 @@ static int intel_set_power(struct hci_uart *hu, bool powered) devm_free_irq(&idev->pdev->dev, idev->irq, idev); device_wakeup_disable(&idev->pdev->dev); + pm_runtime_dont_use_autosuspend(&idev->pdev->dev); pm_runtime_disable(&idev->pdev->dev); } } From d1b752f552896275e639370288ec44e33e6636c6 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 9 Aug 2026 20:42:41 +0300 Subject: [PATCH 131/857] Bluetooth: L2CAP: access chan->conn safely in get/setsockopt Since commit b66774b48dd9 ("Bluetooth: L2CAP: Fix UAF in channel timeout by holding conn ref") l2cap_chan::conn has held reference and remains non-NULL also after the corresponding hci_conn is deleted. In this state accessing various fields eg. hci_conn::hdev is invalid, which leads to KASAN crash in l2cap_sock_setsockopt() access of conn->hcon->hdev. Check l2cap_chan::conn.hcon corresponds to an alive hci_conn before trying to use it in l2cap_sock.c. Hold l2cap_chan_lock() in getsockopt/setsockopt to ensure it stays alive, and to avoid data races in l2cap_chan fields. Fixes: b66774b48dd9 ("Bluetooth: L2CAP: Fix UAF in channel timeout by holding conn ref") Reported-by: syzbot+b106284c2a0b7bc80cf9@syzkaller.appspotmail.com Link: https://syzkaller.appspot.com/bug?extid=b106284c2a0b7bc80cf9 Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_sock.c | 64 ++++++++++++++++++++++++++++---------- 1 file changed, 48 insertions(+), 16 deletions(-) diff --git a/net/bluetooth/l2cap_sock.c b/net/bluetooth/l2cap_sock.c index 735167f73f3125..cca6201f9cdbbd 100644 --- a/net/bluetooth/l2cap_sock.c +++ b/net/bluetooth/l2cap_sock.c @@ -436,11 +436,26 @@ static int l2cap_get_mode(struct l2cap_chan *chan) return -EINVAL; } +static struct l2cap_conn *l2cap_chan_conn(struct l2cap_chan *chan) +{ + lockdep_assert_held(&chan->lock); + + /* l2cap_conn_del() sets FLAG_DEL while holding chan->lock before + * conn->hcon is deleted. If not set and conn is non-NULL, conn->hcon + * remains alive during this chan->lock critical section. + */ + if (test_bit(FLAG_DEL, &chan->flags)) + return NULL; + + return chan->conn; +} + static int l2cap_sock_getsockopt_old(struct socket *sock, int optname, sockopt_t *sopt) { struct sock *sk = sock->sk; struct l2cap_chan *chan = l2cap_pi(sk)->chan; + struct l2cap_conn *conn; struct l2cap_options opts; struct l2cap_conninfo cinfo; int err = 0; @@ -451,6 +466,7 @@ static int l2cap_sock_getsockopt_old(struct socket *sock, int optname, len = sopt->optlen; + l2cap_chan_lock(chan); lock_sock(sk); switch (optname) { @@ -537,9 +553,15 @@ static int l2cap_sock_getsockopt_old(struct socket *sock, int optname, break; } + conn = l2cap_chan_conn(chan); + if (!conn) { + err = -ENOTCONN; + break; + } + memset(&cinfo, 0, sizeof(cinfo)); - cinfo.hci_handle = chan->conn->hcon->handle; - memcpy(cinfo.dev_class, chan->conn->hcon->dev_class, 3); + cinfo.hci_handle = conn->hcon->handle; + memcpy(cinfo.dev_class, conn->hcon->dev_class, 3); len = min(len, sizeof(cinfo)); if (copy_to_iter(&cinfo, len, &sopt->iter_out) != len) @@ -553,6 +575,8 @@ static int l2cap_sock_getsockopt_old(struct socket *sock, int optname, } release_sock(sk); + l2cap_chan_unlock(chan); + return err; } @@ -561,6 +585,7 @@ static int l2cap_sock_getsockopt(struct socket *sock, int level, int optname, { struct sock *sk = sock->sk; struct l2cap_chan *chan = l2cap_pi(sk)->chan; + struct l2cap_conn *conn; struct bt_security sec; struct bt_power pwr; int len, mode, err = 0; @@ -578,6 +603,7 @@ static int l2cap_sock_getsockopt(struct socket *sock, int level, int optname, len = sopt->optlen; + l2cap_chan_lock(chan); lock_sock(sk); switch (optname) { @@ -589,12 +615,14 @@ static int l2cap_sock_getsockopt(struct socket *sock, int level, int optname, break; } + conn = l2cap_chan_conn(chan); + memset(&sec, 0, sizeof(sec)); - if (chan->conn) { - sec.level = chan->conn->hcon->sec_level; + if (conn) { + sec.level = conn->hcon->sec_level; if (sk->sk_state == BT_CONNECTED) - sec.key_size = chan->conn->hcon->enc_key_size; + sec.key_size = conn->hcon->enc_key_size; } else { sec.level = chan->sec_level; } @@ -678,12 +706,14 @@ static int l2cap_sock_getsockopt(struct socket *sock, int level, int optname, break; case BT_PHY: - if (sk->sk_state != BT_CONNECTED) { + conn = l2cap_chan_conn(chan); + + if (sk->sk_state != BT_CONNECTED || !conn) { err = -ENOTCONN; break; } - opt = hci_conn_get_phy(chan->conn->hcon); + opt = hci_conn_get_phy(conn->hcon); if (copy_to_iter(&opt, sizeof(opt), &sopt->iter_out) != sizeof(opt)) @@ -719,6 +749,7 @@ static int l2cap_sock_getsockopt(struct socket *sock, int level, int optname, } release_sock(sk); + l2cap_chan_unlock(chan); return err; } @@ -749,6 +780,7 @@ static int l2cap_sock_setsockopt_old(struct socket *sock, int optname, BT_DBG("sk %p", sk); + l2cap_chan_lock(chan); lock_sock(sk); switch (optname) { @@ -850,6 +882,7 @@ static int l2cap_sock_setsockopt_old(struct socket *sock, int optname, } release_sock(sk); + l2cap_chan_unlock(chan); return err; } @@ -913,6 +946,7 @@ static int l2cap_sock_setsockopt(struct socket *sock, int level, int optname, if (level != SOL_BLUETOOTH) return -ENOPROTOOPT; + l2cap_chan_lock(chan); lock_sock(sk); switch (optname) { @@ -938,11 +972,10 @@ static int l2cap_sock_setsockopt(struct socket *sock, int level, int optname, chan->sec_level = sec.level; - if (!chan->conn) + conn = l2cap_chan_conn(chan); + if (!conn) break; - conn = chan->conn; - /* change security for LE channels */ if (chan->scid == L2CAP_CID_ATT) { if (smp_conn_security(conn->hcon, sec.level)) { @@ -997,7 +1030,8 @@ static int l2cap_sock_setsockopt(struct socket *sock, int level, int optname, } if (opt == BT_FLUSHABLE_OFF) { - conn = chan->conn; + conn = l2cap_chan_conn(chan); + /* proceed further only when we have l2cap_conn and No Flush support in the LM */ if (!conn || !lmp_no_flush_capable(conn->hcon->hdev)) { @@ -1083,7 +1117,8 @@ static int l2cap_sock_setsockopt(struct socket *sock, int level, int optname, break; case BT_PHY: - if (sk->sk_state != BT_CONNECTED) { + conn = l2cap_chan_conn(chan); + if (sk->sk_state != BT_CONNECTED || !conn) { err = -ENOTCONN; break; } @@ -1093,10 +1128,6 @@ static int l2cap_sock_setsockopt(struct socket *sock, int level, int optname, if (err) break; - if (!chan->conn) - break; - - conn = chan->conn; err = hci_conn_set_phy(conn->hcon, phys); break; @@ -1139,6 +1170,7 @@ static int l2cap_sock_setsockopt(struct socket *sock, int level, int optname, } release_sock(sk); + l2cap_chan_unlock(chan); return err; } From 9db7e5fffbaecfa76aae285bf3e6c8cd4535897c Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 9 Aug 2026 01:06:05 +0300 Subject: [PATCH 132/857] Bluetooth: L2CAP: reject accept queue add unless BT_LISTEN New sk should not be added to parent socket accept queue after last l2cap_sock_cleanup_listen() has run in l2cap_sock_teardown_cb() and state set to BT_CLOSED, as that can result to UAF on dereferencing the dangling parent reference. l2cap_sock_new_connection_cb() may race with parent l2cap_chan teardown, due to chan->state accessed without consistent locking: [Task 1] [Task 2] l2cap_sock_release(parent) l2cap_connect l2cap_sock_shutdown pchan = l2cap_global_chan_by_psm l2cap_chan_lock(pchan) l2cap_chan_close l2cap_sock_teardown_cb pchan->state = BT_CLOSED l2cap_chan_unlock(pchan) ------> l2cap_chan_lock(pchan) l2cap_new_connection l2cap_sock_new_connection_cb l2cap_chan_lock(pchan) <-------- l2cap_chan_unlock(pchan) l2cap_sock_kill(parent) /* bt_sk(sk)->parent dangling */ Fix by adding check for sk_state == BT_LISTEN after acquiring sk lock in l2cap_sock_new_connection_cb(). Add lock_sock() around sk_state writes where missing, to avoid data races. Although the data races on pchan->state should be fixed too, this defensive sk_state check probably makes sense in any case. Fixes: 2ff1a41a912d ("Bluetooth: L2CAP: Fix null-ptr-deref in l2cap_sock_state_change_cb()") Reported-by: syzbot+9265e754091c2d27ea29@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=9265e754091c2d27ea29 Signed-off-by: Pauli Virtanen Reported-by: syzbot+9265e754091c2d27ea29@syzkaller.appspotmail.com Tested-by: syzbot+9265e754091c2d27ea29@syzkaller.appspotmail.com Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_sock.c | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/net/bluetooth/l2cap_sock.c b/net/bluetooth/l2cap_sock.c index cca6201f9cdbbd..8bf35bc8126f32 100644 --- a/net/bluetooth/l2cap_sock.c +++ b/net/bluetooth/l2cap_sock.c @@ -1600,6 +1600,11 @@ static int l2cap_sock_new_connection_cb(struct l2cap_chan *chan, lock_sock(parent); + if (parent->sk_state != BT_LISTEN) { + release_sock(parent); + return -EINVAL; + } + /* Check for backlog size */ if (sk_acceptq_is_full(parent)) { BT_DBG("backlog full %d", parent->sk_ack_backlog); @@ -1763,10 +1768,14 @@ static void l2cap_sock_state_change_cb(struct l2cap_chan *chan, int state, if (!sk) return; + lock_sock(sk); + sk->sk_state = state; if (err) sk->sk_err = err; + + release_sock(sk); } static struct sk_buff *l2cap_sock_alloc_skb_cb(struct l2cap_chan *chan, @@ -1842,6 +1851,8 @@ static void l2cap_sock_resume_cb(struct l2cap_chan *chan) if (!sk) return; + lock_sock(sk); + if (test_and_clear_bit(FLAG_PENDING_SECURITY, &chan->flags)) { sk->sk_state = BT_CONNECTED; chan->state = BT_CONNECTED; @@ -1849,6 +1860,8 @@ static void l2cap_sock_resume_cb(struct l2cap_chan *chan) clear_bit(BT_SK_SUSPEND, &bt_sk(sk)->flags); sk->sk_state_change(sk); + + release_sock(sk); } static void l2cap_sock_set_shutdown_cb(struct l2cap_chan *chan) From 444612a87229d836193685eeda77067c79aed990 Mon Sep 17 00:00:00 2001 From: ZhaoJinming Date: Tue, 11 Aug 2026 16:47:37 +0800 Subject: [PATCH 133/857] Bluetooth: virtio_bt: Fix use-after-free and memory leak in probe error paths When virtbt_open_vdev() fails in virtbt_probe(), hci_free_dev(hdev) is called without first calling hci_unregister_dev(hdev). Since hci_register_dev() already succeeded, the HCI device remains registered while its memory is freed, leading to a use-after-free when accessed via sysfs or HCI sockets. Additionally, the probe function leaks the virtio_bluetooth structure (vbt) in several error paths: - When virtio_find_vqs() fails, vbt is not freed. - When hci_alloc_dev() or hci_register_dev() fails, vbt is not freed. - When virtbt_open_vdev() fails, vbt is not freed. Furthermore, when virtbt_open_vdev() fails after virtio_device_ready() has been called, the device is left live (DRIVER_OK set) while its virtqueues are torn down, and any scheduled work is not flushed, potentially allowing a use-after-free from device-initiated callbacks. Fix all of these by restructuring the error labels to properly unwind in reverse order of the allocation/registration sequence. The new labels err_del_vqs and err_free_vbt ensure that del_vqs and kfree(vbt) are called as appropriate for each failure point. For the virtbt_open_vdev() failure path, call virtio_reset_device() and virtbt_close_vdev() before unregistering the HCI device, matching the cleanup pattern in virtbt_remove(). Signed-off-by: ZhaoJinming Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/virtio_bt.c | 21 +++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/drivers/bluetooth/virtio_bt.c b/drivers/bluetooth/virtio_bt.c index c20d54088c8c4e..8c55b538deefdb 100644 --- a/drivers/bluetooth/virtio_bt.c +++ b/drivers/bluetooth/virtio_bt.c @@ -315,12 +315,12 @@ static int virtbt_probe(struct virtio_device *vdev) err = virtio_find_vqs(vdev, VIRTBT_NUM_VQS, vbt->vqs, vqs_info, NULL); if (err) - return err; + goto err_free_vbt; hdev = hci_alloc_dev(); if (!hdev) { err = -ENOMEM; - goto failed; + goto err_del_vqs; } vbt->hdev = hdev; @@ -390,20 +390,25 @@ static int virtbt_probe(struct virtio_device *vdev) if (hci_register_dev(hdev) < 0) { hci_free_dev(hdev); err = -EBUSY; - goto failed; + goto err_del_vqs; } virtio_device_ready(vdev); err = virtbt_open_vdev(vbt); - if (err) - goto open_failed; + if (err) { + hci_unregister_dev(hdev); + virtio_reset_device(vdev); + virtbt_close_vdev(vbt); + hci_free_dev(hdev); + goto err_del_vqs; + } return 0; -open_failed: - hci_free_dev(hdev); -failed: +err_del_vqs: vdev->config->del_vqs(vdev); +err_free_vbt: + kfree(vbt); return err; } From fca8fe6149048c71ea8863f22674a8fc3cd84f2e Mon Sep 17 00:00:00 2001 From: ZhaoJinming Date: Mon, 10 Aug 2026 18:22:38 +0800 Subject: [PATCH 134/857] Bluetooth: btmtksdio: fix deadlock in close and reset paths btmtksdio_close() and btmtksdio_reset() call cancel_work_sync() on bdev->txrx_work while holding the sdio host lock, which is also acquired by btmtksdio_txrx_work(). If txrx_work is queued when close/reset runs, a worker thread may start it after the host lock is taken and block in sdio_claim_host(), while cancel_work_sync() waits for the work to finish. The host lock is only released after cancel_work_sync() returns, so both sides wait forever, deadlocking close/reset. Fix this by releasing the sdio host lock before calling cancel_work_sync(), then re-acquiring it afterwards. In btmtksdio_close() the interrupt is already disabled by sdio_release_irq(), which also unregisters the IRQ handler, so no new work can be scheduled and cancel_work_sync() fully quiesces txrx_work. btmtksdio_reset() must additionally unregister the IRQ handler before dropping the host lock: btmtksdio_txrx_work() unconditionally re-enables the device interrupt (C_INT_EN_SET) when the handler is still registered, so an in-flight worker would re-enable interrupts and be rescheduled while the device is being reset, defeating the cancellation. The IRQ is re-claimed by btmtksdio_open() when the HCI device is re-opened after the reset. This mirrors the pattern already used by btmtksdio_flush(), which cancels the work without holding the host lock. Signed-off-by: ZhaoJinming Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmtksdio.c | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/drivers/bluetooth/btmtksdio.c b/drivers/bluetooth/btmtksdio.c index 4e1012e90979d8..67d055d5f197bf 100644 --- a/drivers/bluetooth/btmtksdio.c +++ b/drivers/bluetooth/btmtksdio.c @@ -746,8 +746,16 @@ static int btmtksdio_close(struct hci_dev *hdev) sdio_release_irq(bdev->func); + /* No new work can be scheduled after sdio_release_irq(), so cancel the + * work outside the sdio host lock. btmtksdio_txrx_work() also claims + * the host, so canceling it while holding the lock would deadlock. + */ + sdio_release_host(bdev->func); + cancel_work_sync(&bdev->txrx_work); + sdio_claim_host(bdev->func); + btmtksdio_fw_pmctrl(bdev); clear_bit(BTMTKSDIO_FUNC_ENABLED, &bdev->tx_state); @@ -1293,8 +1301,22 @@ static void btmtksdio_reset(struct hci_dev *hdev) sdio_writel(bdev->func, C_INT_EN_CLR, MTK_REG_CHLPCR, NULL); skb_queue_purge(&bdev->txq); + + /* Unregister the IRQ before releasing the host lock so that a + * concurrently running btmtksdio_txrx_work() cannot re-enable the + * device interrupt (C_INT_EN_SET) and be rescheduled while the device + * is being reset. btmtksdio_txrx_work() also claims the host, so the + * work must be cancelled outside the sdio host lock to avoid a + * deadlock. The IRQ is re-claimed by btmtksdio_open() when the HCI + * device is re-opened after the reset. + */ + sdio_release_irq(bdev->func); + sdio_release_host(bdev->func); + cancel_work_sync(&bdev->txrx_work); + sdio_claim_host(bdev->func); + gpiod_set_value_cansleep(bdev->reset, 1); msleep(100); gpiod_set_value_cansleep(bdev->reset, 0); From 3ea6bd32027681ca83780fdad398804bdb5e3e4d Mon Sep 17 00:00:00 2001 From: ZhaoJinming Date: Mon, 10 Aug 2026 15:08:45 +0800 Subject: [PATCH 135/857] Bluetooth: hci_serdev: Fix use-after-free in hci_uart_unregister_device() hci_uart_unregister_device() frees the HCI device (hci_free_dev) before cancelling write_work via cancel_work_sync(). If write_work is executing concurrently on another CPU, it can access hu->hdev and write to hdev->stat after the memory has been freed. Additionally, HCI_UART_PROTO_READY is not cleared until after cancel_work_sync, so the write_wakeup serdev callback can still schedule write_work via hci_uart_tx_wakeup() even after hci_free_dev has freed the device. Fix this by mirroring the same ordering used in the tty/ldisc path (hci_uart_tty_close, hci_ldisc.c:565-593): 1. Save the PROTO_READY state and clear it under the write lock so a concurrent hci_uart_tx_wakeup() cannot re-schedule write_work 2. Cancel write_work (no new work can be scheduled and no work is in flight) 3. Unregister the HCI device 4. Close the protocol (may access hu->hdev and the serdev device) 5. Close the serdev port (safe now that write_work is quiesced and protocol is done) 6. Free the HCI device Also free any partially transmitted frame (hu->tx_skb) left over by write_work once the transmit path is quiesced, since hci_uart_close() would skip hci_uart_flush() because HCI_UART_PROTO_READY is cleared. Signed-off-by: ZhaoJinming Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/hci_serdev.c | 46 +++++++++++++++++++++++++++++----- 1 file changed, 40 insertions(+), 6 deletions(-) diff --git a/drivers/bluetooth/hci_serdev.c b/drivers/bluetooth/hci_serdev.c index 593d9cefbbf925..13346c20559105 100644 --- a/drivers/bluetooth/hci_serdev.c +++ b/drivers/bluetooth/hci_serdev.c @@ -395,20 +395,54 @@ EXPORT_SYMBOL_GPL(hci_uart_register_device_priv); void hci_uart_unregister_device(struct hci_uart *hu) { struct hci_dev *hdev = hu->hdev; + bool proto_ready; + /* Wait for init_ready to finish to prevent registration races */ cancel_work_sync(&hu->init_ready); - if (test_bit(HCI_UART_REGISTERED, &hu->flags)) - hci_unregister_dev(hdev); - hci_free_dev(hdev); + proto_ready = test_bit(HCI_UART_PROTO_READY, &hu->flags); + if (proto_ready) { + /* Clear HCI_UART_PROTO_READY under the write lock so a + * concurrent hci_uart_tx_wakeup() cannot re-schedule + * write_work via the write_wakeup callback once the device + * is torn down. + */ + percpu_down_write(&hu->proto_lock); + clear_bit(HCI_UART_PROTO_READY, &hu->flags); + percpu_up_write(&hu->proto_lock); + } + + /* Unconditionally cancel write_work AFTER clearing PROTO_READY. + * This ensures that concurrent protocol timers cannot requeue + * write_work, permanently preventing double-free races and UAFs, + * and guarantees no write_work is in flight before the serdev + * device is closed. + */ cancel_work_sync(&hu->write_work); + /* Free any partially transmitted frame left over by write_work now + * that the transmit path is fully quiesced. hci_uart_close() would + * skip hci_uart_flush() because HCI_UART_PROTO_READY is cleared. + */ + if (hu->tx_skb) { + kfree_skb(hu->tx_skb); + hu->tx_skb = NULL; + } + + if (test_bit(HCI_UART_REGISTERED, &hu->flags)) + hci_unregister_dev(hdev); + + /* Close the protocol before freeing hdev (intrinsically purges queues). + * Some protocol close handlers (e.g. qca_close) may still access the + * serdev device, so keep the serdev port open until this completes. + */ hu->proto->close(hu); - if (test_bit(HCI_UART_PROTO_READY, &hu->flags)) { - clear_bit(HCI_UART_PROTO_READY, &hu->flags); + if (proto_ready) serdev_device_close(hu->serdev); - } + + hci_free_dev(hdev); + percpu_free_rwsem(&hu->proto_lock); } EXPORT_SYMBOL_GPL(hci_uart_unregister_device); From 99672791e9c9f56257075ad95c285f03a6309720 Mon Sep 17 00:00:00 2001 From: Pavel Shpakovskiy Date: Sat, 8 Aug 2026 19:31:11 +0300 Subject: [PATCH 136/857] Bluetooth: mgmt: fix 'hdev->discovery.uuids' NULL dereference 'uuid_count' member of struct 'discovery_state' is assigned and read without any locks, so there is a chance of situation when uuid_count != 0, but uuids is NULL and there will be NULL pointer dereference. Possible race: 'hci_update_passive_scan_sync' 'hci_discovery_filter_clear' hdev->discovery.uuid_count = 0; <----------------------preempted-----------------------------> 'start_service_discovery' // Set uuid_count to value != 0 hdev->discovery.uuid_count = uuid_count; hdev->discovery.uuids = kmemdup(...); <----------------------preempted-----------------------------> spin_lock(&hdev->discovery.lock); kfree(hdev->discovery.uuids); hdev->discovery.uuids = NULL; spin_unlock(&hdev->discovery.lock); Now uuids == NULL and uuid_count != 0. So 'mgmt_device_found' -> 'is_filter_match' -> 'eir_has_uuids' receives non consistent discovery state, where NULL dereference of uuids happens. To fix it let's add discovery.lock around every read/write of uuid_count, uuids pair of struct members. It is also important to assign uuid_count value only after success kmemdup() allocation in start_service_discovery(), otherwise uuids is NULL, because kmemdup failed, but uuid_count is already assigned to non zero value. The following panic happens: [ ] ------------[ cut here ]------------ [ ] Unable to handle kernel NULL pointer dereference at virtual address 0000000000000000 [ ] Internal error: Oops: 0000000096000006 [#1] PREEMPT SMP [ ] CPU: 0 PID: 15056 Comm: kworker/u9:2 [ ] Workqueue: hci0 hci_rx_work [ ] pstate: 10400009 (nzcV daif +PAN -UAO -TCO -DIT -SSBS BTYPE=--) [ ] pc : eir_has_uuids+0x2d8/0x590 [ ] lr : is_filter_match+0x258/0x320 ... [ ] Call trace: [ ] eir_has_uuids+0x2d8/0x590 [ ] is_filter_match+0x258/0x320 [ ] mgmt_device_found+0x5b0/0xafc [ ] process_adv_report.part.0+0x8c8/0xf14 [ ] hci_le_adv_report_evt+0x338/0x3f0 [ ] hci_le_meta_evt+0x1f0/0x4c8 [ ] hci_event_packet+0x440/0xc9c [ ] hci_rx_work+0x44c/0xaf8 [ ] process_one_work+0x54c/0x103c [ ] worker_thread+0x6c4/0x10c4 [ ] kthread+0x274/0x2ec [ ] ret_from_fork+0x10/0x20 [ ] Code: 14000004 91004021 eb14003f 54000180 (f9400024) [ ] ---[ end trace 0000000000000000 ]--- Fixes: 2935e556850e ("Bluetooth: hci_sync: fix double free in 'hci_discovery_filter_clear()'") Signed-off-by: Pavel Shpakovskiy Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/hci_core.h | 2 +- net/bluetooth/mgmt.c | 18 +++++++++++++----- 2 files changed, 14 insertions(+), 6 deletions(-) diff --git a/include/net/bluetooth/hci_core.h b/include/net/bluetooth/hci_core.h index e07418a5adce75..4105c446ca983b 100644 --- a/include/net/bluetooth/hci_core.h +++ b/include/net/bluetooth/hci_core.h @@ -935,9 +935,9 @@ static inline void hci_discovery_filter_clear(struct hci_dev *hdev) hdev->discovery.result_filtering = false; hdev->discovery.report_invalid_rssi = true; hdev->discovery.rssi = HCI_RSSI_INVALID; - hdev->discovery.uuid_count = 0; spin_lock(&hdev->discovery.lock); + hdev->discovery.uuid_count = 0; kfree(hdev->discovery.uuids); hdev->discovery.uuids = NULL; spin_unlock(&hdev->discovery.lock); diff --git a/net/bluetooth/mgmt.c b/net/bluetooth/mgmt.c index 860c086011b714..ac4864e56ec727 100644 --- a/net/bluetooth/mgmt.c +++ b/net/bluetooth/mgmt.c @@ -6171,6 +6171,7 @@ static int start_service_discovery(struct sock *sk, struct hci_dev *hdev, struct mgmt_pending_cmd *cmd; const u16 max_uuid_count = ((U16_MAX - sizeof(*cp)) / 16); u16 uuid_count, expected_len; + u8 (*uuids)[16] = NULL; u8 status; int err; @@ -6247,12 +6248,10 @@ static int start_service_discovery(struct sock *sk, struct hci_dev *hdev, hdev->discovery.result_filtering = true; hdev->discovery.type = cp->type; hdev->discovery.rssi = cp->rssi; - hdev->discovery.uuid_count = uuid_count; if (uuid_count > 0) { - hdev->discovery.uuids = kmemdup(cp->uuids, uuid_count * 16, - GFP_KERNEL); - if (!hdev->discovery.uuids) { + uuids = kmemdup(cp->uuids, uuid_count * sizeof(*uuids), GFP_KERNEL); + if (!uuids) { err = mgmt_cmd_complete(sk, hdev->id, MGMT_OP_START_SERVICE_DISCOVERY, MGMT_STATUS_FAILED, @@ -6262,6 +6261,11 @@ static int start_service_discovery(struct sock *sk, struct hci_dev *hdev, } } + spin_lock(&hdev->discovery.lock); + hdev->discovery.uuids = uuids; + hdev->discovery.uuid_count = uuid_count; + spin_unlock(&hdev->discovery.lock); + err = hci_cmd_sync_queue(hdev, start_discovery_sync, cmd, start_discovery_complete); if (err < 0) { @@ -10505,6 +10509,7 @@ static bool is_filter_match(struct hci_dev *hdev, s8 rssi, u8 *eir, !hci_test_quirk(hdev, HCI_QUIRK_STRICT_DUPLICATE_FILTER)))) return false; + spin_lock(&hdev->discovery.lock); if (hdev->discovery.uuid_count != 0) { /* If a list of UUIDs is provided in filter, results with no * matching UUID should be dropped. @@ -10513,9 +10518,12 @@ static bool is_filter_match(struct hci_dev *hdev, s8 rssi, u8 *eir, hdev->discovery.uuids) && !eir_has_uuids(scan_rsp, scan_rsp_len, hdev->discovery.uuid_count, - hdev->discovery.uuids)) + hdev->discovery.uuids)) { + spin_unlock(&hdev->discovery.lock); return false; + } } + spin_unlock(&hdev->discovery.lock); /* If duplicate filtering does not report RSSI changes, then restart * scanning to ensure updated result with updated RSSI values. From d67f4a43e7ef8cff8aa8fe1df2f088390af41b6d Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 8 Aug 2026 12:08:45 +0300 Subject: [PATCH 137/857] Bluetooth: L2CAP: fix race l2cap_sock_cleanup_listen() vs. put_chan For L2CAP sockets without owning sk->sk_socket, reading l2cap_pi(sk)->chan may race against concurrent l2cap_sock_kill() -> l2cap_sock_put_chan(). This excludes simultaneous proto_ops callbacks, but access in l2cap_sock_cleanup_listen() has unsafe lockless read. [Task 1] [Task 2 (hdev->workqueue)] l2cap_sock_release(parent) l2cap_disconn_cfm l2cap_sock_cleanup_listen l2cap_conn_del bt_accept_dequeue l2cap_chan_del lock_sock(sk) l2cap_sock_teardown_cb bt_accept_unlink bt_sk(sk)->parent = NULL release_sock(sk) ----------------> lock_sock(sk) parent = /* NULL */ lock_sock(sk) <--------------------- release_sock(sk) sock_set_flag(sk, SOCK_ZAPPED) l2cap_sock_close_cb l2cap_sock_kill(sk) l2cap_sock_put_chan chan = READ l2cap_pi(sk)->chan l2cap_pi(sk)->chan = NULL l2cap_chan_hold_unless_zero l2cap_put_chan(chan) kref_get_unless_zero(&chan->ref) Task 1 may observe NULL which causes null-ptr-deref. Fix the race by taking lock_sock() in l2cap_sock_kill() to synchronize with l2cap_sock_cleanup_listen(). hold_unless_zero() is not needed here, l2cap_pi(sk)->chan owns reference if it is non-NULL. Clarify code comments vs. locking. Fixes: 6fef032af009 ("Bluetooth: L2CAP: Fix use-after-free in l2cap_sock_new_connection_cb()") Reported-by: syzbot+e6382a2f53f5fc7453ac@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=e6382a2f53f5fc7453ac Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 5 +++++ net/bluetooth/l2cap_sock.c | 23 +++++++++++++---------- 2 files changed, 18 insertions(+), 10 deletions(-) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index ef6ce1c20a4f05..3d9a32094347cd 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -699,7 +699,12 @@ struct l2cap_rx_busy { struct l2cap_pinfo { struct bt_sock bt; + + /* With owning sk_socket chan may be read without lock, other access + * should hold lock_sock. + */ struct l2cap_chan *chan; + struct list_head rx_busy; }; diff --git a/net/bluetooth/l2cap_sock.c b/net/bluetooth/l2cap_sock.c index 8bf35bc8126f32..1194c37e466f18 100644 --- a/net/bluetooth/l2cap_sock.c +++ b/net/bluetooth/l2cap_sock.c @@ -1344,7 +1344,12 @@ static void l2cap_sock_kill(struct sock *sk) BT_DBG("sk %p state %s", sk, state_to_string(sk->sk_state)); + /* Take lock to synchronize against access without owning sk->sk_socket, + * eg. in l2cap_sock_cleanup_listen(). proto_ops etc. don't need lock. + */ + lock_sock(sk); l2cap_sock_put_chan(sk); + release_sock(sk); /* Kill poor orphan */ sock_set_flag(sk, SOCK_DEAD); @@ -1548,14 +1553,10 @@ static void l2cap_sock_cleanup_listen(struct sock *parent) * establish sk_lock -> conn->lock and invert the established * conn->lock -> chan->lock -> sk_lock order (lockdep deadlock). * - * Instead, briefly take the child sk lock to fetch and pin its chan. - * l2cap_conn_del() reaches the chan free only via - * l2cap_chan_del() -> l2cap_sock_teardown_cb(), which itself takes - * the child sk lock; holding it across l2cap_chan_hold_unless_zero() - * therefore guarantees the chan cannot be freed while we read and - * pin it (hold_unless_zero() additionally skips a chan already past - * its last reference). We then drop the sk lock before taking - * chan->lock, so sk and chan locks are never held together. + * Instead, briefly take the child sk lock to synchronize vs. + * l2cap_sock_kill that puts l2cap_pi(sk)->chan. We then drop the sk + * lock before taking chan->lock, so sk and chan locks are never held + * together. * * Since we cannot call l2cap_chan_close() without conn->lock, * schedule l2cap_chan_timeout to close the channel; it already @@ -1565,10 +1566,12 @@ static void l2cap_sock_cleanup_listen(struct sock *parent) struct l2cap_chan *chan; lock_sock_nested(sk, L2CAP_NESTING_NORMAL); - chan = l2cap_chan_hold_unless_zero(l2cap_pi(sk)->chan); + chan = l2cap_pi(sk)->chan; + if (chan) + l2cap_chan_hold(chan); release_sock(sk); if (!chan) { - /* l2cap_conn_del() already tearing this child down */ + /* Already torn down */ sock_put(sk); continue; } From 1cc747bb4cde2a7c8323422c0ff92e87236837b0 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Wed, 12 Aug 2026 23:36:54 +0200 Subject: [PATCH 138/857] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 311eaaf2532da9..660ccc45367ebf 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -106,6 +106,8 @@ soc/dt https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm32-for-7.3 apple/dt https://git.kernel.org/pub/scm/linux/kernel/git/sven/linux tags/apple-soc-dt-7.3 + microchip/dt64 + https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux tags/microchip-dt64-7.3 soc/drivers ixp4xx/soc-drivers From ad16085eefa67595d6714132238d3a1c970771ae Mon Sep 17 00:00:00 2001 From: Hardik Prakash Date: Fri, 14 Aug 2026 15:37:19 +0530 Subject: [PATCH 139/857] Revert "i2c: designware: defer probe if child GpioInt controllers are not bound" This reverts commit 0a4bb2abc3e56d7be6e69b050c88ba52c87e22bf. The reverted commit causes a regression on ThinkPad T14s Gen 4 (AMD): the touchpad's I2C controller fails with lost arbitration errors, because it delays i2c-designware's probe by roughly 500ms, which shifts the touchpad's first HID descriptor fetch into a window where the platform's embedded controller is still acting as a secondary I2C bus master. Debug tracing confirms the GpioInt dependency check itself behaves correctly (it defers appropriately and confirms the GPIO controller is bound); the arbitration failure happens roughly a second after the check passes, when i2c_hid_acpi's own probe attempts its first transaction. The original fix is still needed for the Lenovo Yoga 7 14AGP11 touchscreen race the commit addressed, but a corrected version will be resubmitted once the EC bus-mastering interaction is understood and handled properly, rather than reintroducing a different regression on more widely-used ThinkPad hardware in the meantime. Reported-by: Thorsten Leemhuis Closes: https://lore.kernel.org/all/b4a4eadb-282f-464c-843a-19d415a34d0c@leemhuis.info/ Signed-off-by: Hardik Prakash Reviewed-by: Mario Limonciello (AMD) > --- Tested-by: Thorsten Leemhuis Signed-off-by: Andi Shyti Link: https://patch.msgid.link/20260814100719.9548-1-hardikprakash.official@gmail.com --- drivers/i2c/busses/i2c-designware-platdrv.c | 80 --------------------- 1 file changed, 80 deletions(-) diff --git a/drivers/i2c/busses/i2c-designware-platdrv.c b/drivers/i2c/busses/i2c-designware-platdrv.c index c8a203fff4d1e3..6d6e81242f74e1 100644 --- a/drivers/i2c/busses/i2c-designware-platdrv.c +++ b/drivers/i2c/busses/i2c-designware-platdrv.c @@ -8,14 +8,12 @@ * Copyright (C) 2007 MontaVista Software Inc. * Copyright (C) 2009 Provigent Ltd. */ -#include #include #include #include #include #include #include -#include #include #include #include @@ -132,80 +130,6 @@ static int i2c_dw_probe_lock_support(struct dw_i2c_dev *dev) return 0; } -#if defined(CONFIG_ACPI) && defined(CONFIG_GPIOLIB) -/* - * Check whether an ACPI GpioInt resource's referenced GPIO controller - * has finished probing. Resources with no named controller (resource - * source string) are skipped, since they can't be resolved to a - * struct device. - */ -static int check_gpioint_resource(struct acpi_resource *ares, void *data) -{ - struct acpi_resource_gpio *agpio; - struct acpi_device *gpio_adev; - struct device *gpio_dev; - acpi_handle handle; - acpi_status status; - - if (!acpi_gpio_get_irq_resource(ares, &agpio)) - return 1; /* not a GpioInt resource, skip */ - - if (!agpio->resource_source.string_length) - return 1; /* no named controller, skip */ - - status = acpi_get_handle(NULL, agpio->resource_source.string_ptr, &handle); - if (ACPI_FAILURE(status)) - return 1; - - gpio_adev = acpi_fetch_acpi_dev(handle); - if (!gpio_adev) - return 1; - - struct gpio_device *gdev __free(gpio_device_put) = - gpio_device_find_by_fwnode(acpi_fwnode_handle(gpio_adev)); - if (!gdev) - return -EPROBE_DEFER; /* controller not registered yet: abort walk */ - - gpio_dev = gpio_device_to_device(gdev)->parent; - - guard(device)(gpio_dev); - if (!device_is_bound(gpio_dev)) - return -EPROBE_DEFER; /* controller not bound yet: abort walk */ - - return 1; /* bound, skip adding to resource list, continue walk */ -} - -static int check_child_gpioint(struct acpi_device *adev, void *data) -{ - LIST_HEAD(res_list); - int ret; - - ret = acpi_dev_get_resources(adev, &res_list, check_gpioint_resource, NULL); - if (ret < 0) - return ret; - - acpi_dev_free_resource_list(&res_list); - - return 0; -} - -static int i2c_dw_check_gpio_dependencies(struct device *dev) -{ - struct acpi_device *adev; - - adev = ACPI_COMPANION(dev); - if (!adev) - return 0; - - return acpi_dev_for_each_child(adev, check_child_gpioint, NULL); -} -#else -static int i2c_dw_check_gpio_dependencies(struct device *dev) -{ - return 0; -} -#endif /* CONFIG_ACPI && CONFIG_GPIOLIB */ - static int dw_i2c_plat_probe(struct platform_device *pdev) { u32 flags = (uintptr_t)device_get_match_data(&pdev->dev); @@ -214,10 +138,6 @@ static int dw_i2c_plat_probe(struct platform_device *pdev) struct dw_i2c_dev *dev; int irq, ret; - ret = i2c_dw_check_gpio_dependencies(device); - if (ret) - return ret; - irq = platform_get_irq_optional(pdev, 0); if (irq == -ENXIO) flags |= ACCESS_POLLING; From 2ce8c58ee70075fc25d806434b80709680a1c0f4 Mon Sep 17 00:00:00 2001 From: Ruoyu Wang Date: Thu, 13 Aug 2026 23:31:55 +0800 Subject: [PATCH 140/857] i2c: ocores: Disable clock on failed resume ocores_i2c_resume() enables the controller clock before reinitializing the hardware. If the clock rate changed while the device was suspended, ocores_init() may reject the resulting prescaler. The callback then returns an error with the clock still enabled, while the controller itself remains disabled. Disable and unprepare the clock when ocores_init() fails so the failed resume path balances the successful clk_prepare_enable() call. This issue was found by a static analysis checker and confirmed by manual source review. Fixes: e961a094afe0 ("i2c: ocores: add common clock support") Signed-off-by: Ruoyu Wang Reviewed-by: Max Filippov Signed-off-by: Andi Shyti Link: https://patch.msgid.link/20260813153155.3953577-1-ruoyuw560@gmail.com --- drivers/i2c/busses/i2c-ocores.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/drivers/i2c/busses/i2c-ocores.c b/drivers/i2c/busses/i2c-ocores.c index df6ebf32d6e8f3..2d18c10358375f 100644 --- a/drivers/i2c/busses/i2c-ocores.c +++ b/drivers/i2c/busses/i2c-ocores.c @@ -755,7 +755,11 @@ static int ocores_i2c_resume(struct device *dev) rate = clk_get_rate(i2c->clk) / 1000; if (rate) i2c->ip_clock_khz = rate; - return ocores_init(dev, i2c); + ret = ocores_init(dev, i2c); + if (ret) + clk_disable_unprepare(i2c->clk); + + return ret; } static DEFINE_NOIRQ_DEV_PM_OPS(ocores_i2c_pm, From ccc3128e0e4e1a9afe6d39720ed2609cf9133215 Mon Sep 17 00:00:00 2001 From: Linkai Gong Date: Thu, 13 Aug 2026 17:56:17 +0800 Subject: [PATCH 141/857] i2c: mux: demux-pinctrl: fix OF node leak on kstrdup failure of_parse_phandle() takes a reference on the parent node. If a later devm_kstrdup() fails, err_rollback only releases nodes for indices 0..i-1, so the current node is leaked. of_node_put() the current parent before rolling back. Fixes: 7c0195fa9a9e ("i2c: mux: demux-pinctrl: check the return value of devm_kstrdup()") Signed-off-by: Linkai Gong Cc: # v6.6+ Signed-off-by: Andi Shyti Link: https://patch.msgid.link/20260813095617.2246320-1-gonglinkai@kylinos.cn --- drivers/i2c/muxes/i2c-demux-pinctrl.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/i2c/muxes/i2c-demux-pinctrl.c b/drivers/i2c/muxes/i2c-demux-pinctrl.c index f2a1f47449789c..2403c0bf7c4319 100644 --- a/drivers/i2c/muxes/i2c-demux-pinctrl.c +++ b/drivers/i2c/muxes/i2c-demux-pinctrl.c @@ -247,6 +247,7 @@ static int i2c_demux_pinctrl_probe(struct platform_device *pdev) props[i].value = devm_kstrdup(&pdev->dev, "ok", GFP_KERNEL); if (!props[i].name || !props[i].value) { err = -ENOMEM; + of_node_put(adap_np); goto err_rollback; } props[i].length = 3; From ae77d827fb0e6fb2faf8186f075c6c337f2dcf08 Mon Sep 17 00:00:00 2001 From: Alexandre Belloni Date: Mon, 17 Aug 2026 15:56:12 +0200 Subject: [PATCH 142/857] soc: document merges Signed-off-by: Alexandre Belloni --- arch/arm/arm-soc-for-next-contents.txt | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 660ccc45367ebf..ddf64ddd5df057 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -108,6 +108,10 @@ soc/dt https://git.kernel.org/pub/scm/linux/kernel/git/sven/linux tags/apple-soc-dt-7.3 microchip/dt64 https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux tags/microchip-dt64-7.3 + marvell/dt + git://git.kernel.org/pub/scm/linux/kernel/git/gclement/mvebu tags/mvebu-dt-7.3-1 + marvell/dt64 + git://git.kernel.org/pub/scm/linux/kernel/git/gclement/mvebu tags/mvebu-dt64-7.3-1 soc/drivers ixp4xx/soc-drivers From 771e812f94b320614147b7cd64d0a7b1186933ea Mon Sep 17 00:00:00 2001 From: Ismail Tarim Date: Sat, 15 Aug 2026 14:56:23 +0300 Subject: [PATCH 143/857] Bluetooth: btmtk: Do not report success when subsys reset fails btmtk_usb_subsys_reset() validates the subsystem reset by reading the chip id back. When that read succeeds at the bus level but yields an id of zero, the reset has demonstrably not taken effect: the function logs "Can't get device id, subsys reset fail." and then returns the return value of btmtk_usb_id_get(), which in that case is zero, i.e. success. btusb_mtk_reset() returns that value unchanged, so its caller cannot tell a completed reset from a failed one. Return -ENODEV when the chip id reads back as zero, leaving the existing MT6639 exemption intact. Observed on an MT7902 [13d3:3579]. The path can be reached on demand by asking the controller for a coredump, since btmtk requests a reset once the dump completes: # echo 1 > /sys/class/bluetooth/hci0/device/coredump Bluetooth: hci0: Mediatek coredump end Bluetooth: hci0: Can't get device id, subsys reset fail. usb 3-10: reset high-speed USB device number 5 using xhci_hcd usb 3-10: device descriptor read/64, error -110 usb usb3-port10: attempt power cycle usb usb3-port10: unable to enumerate USB device The same sequence occurs unprompted when the controller firmware asserts on its own. Note that this corrects the error reporting only; it does not by itself make the controller recoverable in the case above. Fixes: 25b6d7593a3a ("Bluetooth: btmtk: introduce btmtk reset work") Signed-off-by: Ismail Tarim Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmtk.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/drivers/bluetooth/btmtk.c b/drivers/bluetooth/btmtk.c index 66b34676104362..dc702c0a6034de 100644 --- a/drivers/bluetooth/btmtk.c +++ b/drivers/bluetooth/btmtk.c @@ -968,8 +968,10 @@ int btmtk_usb_subsys_reset(struct hci_dev *hdev, u32 dev_id) } err = btmtk_usb_id_get(hdev, 0x70010200, &val); - if (err || (!val && dev_id != 0x6639)) + if (err || (!val && dev_id != 0x6639)) { bt_dev_err(hdev, "Can't get device id, subsys reset fail."); + return err ? err : -ENODEV; + } return err; } From 54c03e6bc71882a46f6f4fe2fd09409950c7c814 Mon Sep 17 00:00:00 2001 From: Ismail Tarim Date: Sat, 15 Aug 2026 14:56:24 +0300 Subject: [PATCH 144/857] Bluetooth: btmtk: Do not discard the subsystem reset timeout When the MTK_BT_RST_DONE poll times out, btmtk_usb_subsys_reset() logs "Reset timeout" and keeps the error in err, but err is then overwritten by the return value of the following btmtk_usb_id_get() call, so the timeout is never reported to the caller. Commit 25b6d7593a3a ("Bluetooth: btmtk: introduce btmtk reset work") discarded the return value of the chip id read, so the function returned the timeout error as intended. Commit 3dcb122b3064 ("Bluetooth: btusb: mediatek: return error for failed reg access") started assigning err at that call and silently dropped it. Keep the timeout in a separate variable and return it, restoring the original behaviour without changing the control flow. Fixes: 3dcb122b3064 ("Bluetooth: btusb: mediatek: return error for failed reg access") Signed-off-by: Ismail Tarim Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmtk.c | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/drivers/bluetooth/btmtk.c b/drivers/bluetooth/btmtk.c index dc702c0a6034de..c0ed51567ed4dd 100644 --- a/drivers/bluetooth/btmtk.c +++ b/drivers/bluetooth/btmtk.c @@ -860,6 +860,7 @@ static u32 btmtk_usb_reset_done(struct hci_dev *hdev) int btmtk_usb_subsys_reset(struct hci_dev *hdev, u32 dev_id) { + int reset_err = 0; u32 val; int err; @@ -958,8 +959,10 @@ int btmtk_usb_subsys_reset(struct hci_dev *hdev, u32 dev_id) err = readx_poll_timeout(btmtk_usb_reset_done, hdev, val, val & MTK_BT_RST_DONE, 20000, 1000000); - if (err < 0) + if (err < 0) { bt_dev_err(hdev, "Reset timeout"); + reset_err = err; + } if (dev_id == 0x7922) { err = btmtk_usb_uhw_reg_write(hdev, MTK_UDMA_INT_STA_BT, 0x000000FF); @@ -973,7 +976,7 @@ int btmtk_usb_subsys_reset(struct hci_dev *hdev, u32 dev_id) return err ? err : -ENODEV; } - return err; + return reset_err; } EXPORT_SYMBOL_GPL(btmtk_usb_subsys_reset); From 951d9f743029bc73032aa32140ed3ca5af47de18 Mon Sep 17 00:00:00 2001 From: Chris Lu Date: Mon, 17 Aug 2026 17:53:31 +0800 Subject: [PATCH 145/857] Bluetooth: btmtksdio: Take exclusive ownership of the SKB before TX btmtksdio_tx_packet() prepends the MediaTek SDIO header with skb_push() and writes into that space after only checking the headroom size. On a cloned SKB that headroom belongs to a buffer shared with the other owner, which the driver has no right to write to. Cloned SKBs do reach this path: hci_send_cmd_sync() keeps a clone of every HCI command in hdev->sent_cmd before handing the SKB to the driver, and l2cap_ertm_send() clones SKBs for retransmission. Replace the open-coded headroom check with skb_cow_head(), which both guarantees the headroom and reallocates a private buffer when the SKB is cloned. The cost is one reallocation and copy per cloned packet, the usual price of this pattern in network drivers. This has no observable effect on its own, as the driver only writes in front of skb->data where no other owner looks. It is a prerequisite for "Bluetooth: btmtksdio: Fix out-of-bounds DMA read in the TX path", which writes padding behind skb->tail, and carries the same Fixes: tag so that both are backported together. Fixes: 9aebfd4a2200 ("Bluetooth: mediatek: add support for MediaTek MT7663S and MT7668S SDIO devices") Signed-off-by: Chris Lu Assisted-by: Claude:claude-opus-5 Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmtksdio.c | 13 ++++++------- 1 file changed, 6 insertions(+), 7 deletions(-) diff --git a/drivers/bluetooth/btmtksdio.c b/drivers/bluetooth/btmtksdio.c index 67d055d5f197bf..bcbbcf8ecc2430 100644 --- a/drivers/bluetooth/btmtksdio.c +++ b/drivers/bluetooth/btmtksdio.c @@ -274,13 +274,12 @@ static int btmtksdio_tx_packet(struct btmtksdio_dev *bdev, struct mtkbtsdio_hdr *sdio_hdr; int err; - /* Make sure that there are enough rooms for SDIO header */ - if (unlikely(skb_headroom(skb) < sizeof(*sdio_hdr))) { - err = pskb_expand_head(skb, sizeof(*sdio_hdr), 0, - GFP_ATOMIC); - if (err < 0) - return err; - } + /* Make sure that the data buffer is not shared with anyone else and + * that there is enough room for the SDIO header + */ + err = skb_cow_head(skb, sizeof(*sdio_hdr)); + if (err < 0) + return err; /* Prepend MediaTek SDIO Specific Header */ skb_push(skb, sizeof(*sdio_hdr)); From 262cb784c96cbcd4cb511466e492b2f077135341 Mon Sep 17 00:00:00 2001 From: Chris Lu Date: Mon, 17 Aug 2026 17:53:32 +0800 Subject: [PATCH 146/857] Bluetooth: btmtksdio: Fix out-of-bounds DMA read in the TX path btmtksdio_tx_packet() rounds the transfer size up to the SDIO block size of 256 bytes, but hands the host controller the SKB buffer as is: err = sdio_writesb(bdev->func, MTK_REG_CTDR, skb->data, round_up(skb->len, MTK_SDIO_BLOCK_SIZE)); Only skb->len bytes hold packet data, so the controller reads up to 255 bytes of uninitialised memory and sends it to the device over the SDIO bus. Depending on how much tailroom slack the SKB allocation happens to carry, that read can also extend past the end of the buffer. Compute the padded length up front, ensure the SKB has tailroom for it, and zero-fill the padding with skb_put_zero(). skb->len then covers the padding, so sdio_writesb() no longer needs to round up. byte_tx keeps counting the header and the payload only, and the error path restores the SKB so that the caller can requeue it. Writing behind skb->tail is only safe because the driver owns the buffer, which "Bluetooth: btmtksdio: Take exclusive ownership of the SKB before TX" ensures. Fixes: 9aebfd4a2200 ("Bluetooth: mediatek: add support for MediaTek MT7663S and MT7668S SDIO devices") Signed-off-by: Chris Lu Assisted-by: Claude:claude-opus-5 Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmtksdio.c | 26 +++++++++++++++++++++----- 1 file changed, 21 insertions(+), 5 deletions(-) diff --git a/drivers/bluetooth/btmtksdio.c b/drivers/bluetooth/btmtksdio.c index bcbbcf8ecc2430..b7f0be7fc42a92 100644 --- a/drivers/bluetooth/btmtksdio.c +++ b/drivers/bluetooth/btmtksdio.c @@ -272,6 +272,7 @@ static int btmtksdio_tx_packet(struct btmtksdio_dev *bdev, struct sk_buff *skb) { struct mtkbtsdio_hdr *sdio_hdr; + unsigned int len, pad_len; int err; /* Make sure that the data buffer is not shared with anyone else and @@ -281,6 +282,18 @@ static int btmtksdio_tx_packet(struct btmtksdio_dev *bdev, if (err < 0) return err; + /* The transfer is rounded up to the SDIO block size, so the buffer + * has to provide tailroom for the padding as well + */ + len = skb->len + sizeof(*sdio_hdr); + pad_len = round_up(len, MTK_SDIO_BLOCK_SIZE) - len; + + if (unlikely(skb_tailroom(skb) < pad_len)) { + err = pskb_expand_head(skb, 0, pad_len, GFP_ATOMIC); + if (err < 0) + return err; + } + /* Prepend MediaTek SDIO Specific Header */ skb_push(skb, sizeof(*sdio_hdr)); @@ -289,19 +302,22 @@ static int btmtksdio_tx_packet(struct btmtksdio_dev *bdev, sdio_hdr->reserved = cpu_to_le16(0); sdio_hdr->bt_type = hci_skb_pkt_type(skb); + /* Zero the padding so that no uninitialised memory is sent out */ + skb_put_zero(skb, pad_len); + clear_bit(BTMTKSDIO_HW_TX_READY, &bdev->tx_state); - err = sdio_writesb(bdev->func, MTK_REG_CTDR, skb->data, - round_up(skb->len, MTK_SDIO_BLOCK_SIZE)); + err = sdio_writesb(bdev->func, MTK_REG_CTDR, skb->data, skb->len); if (err < 0) - goto err_skb_pull; + goto err_skb_restore; - bdev->hdev->stat.byte_tx += skb->len; + bdev->hdev->stat.byte_tx += len; kfree_skb(skb); return 0; -err_skb_pull: +err_skb_restore: + skb_trim(skb, len); skb_pull(skb, sizeof(*sdio_hdr)); return err; From f716a05f496718a7f70a7765291ad8e7858c81a2 Mon Sep 17 00:00:00 2001 From: Sherry Sun Date: Mon, 17 Aug 2026 10:27:39 +0800 Subject: [PATCH 147/857] Bluetooth: btnxpuart: Check remote M.2 connector availability before pwrseq The current code uses of_graph_is_present() to decide whether to enter the pwrseq path. However, of_graph_is_present() only checks for the structural presence of a port/ports sub-node and does not check the status property. This causes problems when a DT overlay disables the remote M.2 connector node (e.g., switching from PCIe WiFi to SDIO WiFi): the port node still exists, so of_graph_is_present() returns true, but the pwrseq provider never registers because the connector is disabled, leading to an infinite -EPROBE_DEFER loop. Replace of_graph_is_present() with a new helper that traverses the OF graph to the remote port parent (the M.2 connector node) and checks of_device_is_available(). When the remote connector is disabled, the pwrseq path is skipped, allowing the BT driver to fall through to the direct bluetooth child node path. Fixes: e48e332d84d8 ("Bluetooth: btnxpuart: Add M.2 Bluetooth device support using pwrseq") Signed-off-by: Sherry Sun Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btnxpuart.c | 24 +++++++++++++++++++++++- 1 file changed, 23 insertions(+), 1 deletion(-) diff --git a/drivers/bluetooth/btnxpuart.c b/drivers/bluetooth/btnxpuart.c index 81cdd8da563669..e2b8f7997e4e5a 100644 --- a/drivers/bluetooth/btnxpuart.c +++ b/drivers/bluetooth/btnxpuart.c @@ -1809,6 +1809,28 @@ static void nxp_coredump_notify(struct hci_dev *hdev, int state) kobject_uevent_env(&serdev->dev.kobj, KOBJ_CHANGE, envp); } +/* + * Check if the remote M.2 connector device linked via OF graph is present + * and available. This is used to determine whether the pwrseq path should + * be taken. When the remote connector node is disabled (e.g., by a DT + * overlay switching from PCIe WiFi to SDIO WiFi), the pwrseq path is + * skipped, allowing the BT driver to use a direct bluetooth child node + * instead. + */ +static bool nxp_m2_connector_is_available(struct device *dev) +{ + struct device_node *ep __free(device_node) = + of_graph_get_next_endpoint(dev_of_node(dev), NULL); + + if (!ep) + return false; + + struct device_node *remote __free(device_node) = + of_graph_get_remote_port_parent(ep); + + return remote && of_device_is_available(remote); +} + static int nxp_serdev_probe(struct serdev_device *serdev) { struct hci_dev *hdev; @@ -1863,7 +1885,7 @@ static int nxp_serdev_probe(struct serdev_device *serdev) return err; } - if (of_graph_is_present(dev_of_node(&serdev->ctrl->dev))) { + if (nxp_m2_connector_is_available(&serdev->ctrl->dev)) { struct pwrseq_desc *pwrseq; pwrseq = pwrseq_get(&serdev->ctrl->dev, "uart"); From f4fe51177b82176080035025754d89f8e730f72f Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 16 Aug 2026 13:26:21 +0300 Subject: [PATCH 148/857] Bluetooth: hci_core: add lockdep check to hci_conn lookups Add lockdep check for RCU || hdev->lock in hci_conn_hash lookups that return hci_conn pointer, as dereferencing that without locks can be TOCTOU issue. It used to be several callsites did not hold appropriate locks. The check is equivalent to removing rcu_read_lock() and doing instead list_for_each_entry_rcu(c, &h->list, list, lockdep_is_held(&hdev->lock)) Although there should not be any remaining callsites without locks, don't remove the rcu_read_lock() for now, and just add the warning here. Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/hci_core.h | 44 ++++++++++++++++++++++++++++++++ 1 file changed, 44 insertions(+) diff --git a/include/net/bluetooth/hci_core.h b/include/net/bluetooth/hci_core.h index 4105c446ca983b..c12cd6873f65eb 100644 --- a/include/net/bluetooth/hci_core.h +++ b/include/net/bluetooth/hci_core.h @@ -1030,6 +1030,9 @@ static inline bool hci_conn_sc_enabled(struct hci_conn *conn) static inline void hci_conn_hash_add(struct hci_dev *hdev, struct hci_conn *c) { struct hci_conn_hash *h = &hdev->conn_hash; + + lockdep_assert_held(&hdev->lock); + list_add_tail_rcu(&c->list, &h->list); switch (c->type) { case ACL_LINK: @@ -1060,6 +1063,8 @@ static inline void hci_conn_hash_del(struct hci_dev *hdev, struct hci_conn *c) { struct hci_conn_hash *h = &hdev->conn_hash; + lockdep_assert_held(&hdev->lock); + list_del_rcu(&c->list); synchronize_rcu(); @@ -1088,6 +1093,15 @@ static inline void hci_conn_hash_del(struct hci_dev *hdev, struct hci_conn *c) } } +#ifdef CONFIG_PROVE_RCU +#define HCI_CONN_HASH_LOCKDEP_CHECK(hdev) \ + RCU_LOCKDEP_WARN(!lockdep_is_held(&(hdev)->lock) && \ + !rcu_read_lock_held(), \ + "suspicious hci_conn locking") +#else +#define HCI_CONN_HASH_LOCKDEP_CHECK(hdev) do { } while (0 && (hdev)) +#endif + static inline unsigned int hci_conn_num(struct hci_dev *hdev, __u8 type) { struct hci_conn_hash *h = &hdev->conn_hash; @@ -1169,6 +1183,8 @@ static inline struct hci_conn *hci_conn_hash_lookup_bis(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1191,6 +1207,8 @@ hci_conn_hash_lookup_create_pa_sync(struct hci_dev *hdev) struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1217,6 +1235,8 @@ hci_conn_hash_lookup_per_adv_bis(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1241,6 +1261,8 @@ static inline struct hci_conn *hci_conn_hash_lookup_handle(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1260,6 +1282,8 @@ static inline struct hci_conn *hci_conn_hash_lookup_ba(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1281,6 +1305,8 @@ static inline struct hci_conn *hci_conn_hash_lookup_role(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1302,6 +1328,8 @@ static inline struct hci_conn *hci_conn_hash_lookup_le(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1328,6 +1356,8 @@ static inline struct hci_conn *hci_conn_hash_lookup_cis(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1360,6 +1390,8 @@ static inline struct hci_conn *hci_conn_hash_lookup_cig(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1383,6 +1415,8 @@ static inline struct hci_conn *hci_conn_hash_lookup_big(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1407,6 +1441,8 @@ hci_conn_hash_lookup_big_sync_pend(struct hci_dev *hdev, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1431,6 +1467,8 @@ hci_conn_hash_lookup_big_state(struct hci_dev *hdev, __u8 handle, __u16 state, struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1454,6 +1492,8 @@ hci_conn_hash_lookup_pa_sync_big_handle(struct hci_dev *hdev, __u8 big) struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1477,6 +1517,8 @@ hci_conn_hash_lookup_pa_sync_handle(struct hci_dev *hdev, __u16 sync_handle) struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { @@ -1546,6 +1588,8 @@ static inline struct hci_conn *hci_lookup_le_connect(struct hci_dev *hdev) struct hci_conn_hash *h = &hdev->conn_hash; struct hci_conn *c; + HCI_CONN_HASH_LOCKDEP_CHECK(hdev); + rcu_read_lock(); list_for_each_entry_rcu(c, &h->list, list) { From 26cf20d065b31a9a591ec01086e7661fbd80040c Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 16 Aug 2026 11:59:01 +0300 Subject: [PATCH 149/857] Bluetooth: hci_sync: add conditional locking annotations Add context analysis annotations to functions doing conditional locking, to suppress analysis warnings. Fixes: cdc36db204ff ("Bluetooth: hci_sync: Fix advertising data UAFs") Tested-by: Nathan Chancellor # build Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/hci_sync.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/net/bluetooth/hci_sync.c b/net/bluetooth/hci_sync.c index ea8baf05178e90..7150037a864b44 100644 --- a/net/bluetooth/hci_sync.c +++ b/net/bluetooth/hci_sync.c @@ -1287,6 +1287,7 @@ hci_set_ext_adv_params_sync(struct hci_dev *hdev, u8 instance, } static int hci_set_ext_adv_data_sync(struct hci_dev *hdev, u8 instance) + __context_unsafe(/* conditional locking */) { DEFINE_FLEX(struct hci_cp_le_set_ext_adv_data, pdu, data, length, HCI_MAX_EXT_AD_LENGTH); @@ -1375,6 +1376,7 @@ int hci_update_adv_data_sync(struct hci_dev *hdev, u8 instance) } int hci_setup_ext_adv_instance_sync(struct hci_dev *hdev, u8 instance) + __context_unsafe(/* conditional locking */) { struct hci_cp_le_set_ext_adv_params cp; struct hci_rp_le_set_ext_adv_params rp; @@ -1535,6 +1537,7 @@ int hci_setup_ext_adv_instance_sync(struct hci_dev *hdev, u8 instance) } static int hci_set_ext_scan_rsp_data_sync(struct hci_dev *hdev, u8 instance) + __context_unsafe(/* conditional locking */) { DEFINE_FLEX(struct hci_cp_le_set_ext_scan_rsp_data, pdu, data, length, HCI_MAX_EXT_AD_LENGTH); @@ -1588,6 +1591,7 @@ static int hci_set_ext_scan_rsp_data_sync(struct hci_dev *hdev, u8 instance) } static int __hci_set_scan_rsp_data_sync(struct hci_dev *hdev, u8 instance) + __context_unsafe(/* conditional locking */) { struct hci_cp_le_set_scan_rsp_data cp; u8 len; @@ -1729,6 +1733,7 @@ static int hci_set_per_adv_params_sync(struct hci_dev *hdev, u8 instance, } static int hci_set_per_adv_data_sync(struct hci_dev *hdev, u8 instance) + __context_unsafe(/* conditional locking */) { DEFINE_FLEX(struct hci_cp_le_set_per_adv_data, pdu, data, length, HCI_MAX_PER_AD_LENGTH); From a7b612da9059f045103bad244a0adba9e233c63d Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 16 Aug 2026 12:47:03 +0300 Subject: [PATCH 150/857] Bluetooth: L2CAP: avoid maybe-return-locked in l2cap_get_chan_by_scid/dcid Replace the maybe-return-locked pattern in l2cap_get_chan_by_scid/dcid() by doing locking in the caller after NULL check. This allows adding context analysis annotations for the locking. Reviewed-by: Bart Van Assche Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_core.c | 28 ++++++++++++++++------------ 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index ee459dd411f5db..df41ef95250050 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -109,7 +109,7 @@ static struct l2cap_chan *__l2cap_get_chan_by_scid(struct l2cap_conn *conn, } /* Find channel with given SCID. - * Returns a reference locked channel. + * Returns a reference. */ static struct l2cap_chan *l2cap_get_chan_by_scid(struct l2cap_conn *conn, u16 cid) @@ -117,18 +117,14 @@ static struct l2cap_chan *l2cap_get_chan_by_scid(struct l2cap_conn *conn, struct l2cap_chan *c; c = __l2cap_get_chan_by_scid(conn, cid); - if (c) { - /* Only lock if chan reference is not 0 */ + if (c) c = l2cap_chan_hold_unless_zero(c); - if (c) - l2cap_chan_lock(c); - } return c; } /* Find channel with given DCID. - * Returns a reference locked channel. + * Returns a reference. */ static struct l2cap_chan *l2cap_get_chan_by_dcid(struct l2cap_conn *conn, u16 cid) @@ -136,12 +132,8 @@ static struct l2cap_chan *l2cap_get_chan_by_dcid(struct l2cap_conn *conn, struct l2cap_chan *c; c = __l2cap_get_chan_by_dcid(conn, cid); - if (c) { - /* Only lock if chan reference is not 0 */ + if (c) c = l2cap_chan_hold_unless_zero(c); - if (c) - l2cap_chan_lock(c); - } return c; } @@ -4368,6 +4360,8 @@ static inline int l2cap_config_req(struct l2cap_conn *conn, return 0; } + l2cap_chan_lock(chan); + if (chan->state != BT_CONFIG && chan->state != BT_CONNECT2 && chan->state != BT_CONNECTED) { cmd_reject_invalid_cid(conn, cmd->ident, chan->scid, @@ -4479,6 +4473,8 @@ static inline int l2cap_config_rsp(struct l2cap_conn *conn, if (!chan) return 0; + l2cap_chan_lock(chan); + switch (result) { case L2CAP_CONF_SUCCESS: l2cap_conf_rfc_get(chan, rsp->data, len); @@ -4585,6 +4581,8 @@ static inline int l2cap_disconnect_req(struct l2cap_conn *conn, return 0; } + l2cap_chan_lock(chan); + rsp.dcid = cpu_to_le16(chan->scid); rsp.scid = cpu_to_le16(chan->dcid); l2cap_send_cmd(conn, cmd->ident, L2CAP_DISCONN_RSP, sizeof(rsp), &rsp); @@ -4622,6 +4620,8 @@ static inline int l2cap_disconnect_rsp(struct l2cap_conn *conn, return 0; } + l2cap_chan_lock(chan); + if (chan->state != BT_DISCONN) { l2cap_chan_unlock(chan); l2cap_chan_put(chan); @@ -5124,6 +5124,8 @@ static inline int l2cap_le_credits(struct l2cap_conn *conn, if (!chan) return -EBADSLT; + l2cap_chan_lock(chan); + max_credits = LE_FLOWCTL_MAX_CREDITS - chan->tx_credits; if (credits > max_credits) { BT_ERR("LE credits overflow"); @@ -6983,6 +6985,8 @@ static void l2cap_data_channel(struct l2cap_conn *conn, u16 cid, return; } + l2cap_chan_lock(chan); + BT_DBG("chan %p, len %d", chan, skb->len); /* If we receive data on a fixed channel before the info req/rsp From a3dd57c495646a7b56f7b9e64f37a6d9b27e5254 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 16 Aug 2026 12:47:04 +0300 Subject: [PATCH 151/857] Bluetooth: L2CAP: add locking annotations for l2cap_chan_lock/unlock Add minimal context analysis annotations to l2cap_chan_lock/unlock() and callers required for no warnings. Reviewed-by: Bart Van Assche Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 2 ++ net/bluetooth/l2cap_core.c | 1 + net/bluetooth/l2cap_sock.c | 1 + 3 files changed, 4 insertions(+) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index 3d9a32094347cd..69d193fee351a6 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -830,11 +830,13 @@ struct l2cap_chan *l2cap_chan_hold_unless_zero(struct l2cap_chan *c); void l2cap_chan_put(struct l2cap_chan *c); static inline void l2cap_chan_lock(struct l2cap_chan *chan) + __acquires(&chan->lock) { mutex_lock_nested(&chan->lock, atomic_read(&chan->nesting)); } static inline void l2cap_chan_unlock(struct l2cap_chan *chan) + __releases(&chan->lock) { mutex_unlock(&chan->lock); } diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index df41ef95250050..358b11eabd4f5b 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -4082,6 +4082,7 @@ static struct l2cap_chan *l2cap_new_connection(struct l2cap_conn *conn, static void l2cap_connect(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u8 *data, u8 rsp_code) + __context_unsafe(/* conditional locking */) { struct l2cap_conn_req *req = (struct l2cap_conn_req *) data; struct l2cap_conn_rsp rsp; diff --git a/net/bluetooth/l2cap_sock.c b/net/bluetooth/l2cap_sock.c index 1194c37e466f18..b553b6356af81c 100644 --- a/net/bluetooth/l2cap_sock.c +++ b/net/bluetooth/l2cap_sock.c @@ -1784,6 +1784,7 @@ static void l2cap_sock_state_change_cb(struct l2cap_chan *chan, int state, static struct sk_buff *l2cap_sock_alloc_skb_cb(struct l2cap_chan *chan, unsigned long hdr_len, unsigned long len, int nb) + __must_hold(&chan->lock) { struct sock *sk = chan->data; struct sk_buff *skb; From c519ffc1e2c669296b976d11f5e7a79d2f82debb Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 16 Aug 2026 12:47:05 +0300 Subject: [PATCH 152/857] Bluetooth: enable context analysis for headers Remove context analysis suppression for include/net/bluetooth/*, now that previous commits have resolved the warnings. Reviewed-by: Bart Van Assche Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- scripts/context-analysis-suppression.txt | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/context-analysis-suppression.txt b/scripts/context-analysis-suppression.txt index 1c51b6153f0854..d4476d9ed10a38 100644 --- a/scripts/context-analysis-suppression.txt +++ b/scripts/context-analysis-suppression.txt @@ -32,3 +32,4 @@ src:*include/linux/seqlock*.h=emit src:*include/linux/spinlock*.h=emit src:*include/linux/srcu*.h=emit src:*include/linux/ww_mutex.h=emit +src:*include/net/bluetooth/*=emit From 1fcf216462ec38f634ca1955572fe01372513370 Mon Sep 17 00:00:00 2001 From: Ali Ahmet Memis Date: Fri, 14 Aug 2026 18:28:48 +0000 Subject: [PATCH 153/857] Bluetooth: btnxpuart: Validate the FW dump header length nxp_process_fw_dump() pulls the ACL header off the frame and then reads seq_num and buf_len from a struct nxp_fw_dump_hdr placed at skb->data, without checking that the ACL payload is long enough to contain it. h4_recv_buf() collects HCI_ACL_HDR_SIZE bytes of header followed by the number of payload bytes named in that header, so skb->len is 4 + dlen with dlen supplied by the controller and possibly smaller than the 8 byte dump header, or zero. A short frame with connection handle 0xfff therefore reads both fields from beyond the received data. Beyond the read itself, buf_len is what terminates a dump: a value of zero makes the driver call hci_devcd_complete() and reset the controller, so a truncated frame can end a dump early. Use skb_pull_data() to validate and pull the FW dump header before accessing its fields. Warn and reject the chunk if the header is truncated. Fixes: 998e447f443f ("Bluetooth: btnxpuart: Add support for HCI coredump feature") Signed-off-by: Ali Ahmet Memis Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btnxpuart.c | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/drivers/bluetooth/btnxpuart.c b/drivers/bluetooth/btnxpuart.c index e2b8f7997e4e5a..f2bbe6e462aab4 100644 --- a/drivers/bluetooth/btnxpuart.c +++ b/drivers/bluetooth/btnxpuart.c @@ -1359,12 +1359,21 @@ static int nxp_process_fw_dump(struct hci_dev *hdev, struct sk_buff *skb) { struct hci_acl_hdr *acl_hdr = (struct hci_acl_hdr *)skb_pull_data(skb, sizeof(*acl_hdr)); - struct nxp_fw_dump_hdr *fw_dump_hdr = (struct nxp_fw_dump_hdr *)skb->data; + struct nxp_fw_dump_hdr *fw_dump_hdr; struct btnxpuart_dev *nxpdev = hci_get_drvdata(hdev); - __u16 seq_num = __le16_to_cpu(fw_dump_hdr->seq_num); - __u16 buf_len = __le16_to_cpu(fw_dump_hdr->buf_len); + __u16 seq_num; + __u16 buf_len; int err; + fw_dump_hdr = skb_pull_data(skb, sizeof(*fw_dump_hdr)); + if (!fw_dump_hdr) { + bt_dev_warn(hdev, "FW dump: invalid or corrupt fw dump chunk"); + goto free_skb; + } + + seq_num = __le16_to_cpu(fw_dump_hdr->seq_num); + buf_len = __le16_to_cpu(fw_dump_hdr->buf_len); + if (seq_num == 0x0001) { if (test_and_set_bit(BTNXPUART_FW_DUMP_IN_PROGRESS, &nxpdev->tx_state)) { bt_dev_err(hdev, "FW dump already in progress"); From 14a97a38ba8f2208fd394cebaa3889566d6869f5 Mon Sep 17 00:00:00 2001 From: HyeongJun An Date: Sat, 15 Aug 2026 15:24:19 +0900 Subject: [PATCH 154/857] Bluetooth: eir: Fix OOB read in eir_get_service_data() eir_get_service_data() walks the advertising data for a Service Data field with a matching UUID. On a mismatch it advances: eir += dlen; eir_len -= dlen; eir_get_data() reports dlen as the field's data length, but the field spans dlen + 2 bytes once its length and type bytes count, and more when non-Service-Data fields were skipped to reach it. The pointer lands correctly on the next field. eir_len does not, and the shortfall compounds across fields until eir_get_data() reads the length and type bytes of a "field" past the end of the buffer. For an ISO broadcast sink that buffer is hcon->le_per_adv_data[], filled from the periodic advertising reports of a remote broadcaster. A PA payload packed with mismatching Service Data fields walks off the array into the rest of struct hci_conn. A drifted field that matches the BAA UUID puts those bytes in iso_pi(sk)->base, where user space reads them back with getsockopt(BT_ISO_BASE). Recompute eir_len from the end of the buffer each iteration. Fixes: 8f9ae5b3ae80 ("Bluetooth: eir: Add helpers for managing service data") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-5 Signed-off-by: HyeongJun An Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/eir.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/net/bluetooth/eir.c b/net/bluetooth/eir.c index 1de5f9df6eec00..a55696820b227d 100644 --- a/net/bluetooth/eir.c +++ b/net/bluetooth/eir.c @@ -369,6 +369,7 @@ u8 eir_create_scan_rsp(struct hci_dev *hdev, u8 instance, u8 *ptr) void *eir_get_service_data(u8 *eir, size_t eir_len, u16 uuid, size_t *len) { + const u8 *eir_end = eir + eir_len; size_t dlen; while ((eir = eir_get_data(eir, eir_len, EIR_SERVICE_DATA, &dlen))) { @@ -381,7 +382,7 @@ void *eir_get_service_data(u8 *eir, size_t eir_len, u16 uuid, size_t *len) } eir += dlen; - eir_len -= dlen; + eir_len = eir_end - eir; } return NULL; From 57f558cf1a3d333bd6c9601f48459c1f60ecf5bb Mon Sep 17 00:00:00 2001 From: Ibrahim Abdelkader Date: Mon, 17 Aug 2026 22:50:16 +0200 Subject: [PATCH 155/857] Bluetooth: hci_core: Return -ENOMEM when the sent_cmd clone fails hci_send_cmd_sync() returns -EINVAL when skb_clone() fails for sent_cmd, which describes an invalid argument rather than an allocation failure. Return -ENOMEM instead. The only caller, hci_cmd_work(), tests the result for zero, so there is no functional change. Signed-off-by: Ibrahim Abdelkader Signed-off-by: Hans de Goede Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/hci_core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/net/bluetooth/hci_core.c b/net/bluetooth/hci_core.c index 509c820a693d68..b87eef0479d665 100644 --- a/net/bluetooth/hci_core.c +++ b/net/bluetooth/hci_core.c @@ -4075,7 +4075,7 @@ static int hci_send_cmd_sync(struct hci_dev *hdev, struct sk_buff *skb) if (!hdev->sent_cmd) { skb_queue_head(&hdev->cmd_q, skb); queue_work(hdev->workqueue, &hdev->cmd_work); - return -EINVAL; + return -ENOMEM; } if (hci_skb_opcode(skb) != HCI_OP_NOP) { From 762385e8620095062238d5ce527905a68c552a85 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Tue, 18 Aug 2026 10:49:34 +0100 Subject: [PATCH 156/857] Bluetooth: hci_bcm4377: Ignore reserved PHY in ext adv reports on BCM4378 Commit ed2a2ef16a6b ("Bluetooth: Add quirk to ignore reserved PHY bits in LE Extended Adv Report") added a quirk to handle creative use of the reserved bits in the PHY fields for 4388 controllers in Apple silicon. I observed the same issue with the BCM4378 Bluetooth controller (14e4:5f69, rev 05) on an Apple MacBook Pro (13-inch, M2, 2022): > HCI Event: LE Meta Event (0x3e) plen 51 LE Extended Advertising Report (0x0d) Num reports: 1 Entry 0 Event type: 0x2513 Props: 0x0013 Connectable Scannable Use legacy advertising PDUs Data status: Complete Reserved (0x2500) Legacy PDU Type: Reserved (0x2513) Address type: Random (0x01) Address: EA:C1:82:F0:24:C6 (Static) Primary PHY: Reserved Secondary PHY: No packets SID: no ADI field (0xff) TX power: 127 dBm RSSI: -57 dBm (0xc7) Periodic advertising interval: 0.00 msec (0x0000) Direct address type: Public (0x00) Direct address: 00:00:00:00:00:00 (OUI 00-00-00) Data length: 25 This results in the firmware rejecting connection attempts with "Unsupported Feature or Parameter Value" (0x11). Fix the issue by using the same quirk for BCM4378 devices too. I tested this locally and confirmed that the issue is resolved. This was observed when attempting to connect a Kinesis Advantage 360 keyboard to the MacBook. Assisted-by: Claude:claude-fable-5 Fixes: 2e7ed5f5e69b ("Bluetooth: hci_sync: Use advertised PHYs on hci_le_ext_create_conn_sync") Cc: stable@vger.kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Reviewed-by: Sven Peter Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/hci_bcm4377.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/bluetooth/hci_bcm4377.c b/drivers/bluetooth/hci_bcm4377.c index 925d0a6359453e..66d49b4715440a 100644 --- a/drivers/bluetooth/hci_bcm4377.c +++ b/drivers/bluetooth/hci_bcm4377.c @@ -2490,6 +2490,7 @@ static const struct bcm4377_hw bcm4377_hw_variants[] = { .has_bar0_core2_window2 = true, .broken_mws_transport_config = true, .broken_le_coded = true, + .broken_le_ext_adv_report_phy = true, .send_calibration = bcm4378_send_calibration, .send_ptb = bcm4378_send_ptb, }, From b03f74d42e24970bb20a3044ad8ccfe04ead61a7 Mon Sep 17 00:00:00 2001 From: Shuai Zhang Date: Tue, 18 Aug 2026 19:41:16 +0800 Subject: [PATCH 157/857] Bluetooth: mgmt: reply to cancelled mgmt commands instead of silently dropping The kernel sets HCI_AUTO_OFF when a controller is first registered and starts a 2-second timer. On slower boots bluetoothd and the HCI_AUTO_OFF timer can race: hci_power_off() is already queued while bluetoothd is still in the middle of its adapter setup sequence. hci_cmd_sync_clear() then cancels any pending mgmt commands with -ECANCELED, including the MGMT_OP_REMOVE_ADV_MONITOR sent by reset_adv_monitors() early in the setup sequence. When auto_off=1, hci_dev_close_sync() skips __mgmt_power_off() entirely, so there is no fallback path to reply to the cancelled commands. mgmt_remove_adv_monitor_complete() silently returns on -ECANCELED, leaving the command with no reply. Since bluez's mgmt queue is strictly serialised, this stalls all subsequent commands indefinitely, leaving bluetoothd unable to register the adapter. Fix by mapping -ECANCELED to MGMT_STATUS_CANCELLED in mgmt_errno_status() and replying to the cancelled command in mgmt_remove_adv_monitor_complete() instead of returning early. Signed-off-by: Shuai Zhang Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/mgmt.c | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/net/bluetooth/mgmt.c b/net/bluetooth/mgmt.c index ac4864e56ec727..fd045460e2365b 100644 --- a/net/bluetooth/mgmt.c +++ b/net/bluetooth/mgmt.c @@ -301,6 +301,8 @@ static u8 mgmt_errno_status(int err) return MGMT_STATUS_ALREADY_CONNECTED; case -ENOTCONN: return MGMT_STATUS_DISCONNECTED; + case -ECANCELED: + return MGMT_STATUS_CANCELLED; } return MGMT_STATUS_FAILED; @@ -5675,8 +5677,18 @@ static void mgmt_remove_adv_monitor_complete(struct hci_dev *hdev, struct mgmt_pending_cmd *cmd = data; struct mgmt_cp_remove_adv_monitor *cp; - if (status == -ECANCELED) + /* Reply to a cancelled command so bluetoothd's serialised mgmt queue + * is not blocked. + */ + if (status == -ECANCELED) { + cp = cmd->param; + rp.monitor_handle = cp->monitor_handle; + + mgmt_cmd_complete(cmd->sk, cmd->hdev->id, cmd->opcode, + mgmt_status(status), &rp, sizeof(rp)); + mgmt_pending_free(cmd); return; + } hci_dev_lock(hdev); From d2d1c215ce1170b219fff16e7c0471b316b97685 Mon Sep 17 00:00:00 2001 From: Valentin Kindschi Date: Tue, 18 Aug 2026 15:29:34 +0200 Subject: [PATCH 158/857] Bluetooth: hci_conn: re-enable advertising only for peripheral role hci_le_conn_failed() unconditionally calls hci_enable_advertising(), although its own comment states advertising should be re-enabled only when the failed attempt was made as a peripheral. hci_le_conn_failed() is reached from hci_conn_failed() for every failed LE connection, including outgoing central connections. For a central attempt this enable is redundant: hci_le_create_conn_sync() already restores advertising via hci_resume_advertising_sync() in its done: block. Because hci_enable_advertising() only queues the work on cmd_sync_work, it runs *after* that resume has already succeeded and set HCI_LE_ADV. The resulting HCI sequence, captured on a BCM43455 (no LE Extended Advertising, so legacy advertising is used): LE Create Connection Status Success ... 13.8 s, peer never answers ... LE Set Advertising Parameters (0x2006) Success <- done: resume, LE Set Advertising Enable (0x200a) Success HCI_LE_ADV set LE Create Connection Cancel (0x200e) Success LE Connection Complete Unknown Conn Id LE Set Advertising Parameters (0x2006) Command Disallowed (0x0c) The last command is the queued enable from hci_le_conn_failed() running as a second hci_enable_advertising_sync() pass. It clears HCI_LE_ADV (hci_sync.c, "Clear the HCI_LE_ADV bit temporarily"), then sends LE Set Advertising Parameters while the controller is still advertising, which the controller correctly rejects with Command Disallowed. The disable-first call at the top of hci_enable_advertising_sync() cannot prevent this: hci_disable_advertising_sync() returns early without sending anything when HCI_LE_ADV is clear, so it is a no-op exactly when the flag is wrong. hci_enable_advertising_sync() then returns without sending LE Set Advertising Enable, so HCI_LE_ADV is never set again. The legacy software rotation loop re-arms hci_schedule_adv_instance_sync() every HCI_DEFAULT_ADV_DURATION (2 s), and its "already advertising" shortcut tests HCI_LE_ADV, which can no longer become true. The command is therefore retried every 2 s indefinitely: Bluetooth: hci0: Opcode 0x2006 failed: -16 Observed on a gateway as 5326 occurrences over 3 hours, ending only when bluetoothd was restarted. Connection attempts that succeed do not call hci_le_conn_failed() and never trigger this. Add the role test the comment already describes. Both other hci_enable_advertising() call sites reached from a failed/closed LE connection (hci_cs_disconnect() and hci_disconn_complete_evt()) already guard on conn->role == HCI_ROLE_SLAVE; this one was missed. Reproducing needs legacy advertising (ext_adv_capable() false, so the software rotation loop is used), simultaneous peripheral advertising and outgoing central connects, and a central connect that times out rather than failing fast. The Fixes tag points at the commit that introduced the advertising restart into this path for the directed-advertising (peripheral) case; the role test that the later commit 0b1db38ca26b ("Bluetooth: Fix check for direct advertising") added to the sibling paths was never applied here. Fixes: 3c857757ef6e ("Bluetooth: Add directed advertising support through connect()") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-5 btmon Signed-off-by: Valentin Kindschi Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/hci_conn.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/net/bluetooth/hci_conn.c b/net/bluetooth/hci_conn.c index 19b7629b1cc101..8de98af2fb5818 100644 --- a/net/bluetooth/hci_conn.c +++ b/net/bluetooth/hci_conn.c @@ -1391,7 +1391,8 @@ static void hci_le_conn_failed(struct hci_conn *conn, u8 status) /* Enable advertising in case this was a failed connection * attempt as a peripheral. */ - hci_enable_advertising(hdev); + if (conn->role == HCI_ROLE_SLAVE) + hci_enable_advertising(hdev); } /* This function requires the caller holds hdev->lock */ From e4d6d16f1be19752f1119555b8cdb388ef2964c9 Mon Sep 17 00:00:00 2001 From: Valentin Kindschi Date: Tue, 18 Aug 2026 15:29:35 +0200 Subject: [PATCH 159/857] Bluetooth: hci_event: clear HCI_LE_ADV only on a created connection le_conn_complete_evt() clears HCI_LE_ADV before looking at the event status, on the premise stated in its comment that all controllers stop advertising when a connection is created. That premise only holds when a connection was actually created. On a non-zero status none was, and the controller is still advertising: after the host issues LE Create Connection Cancel the event arrives with Unknown Connection Identifier (0x02), and a connection timeout behaves the same way. Clearing the flag there leaves the host believing advertising is off while the controller has it on. It is also wrong for extended advertising, where several sets can be advertising at once. hci_cc_le_set_ext_adv_enable() is careful about this - on disabling one set it walks hdev->adv_instances and only clears HCI_LE_ADV once no instance is still enabled. The unconditional clear here discards that bookkeeping, so one set connecting drops the flag while the others keep advertising. The direction of the error matters. A flag left set is self-correcting: hci_disable_advertising_sync() sends LE Set Advertising Enable(0) and the command complete puts the state back. A flag left clear is not, because that same function returns early without sending anything while the flag is clear: - LE Set Advertising Parameters is then sent to a controller that is still advertising, and is correctly rejected with Command Disallowed (0x0c); - hci_enable_advertising_sync() returns at that point, before the LE Set Advertising Enable that would set HCI_LE_ADV again. On a controller without LE Extended Advertising that is reachable from here: hci_schedule_adv_instance_sync() re-arms adv_instance_expire every HCI_DEFAULT_ADV_DURATION (2 s) and its "already advertising" shortcut tests HCI_LE_ADV, which can no longer become true, so the parameter write is retried for as long as advertising is configured: Bluetooth: hci0: Opcode 0x2006 failed: -16 Only clear the flag when a connection was established. Note this is not on its own sufficient to stop that retry loop - the redundant enable queued by hci_le_conn_failed() clears HCI_LE_ADV itself and recreates the same mismatch, which patch 1 addresses. This patch fixes the event handler reporting a state the controller is not in. Verified on the affected device (BCM43455, legacy advertising only) with this patch and patch 1 applied. A 221 s btmon capture with an out-of-range peer at -90 dBm contains two outgoing connection attempts that the host cancelled, each producing exactly the event this patch changes: < LE Set Advertising Parameters 0x2006 Success < LE Set Advertising Enable 0x200a Success < LE Create Connection Cancel 0x200e Success > LE Connection Complete Unknown Connection Identifier (0x02), central Nothing follows either one; the next command is an unrelated scan restart 70 ms later. Over the whole capture: 7 LE Set Advertising Parameters sent, all Success; 10 LE Set Advertising Enable, all Success; no Command Disallowed of any opcode, and no 2 s cadence anywhere. Two central connections to other peers completed normally afterwards, with feature exchange and a connection parameter update, so advertising was still live across the cancelled attempts. The extended advertising case above is a code argument, not a measurement: this controller has no LE Extended Advertising, so that path is not exercised by the capture. Fixes: fbd96c151cdc ("Bluetooth: Fix clearing HCI_LE_ADV for LE connections") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-5 btmon Signed-off-by: Valentin Kindschi Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/hci_event.c | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/net/bluetooth/hci_event.c b/net/bluetooth/hci_event.c index 3eb1eaf6e6a0ad..2f5e21ff975294 100644 --- a/net/bluetooth/hci_event.c +++ b/net/bluetooth/hci_event.c @@ -5763,10 +5763,11 @@ static void le_conn_complete_evt(struct hci_dev *hdev, u8 status, hci_dev_lock(hdev); hci_store_wake_reason(hdev, bdaddr, bdaddr_type); - /* All controllers implicitly stop advertising in the event of a - * connection, so ensure that the state bit is cleared. + /* Advertising stops when a connection is created. On a failed + * connection it keeps running, so leave the state bit alone. */ - hci_dev_clear_flag(hdev, HCI_LE_ADV); + if (!status) + hci_dev_clear_flag(hdev, HCI_LE_ADV); /* Check for existing connection: * From 67db795c498c273851c2be081a45bbe535141046 Mon Sep 17 00:00:00 2001 From: Ruoyu Wang Date: Sat, 15 Aug 2026 23:17:20 +0800 Subject: [PATCH 160/857] i2c: mxs: fix DMA channel leak on probe error mxs_i2c_probe() requests an exclusive DMA channel before resetting the controller and registering the I2C adapter. If either later operation fails, probe returns without releasing the channel because the remove callback is not invoked after a failed probe. Use devm_dma_request_chan() so the device core releases the channel on probe failure and driver detach. Remove the manual release from the remove callback because the channel is now device-managed. This issue was found by a static analysis checker and confirmed by manual source review. Fixes: 62885f59a261 ("MXS: Implement DMA support into mxs-i2c") Assisted-by: unnamed:claude-opus-4.8 typestate Signed-off-by: Ruoyu Wang Cc: # v3.7+ Reviewed-by: Frank Li Signed-off-by: Andi Shyti Link: https://patch.msgid.link/20260815151720.3757460-1-ruoyuw560@gmail.com --- drivers/i2c/busses/i2c-mxs.c | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/drivers/i2c/busses/i2c-mxs.c b/drivers/i2c/busses/i2c-mxs.c index 4e07babea9c3f4..eee4fdcd9df31a 100644 --- a/drivers/i2c/busses/i2c-mxs.c +++ b/drivers/i2c/busses/i2c-mxs.c @@ -839,7 +839,7 @@ static int mxs_i2c_probe(struct platform_device *pdev) } /* Setup the DMA */ - i2c->dmach = dma_request_chan(dev, "rx-tx"); + i2c->dmach = devm_dma_request_chan(dev, "rx-tx"); if (IS_ERR(i2c->dmach)) { return dev_err_probe(dev, PTR_ERR(i2c->dmach), "Failed to request dma\n"); @@ -877,9 +877,6 @@ static void mxs_i2c_remove(struct platform_device *pdev) i2c_del_adapter(&i2c->adapter); - if (i2c->dmach) - dma_release_channel(i2c->dmach); - writel(MXS_I2C_CTRL0_SFTRST, i2c->regs + MXS_I2C_CTRL0_SET); } From be766d775060e0049e7ee798b968d3bd649eb553 Mon Sep 17 00:00:00 2001 From: Xin Chen Date: Wed, 19 Aug 2026 21:53:21 +0800 Subject: [PATCH 161/857] Bluetooth: hci_core: use skb_get() instead of skb_clone() for req_skb BT enable fails intermittently with -ETIMEDOUT (-110). The kernel log shows the HCI Read Local Version command was sent and the firmware replied with status 0x00 (logged by hci_req_cmd_complete() BT_DBG), but the waiter in __hci_cmd_sync_sk() never woke up and timed out after 10 s: bluetooth hci0: Opcode 0xfc00 // __hci_cmd_sync_sk bluetooth hci0: opcode 0xfc00 plen 1 // hci_cmd_sync_add bluetooth hci0: skb len 4 // hci_cmd_sync_alloc bluetooth hci0: length 1 // hci_req_sync_run Bluetooth: hci0 cmd_cnt 1 cmd queued 1 // hci_cmd_work Bluetooth: hci0 type 1 len 4 // hci_send_frame Bluetooth: opcode 0xfc00 status 0x00 // hci_req_cmd_complete <-- req_skb NULL: req_complete_skb not set, hci_cmd_sync_complete() never called, req_status stays HCI_REQ_PEND --> <-- 10 s later: wait_event_interruptible_timeout expires --> bluetooth hci0: end: err -110 // __hci_cmd_sync_sk The root cause is that hci_send_cmd_sync() clones the sent command into hdev->req_skb so that hci_req_cmd_complete() can locate the registered completion callback. Under memory pressure this skb_clone() fails, leaving hdev->req_skb NULL. The firmware reply is received and processed, but hci_req_cmd_complete() finds NULL req_skb, so hci_cmd_sync_complete() is never called, req_status stays HCI_REQ_PEND, and the waiter times out with -ETIMEDOUT. req_skb is only used to read bt_cb(skb)->hci callbacks and opcode -- it is never modified. Replace skb_clone() with skb_get(), which simply increments the reference count of hdev->sent_cmd without allocating new memory and therefore cannot fail. This issue was first observed as a use-after-free in ttyport_close() when ttyport_open() failed, which was investigated in an earlier patch series [1]. That investigation led to the discovery of the true root cause described above. [1] https://lore.kernel.org/all/20250430111617.1151390-1-quic_cxin@quicinc.com/ Fixes: 2615fd9a7c25 ("Bluetooth: hci_sync: Fix overwriting request callback") Cc: stable@vger.kernel.org Signed-off-by: Xin Chen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/hci_core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/net/bluetooth/hci_core.c b/net/bluetooth/hci_core.c index b87eef0479d665..88df159d339371 100644 --- a/net/bluetooth/hci_core.c +++ b/net/bluetooth/hci_core.c @@ -4093,7 +4093,7 @@ static int hci_send_cmd_sync(struct hci_dev *hdev, struct sk_buff *skb) if (READ_ONCE(hdev->req_status) == HCI_REQ_PEND && !hci_dev_test_and_set_flag(hdev, HCI_CMD_PENDING)) { kfree_skb(hdev->req_skb); - hdev->req_skb = skb_clone(hdev->sent_cmd, GFP_KERNEL); + hdev->req_skb = skb_get(hdev->sent_cmd); } return err; From 6f5f8dcf62444590a66c463a487892aa9cf0d971 Mon Sep 17 00:00:00 2001 From: Chandrashekar Devegowda Date: Wed, 19 Aug 2026 20:02:17 +0530 Subject: [PATCH 162/857] Bluetooth: btintel_pcie: parse FW memory addresses via mailbox TLV Implement GP1 mailbox interrupt handling to receive memory region addresses from firmware via a TLV-based protocol. When firmware sends a BUILD_SPECIFIC_RESOURCES_MAPPING mailbox message, the driver reads a TLV table from device memory containing addresses and sizes of debug memory regions (exception dump, DCCM, SDS, ECL, SMEM). This enables the driver to dynamically discover dump region locations instead of using hardcoded addresses, supporting current and future Intel BT PCIe controller variants. Replace per-device hardcoded exception memory address and size constants in btintel_pcie_read_hwexp() with the dynamically populated values from dump_info, making exception dump handling consistent with other dump regions. Key changes: - Rewrite GP1 handler to parse mailbox registers and queue TLV work - Add btintel_parse_mbox_tlv() for parsing FW-provided TLV data - Add mbox_work workqueue for deferred TLV processing - Store parsed region addresses in btintel_pcie_dump_mem_info - Add cnvi_bt field to btintel_data for HW variant identification - Rename fw_git_sha1 to fw_sha for consistency - Remove hardcoded HWEXP address/size macros and use dump_info Assisted-by: GitHub-Copilot:claude-opus-4.7 Signed-off-by: Chandrashekar Devegowda Signed-off-by: Kiran K Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btintel.c | 7 + drivers/bluetooth/btintel.h | 1 + drivers/bluetooth/btintel_pcie.c | 349 +++++++++++++++++++++++++++++-- drivers/bluetooth/btintel_pcie.h | 53 ++++- 4 files changed, 387 insertions(+), 23 deletions(-) diff --git a/drivers/bluetooth/btintel.c b/drivers/bluetooth/btintel.c index bcb2514b7bc027..cbeb27033aa1e7 100644 --- a/drivers/bluetooth/btintel.c +++ b/drivers/bluetooth/btintel.c @@ -66,6 +66,7 @@ static struct { const char *driver_name; u8 hw_variant; u32 fw_build_num; + u32 fw_sha; } coredump_info; const guid_t btintel_guid_dsm = @@ -560,6 +561,7 @@ int btintel_version_info_tlv(struct hci_dev *hdev, coredump_info.hw_variant = INTEL_HW_VARIANT(version->cnvi_bt); coredump_info.fw_build_num = version->build_num; + coredump_info.fw_sha = version->git_sha1; bt_dev_info(hdev, "%s timestamp %u.%u buildtype %u build %u", variant, 2000 + (version->timestamp >> 8), version->timestamp & 0xff, @@ -2337,6 +2339,7 @@ static int btintel_prepare_fw_download_tlv(struct hci_dev *hdev, struct intel_version_tlv *ver, u32 *boot_param) { + struct btintel_data *intel_data = hci_get_priv(hdev); const struct firmware *fw; char fwname[128]; int err; @@ -2449,6 +2452,7 @@ static int btintel_prepare_fw_download_tlv(struct hci_dev *hdev, btintel_reset_to_bootloader(hdev); done: + intel_data->cnvi_bt = ver->cnvi_bt; release_firmware(fw); return err; } @@ -3486,6 +3490,9 @@ int btintel_bootloader_setup_tlv(struct hci_dev *hdev, btintel_version_info_tlv(hdev, &new_ver); + /* Update ver with the operational firmware version */ + *ver = new_ver; + finish: /* Set the event mask for Intel specific vendor events. This enables * a few extra events that are useful during general operation. It diff --git a/drivers/bluetooth/btintel.h b/drivers/bluetooth/btintel.h index 966ec1b02be257..ef232820c31b56 100644 --- a/drivers/bluetooth/btintel.h +++ b/drivers/bluetooth/btintel.h @@ -247,6 +247,7 @@ enum { struct btintel_data { DECLARE_BITMAP(flags, __INTEL_NUM_FLAGS); int (*acpi_reset_method)(struct hci_dev *hdev); + u32 cnvi_bt; }; #define btintel_set_flag(hdev, nr) \ diff --git a/drivers/bluetooth/btintel_pcie.c b/drivers/bluetooth/btintel_pcie.c index 005c77a4f5eb49..4d1b6e9a2bdc7b 100644 --- a/drivers/bluetooth/btintel_pcie.c +++ b/drivers/bluetooth/btintel_pcie.c @@ -75,14 +75,8 @@ struct btintel_pcie_dev_recovery { #define BTINTEL_PCIE_MAGIC_NUM 0xA5A5A5A5 -#define BTINTEL_PCIE_BLZR_HWEXP_SIZE 1024 -#define BTINTEL_PCIE_BLZR_HWEXP_DMP_ADDR 0xB00A7C00 -#define BTINTEL_PCIE_SCP_HWEXP_SIZE 4096 -#define BTINTEL_PCIE_SCP_HWEXP_DMP_ADDR 0xB030F800 -#define BTINTEL_PCIE_SCP2_HWEXP_SIZE 4096 -#define BTINTEL_PCIE_SCP2_HWEXP_DMP_ADDR 0xB031D000 #define BTINTEL_PCIE_MAGIC_NUM 0xA5A5A5A5 @@ -726,7 +720,7 @@ static int btintel_pcie_read_dram_buffers(struct btintel_pcie_data *data) sizeof(*tlv) + sizeof(data->dmp_hdr.write_ptr) + sizeof(*tlv) + sizeof(data->dmp_hdr.wrap_ctr) + sizeof(*tlv) + sizeof(data->dmp_hdr.trigger_reason) + - sizeof(*tlv) + sizeof(data->dmp_hdr.fw_git_sha1) + + sizeof(*tlv) + sizeof(data->dmp_hdr.fw_sha) + sizeof(*tlv) + sizeof(data->dmp_hdr.cnvr_top) + sizeof(*tlv) + sizeof(data->dmp_hdr.cnvi_top) + sizeof(*tlv) + strlen(ts) + @@ -775,8 +769,8 @@ static int btintel_pcie_read_dram_buffers(struct btintel_pcie_data *data) sizeof(data->dmp_hdr.wrap_ctr)); p = btintel_pcie_copy_tlv(p, BTINTEL_TRIGGER_REASON, &data->dmp_hdr.trigger_reason, sizeof(data->dmp_hdr.trigger_reason)); - p = btintel_pcie_copy_tlv(p, BTINTEL_FW_SHA, &data->dmp_hdr.fw_git_sha1, - sizeof(data->dmp_hdr.fw_git_sha1)); + p = btintel_pcie_copy_tlv(p, BTINTEL_FW_SHA, &data->dmp_hdr.fw_sha, + sizeof(data->dmp_hdr.fw_sha)); p = btintel_pcie_copy_tlv(p, BTINTEL_CNVR_TOP, &data->dmp_hdr.cnvr_top, sizeof(data->dmp_hdr.cnvr_top)); p = btintel_pcie_copy_tlv(p, BTINTEL_CNVI_TOP, &data->dmp_hdr.cnvi_top, @@ -957,10 +951,297 @@ static inline bool btintel_pcie_in_error(struct btintel_pcie_data *data) return data->boot_stage_cache & BTINTEL_PCIE_CSR_BOOT_STAGE_ABORT_HANDLER; } +static const char *btintel_pcie_tlv_str(u8 tlv_type) +{ + switch (tlv_type) { + case BTINTEL_PCIE_TLV_TYPE_EXCEPTION_DUMP_ADDRESS: + return "EXCEPTION_DUMP_ADDRESS"; + case BTINTEL_PCIE_TLV_TYPE_DCCM_MEM_ADDRESS: + return "DCCM_MEM_ADDRESS"; + case BTINTEL_PCIE_TLV_TYPE_SDS_MEM_ADDRESS: + return "SDS_MEM_ADDRESS"; + case BTINTEL_PCIE_TLV_TYPE_ECL_MEM_ADDRESS: + return "ECL_MEM_ADDRESS"; + case BTINTEL_PCIE_TLV_TYPE_SMEM_ADDRESS: + return "SMEM_ADDRESS"; + default: + return "UNKNOWN"; + } +} + +static int btintel_parse_mbox_tlv(struct btintel_pcie_data *data) +{ + /* Custom TLV structure for mailbox parsing + * len is __le16 as per agreement with FW + */ + struct mbox_tlv { + u8 type; + __le16 len; + u8 val[]; + } __packed; + + u8 *buffer, *ptr; + u32 buffer_len, remaining; + int err; + u32 tbl_addr, tbl_size; + struct mbox_tlv *tlv; + struct btintel_data *cnvi_data = hci_get_priv(data->hdev); + u8 hw_variant = INTEL_HW_VARIANT(cnvi_data->cnvi_bt); + + memset(&data->dump_info, 0, sizeof(data->dump_info)); + + /* Snapshot to avoid TOCTOU with the GP1 IRQ handler */ + tbl_size = READ_ONCE(data->debug_table_size); + tbl_addr = READ_ONCE(data->debug_table_addr); + + if (!tbl_size || !tbl_addr) + return -EINVAL; + + /* Ensure size is 4-byte aligned; btintel_pcie_read_device_mem() + * reads device memory in 4-byte units. + */ + tbl_size = ALIGN_DOWN(tbl_size, 4); + + if (!tbl_size) + return -EINVAL; + + if (tbl_size > SZ_1M) { + bt_dev_err(data->hdev, "Debug table size too large: %u", + tbl_size); + return -EINVAL; + } + + buffer_len = tbl_size; + + buffer = vmalloc(buffer_len); + if (!buffer) + return -ENOMEM; + + btintel_pcie_mac_init(data); + + err = btintel_pcie_read_device_mem(data, buffer, tbl_addr, + buffer_len); + if (err) + goto exit_on_error; + + print_hex_dump(KERN_INFO, "Bluetooth: mbox_tlv: ", DUMP_PREFIX_OFFSET, 16, 1, + buffer, buffer_len, false); + + ptr = buffer; + remaining = buffer_len; + + /* Parse TLV structures: type(1) + length(2) + value */ + while (remaining >= sizeof(struct mbox_tlv)) { + u16 tlv_len; + u32 tlv_total; + + tlv = (struct mbox_tlv *)ptr; + tlv_len = le16_to_cpu(tlv->len); + tlv_total = sizeof(tlv->type) + + sizeof(tlv->len) + tlv_len; + + if (tlv_total > remaining) { + bt_dev_err(data->hdev, + "TLV parse error: type=%u, len=%u", + tlv->type, tlv_len); + break; + } + + switch (tlv->type) { + case BTINTEL_PCIE_TLV_TYPE_EXCEPTION_DUMP_ADDRESS: + if (tlv_len < 8) { + bt_dev_err(data->hdev, + "TLV %s too short: %u", + btintel_pcie_tlv_str(tlv->type), + tlv_len); + break; + } + data->dump_info.exception_dump_addr = + get_unaligned_le32(&tlv->val[0]); + data->dump_info.exception_dump_len = + get_unaligned_le32(&tlv->val[4]); + break; + case BTINTEL_PCIE_TLV_TYPE_DCCM_MEM_ADDRESS: + if (tlv_len < 8) { + bt_dev_err(data->hdev, + "TLV %s too short: %u", + btintel_pcie_tlv_str(tlv->type), + tlv_len); + break; + } + data->dump_info.dccm_addr_start = + get_unaligned_le32(&tlv->val[0]); + data->dump_info.dccm_addr_end = + get_unaligned_le32(&tlv->val[4]); + break; + case BTINTEL_PCIE_TLV_TYPE_SDS_MEM_ADDRESS: + /* hw_variant comes from cnvi_bt which is set during + * setup. If mailbox fires before setup completes, + * hw_variant is 0. Skip SDS parsing in that case. + */ + if (!hw_variant) { + bt_dev_dbg(data->hdev, "SDS TLV: skipped, hw_variant not yet known"); + break; + } + if (tlv_len == 16 && + hw_variant > BTINTEL_HWID_BZRI) { + data->dump_info.sds_start_addr_start = + get_unaligned_le32(&tlv->val[0]); + data->dump_info.sds_start_addr_end = + get_unaligned_le32(&tlv->val[4]); + data->dump_info.sds_iosf_data_addr_start = + get_unaligned_le32(&tlv->val[8]); + data->dump_info.sds_iosf_data_addr_end = + get_unaligned_le32(&tlv->val[12]); + } else if (tlv_len == 24 && + (hw_variant == BTINTEL_HWID_BZRI || + hw_variant == BTINTEL_HWID_BZRIW)) { + data->dump_info.sds_fixed_rom_addr_start = + get_unaligned_le32(&tlv->val[0]); + data->dump_info.sds_fixed_rom_addr_end = + get_unaligned_le32(&tlv->val[4]); + data->dump_info.sds_start_addr_start = + get_unaligned_le32(&tlv->val[8]); + data->dump_info.sds_start_addr_end = + get_unaligned_le32(&tlv->val[12]); + data->dump_info.sds_iosf_data_addr_start = + get_unaligned_le32(&tlv->val[16]); + data->dump_info.sds_iosf_data_addr_end = + get_unaligned_le32(&tlv->val[20]); + } else { + bt_dev_err(data->hdev, + "SDS TLV: hw=0x%2.2x len=%u", + hw_variant, tlv_len); + } + break; + case BTINTEL_PCIE_TLV_TYPE_ECL_MEM_ADDRESS: + if (tlv_len < 8) { + bt_dev_err(data->hdev, + "TLV %s too short: %u", + btintel_pcie_tlv_str(tlv->type), + tlv_len); + break; + } + data->dump_info.ecl_addr_start = + get_unaligned_le32(&tlv->val[0]); + data->dump_info.ecl_addr_end = + get_unaligned_le32(&tlv->val[4]); + break; + case BTINTEL_PCIE_TLV_TYPE_SMEM_ADDRESS: + if (tlv_len < 8) { + bt_dev_err(data->hdev, + "TLV %s too short: %u", + btintel_pcie_tlv_str(tlv->type), + tlv_len); + break; + } + data->dump_info.smem_addr_start = + get_unaligned_le32(&tlv->val[0]); + data->dump_info.smem_addr_end = + get_unaligned_le32(&tlv->val[4]); + break; + default: + bt_dev_dbg(data->hdev, + "Unknown TLV type: %u length: %u", + tlv->type, tlv_len); + break; + } + + /* Move to next TLV */ + ptr += tlv_total; + remaining -= tlv_total; + } + + bt_dev_info(data->hdev, + "exception_dump: addr:0x%08x len:0x%08x", + data->dump_info.exception_dump_addr, + data->dump_info.exception_dump_len); + bt_dev_info(data->hdev, + "dccm: start:0x%08x end:0x%08x", + data->dump_info.dccm_addr_start, + data->dump_info.dccm_addr_end); + bt_dev_info(data->hdev, + "sds_fixed_rom: start:0x%08x end:0x%08x", + data->dump_info.sds_fixed_rom_addr_start, + data->dump_info.sds_fixed_rom_addr_end); + bt_dev_info(data->hdev, + "sds: start:0x%08x end:0x%08x", + data->dump_info.sds_start_addr_start, + data->dump_info.sds_start_addr_end); + bt_dev_info(data->hdev, + "sds_iosf: start:0x%08x end:0x%08x", + data->dump_info.sds_iosf_data_addr_start, + data->dump_info.sds_iosf_data_addr_end); + bt_dev_info(data->hdev, + "ecl: start:0x%08x end:0x%08x", + data->dump_info.ecl_addr_start, + data->dump_info.ecl_addr_end); + bt_dev_info(data->hdev, + "smem: start:0x%08x end:0x%08x", + data->dump_info.smem_addr_start, + data->dump_info.smem_addr_end); + + vfree(buffer); + return 0; + +exit_on_error: + vfree(buffer); + return err; +} + static void btintel_pcie_msix_gp1_handler(struct btintel_pcie_data *data) { - bt_dev_err(data->hdev, "Received gp1 mailbox interrupt"); - btintel_pcie_dump_debug_registers(data->hdev); + bool target_access = false; + u32 addr = 0, size = 0; + + /* Read the Mail box status and registers */ + data->mbox.mbox_status = btintel_pcie_rd_reg32(data, BTINTEL_PCIE_CSR_MBOX_STATUS_REG); + if (data->mbox.mbox_status & BTINTEL_PCIE_CSR_MBOX_STATUS_MBOX1) { + data->mbox.mbox1 = btintel_pcie_rd_reg32(data, BTINTEL_PCIE_CSR_MBOX_1_REG); + if (data->mbox.mbox1 == + BTINTEL_PCIE_BUILD_SPECIFIC_RESOURCES_MAPPING) { + bt_dev_info(data->hdev, + "mailbox for target access"); + target_access = true; + } + } + + if (data->mbox.mbox_status & BTINTEL_PCIE_CSR_MBOX_STATUS_MBOX2) { + data->mbox.mbox2 = btintel_pcie_rd_reg32(data, BTINTEL_PCIE_CSR_MBOX_2_REG); + if (target_access) + addr = data->mbox.mbox2; + } + + if (data->mbox.mbox_status & BTINTEL_PCIE_CSR_MBOX_STATUS_MBOX3) { + data->mbox.mbox3 = btintel_pcie_rd_reg32(data, BTINTEL_PCIE_CSR_MBOX_3_REG); + if (target_access) + size = data->mbox.mbox3; + } + + if (data->mbox.mbox_status & BTINTEL_PCIE_CSR_MBOX_STATUS_MBOX4) + data->mbox.mbox4 = btintel_pcie_rd_reg32(data, BTINTEL_PCIE_CSR_MBOX_4_REG); + + bt_dev_dbg(data->hdev, + "GP1: sts:0x%08x mb1:0x%08x mb2:0x%08x mb3:0x%08x mb4:0x%08x", + data->mbox.mbox_status, data->mbox.mbox1, + data->mbox.mbox2, data->mbox.mbox3, + data->mbox.mbox4); + + if (target_access && + !test_and_set_bit(BTINTEL_PCIE_MAIL_BOX_INTR, + &data->flags)) { + WRITE_ONCE(data->debug_table_addr, addr); + WRITE_ONCE(data->debug_table_size, size); + if (!queue_work(data->dump_workqueue, + &data->mbox_work)) + clear_bit(BTINTEL_PCIE_MAIL_BOX_INTR, + &data->flags); + } + + /* Mailbox is read, ack to FW */ + btintel_pcie_set_reg_bits(data, + BTINTEL_PCIE_CSR_IPC_DOORBELL_VEC_REG, + BTINTEL_PCIE_CSR_DOORBELL_MBOX_READ_CONFIRM); } /* This function handles the MSI-X interrupt for gp0 cause (bit 0 in @@ -1287,7 +1568,8 @@ static int btintel_pcie_recv_frame(struct btintel_pcie_data *data, static void btintel_pcie_read_hwexp(struct btintel_pcie_data *data) { - int len, err, offset, pending; + int err, offset, pending; + u32 len; struct sk_buff *skb; u8 *buf, prefix[64]; u32 addr, val; @@ -1307,23 +1589,30 @@ static void btintel_pcie_read_hwexp(struct btintel_pcie_data *data) /* only from step B0 onwards */ if (INTEL_CNVX_TOP_STEP(data->dmp_hdr.cnvi_top) != 0x01) return; - len = BTINTEL_PCIE_BLZR_HWEXP_SIZE; /* exception data length */ - addr = BTINTEL_PCIE_BLZR_HWEXP_DMP_ADDR; break; case BTINTEL_CNVI_SCP: - len = BTINTEL_PCIE_SCP_HWEXP_SIZE; - addr = BTINTEL_PCIE_SCP_HWEXP_DMP_ADDR; - break; case BTINTEL_CNVI_SCP2: case BTINTEL_CNVI_SCP2F: - len = BTINTEL_PCIE_SCP2_HWEXP_SIZE; - addr = BTINTEL_PCIE_SCP2_HWEXP_DMP_ADDR; break; default: bt_dev_err(data->hdev, "Unsupported cnvi 0x%8.8x", data->dmp_hdr.cnvi_top); return; } + len = data->dump_info.exception_dump_len; + addr = data->dump_info.exception_dump_addr; + + if (!addr || len < sizeof(__le32) || len > SZ_4K) { + bt_dev_err(data->hdev, "Invalid exception address: 0x%8.8x or length: %u", + addr, len); + return; + } + + /* Ensure size is 4-byte aligned; btintel_pcie_read_device_mem() + * reads device memory in 4-byte units. + */ + len = ALIGN_DOWN(len, 4); + buf = kzalloc(len, GFP_KERNEL); if (!buf) goto exit_on_error; @@ -1576,6 +1865,20 @@ static void btintel_pcie_fwtrigger_worker(struct work_struct *work) clear_bit(BTINTEL_PCIE_FWTRIGGER_DUMP_INPROGRESS, &data->flags); } +static void btintel_pcie_mbox_worker(struct work_struct *work) +{ + struct btintel_pcie_data *data = container_of(work, + struct btintel_pcie_data, mbox_work); + + if (!data->hdev) + goto out; + + btintel_parse_mbox_tlv(data); +out: + /* Release guard last; matches set in gp1 handler. */ + clear_bit(BTINTEL_PCIE_MAIL_BOX_INTR, &data->flags); +} + static void btintel_pcie_rx_work(struct work_struct *work) { struct btintel_pcie_data *data = container_of(work, @@ -2380,6 +2683,7 @@ static int btintel_pcie_setup_internal(struct hci_dev *hdev) goto exit_error; } + data->dmp_hdr.cnvi_bt = ver_tlv.cnvi_bt; switch (INTEL_HW_PLATFORM(ver_tlv.cnvi_bt)) { case 0x37: break; @@ -2432,10 +2736,9 @@ static int btintel_pcie_setup_internal(struct hci_dev *hdev) data->dmp_hdr.fw_timestamp = ver_tlv.timestamp; data->dmp_hdr.fw_build_type = ver_tlv.build_type; data->dmp_hdr.fw_build_num = ver_tlv.build_num; - data->dmp_hdr.cnvi_bt = ver_tlv.cnvi_bt; if (ver_tlv.img_type == 0x02 || ver_tlv.img_type == 0x03) - data->dmp_hdr.fw_git_sha1 = ver_tlv.git_sha1; + data->dmp_hdr.fw_sha = ver_tlv.git_sha1; err = btintel_pcie_get_debug_info_addr(hdev); if (err) @@ -2715,6 +3018,7 @@ static void btintel_pcie_reset_work(struct work_struct *wk) disable_work_sync(&data->coredump_work); disable_work_sync(&data->hwexp_work); disable_work_sync(&data->fwtrigger_work); + disable_work_sync(&data->mbox_work); bt_dev_dbg(data->hdev, "Release bluetooth interface"); @@ -2739,6 +3043,7 @@ static void btintel_pcie_reset_work(struct work_struct *wk) enable_work(&data->coredump_work); enable_work(&data->hwexp_work); enable_work(&data->fwtrigger_work); + enable_work(&data->mbox_work); } out: @@ -3012,6 +3317,7 @@ static int btintel_pcie_probe(struct pci_dev *pdev, INIT_WORK(&data->coredump_work, btintel_pcie_coredump_worker); INIT_WORK(&data->hwexp_work, btintel_pcie_hwexp_worker); INIT_WORK(&data->fwtrigger_work, btintel_pcie_fwtrigger_worker); + INIT_WORK(&data->mbox_work, btintel_pcie_mbox_worker); data->boot_stage_cache = 0x00; data->img_resp_cache = 0x00; @@ -3081,6 +3387,7 @@ static void btintel_pcie_remove(struct pci_dev *pdev) disable_work_sync(&data->coredump_work); disable_work_sync(&data->hwexp_work); disable_work_sync(&data->fwtrigger_work); + disable_work_sync(&data->mbox_work); /* Cancel pending reset work. Skip only when remove() is called from * within the reset work itself (PLDR device_reprobe path) to avoid diff --git a/drivers/bluetooth/btintel_pcie.h b/drivers/bluetooth/btintel_pcie.h index 749369b24031ce..5ceb2ba1276fd8 100644 --- a/drivers/bluetooth/btintel_pcie.h +++ b/drivers/bluetooth/btintel_pcie.h @@ -18,6 +18,7 @@ #define BTINTEL_PCIE_CSR_CI_ADDR_LSB_REG (BTINTEL_PCIE_CSR_BASE + 0x118) #define BTINTEL_PCIE_CSR_CI_ADDR_MSB_REG (BTINTEL_PCIE_CSR_BASE + 0x11C) #define BTINTEL_PCIE_CSR_IMG_RESPONSE_REG (BTINTEL_PCIE_CSR_BASE + 0x12C) +#define BTINTEL_PCIE_CSR_IPC_DOORBELL_VEC_REG (BTINTEL_PCIE_CSR_BASE + 0x130) #define BTINTEL_PCIE_CSR_MBOX_1_REG (BTINTEL_PCIE_CSR_BASE + 0x170) #define BTINTEL_PCIE_CSR_MBOX_2_REG (BTINTEL_PCIE_CSR_BASE + 0x174) #define BTINTEL_PCIE_CSR_MBOX_3_REG (BTINTEL_PCIE_CSR_BASE + 0x178) @@ -52,6 +53,8 @@ #define BTINTEL_PCIE_CSR_BOOT_STAGE_ALIVE (BIT(23)) #define BTINTEL_PCIE_CSR_BOOT_STAGE_D3_STATE_READY (BIT(24)) +#define BTINTEL_PCIE_CSR_DOORBELL_MBOX_READ_CONFIRM (BIT(4)) + /* Registers for MSI-X */ #define BTINTEL_PCIE_CSR_MSIX_BASE (0x2000) #define BTINTEL_PCIE_CSR_MSIX_FH_INT_CAUSES (BTINTEL_PCIE_CSR_MSIX_BASE + 0x0800) @@ -121,7 +124,8 @@ enum { BTINTEL_PCIE_COREDUMP_INPROGRESS, BTINTEL_PCIE_FWTRIGGER_DUMP_INPROGRESS, BTINTEL_PCIE_RECOVERY_IN_PROGRESS, - BTINTEL_PCIE_SETUP_DONE + BTINTEL_PCIE_SETUP_DONE, + BTINTEL_PCIE_MAIL_BOX_INTR }; enum btintel_pcie_tlv_type { @@ -153,6 +157,14 @@ enum btintel_pcie_reset_type { BTINTEL_PCIE_IOSF_PRR_PLDR = 1, }; +enum btintel_pcie_mbox_msg { + BTINTEL_PCIE_NO_USE = 0, + BTINTEL_PCIE_TOP_SILENT_RESET, + BTINTEL_PCIE_SB_AUDIO_DEVICE_REPORT, + BTINTEL_PCIE_BUILD_SPECIFIC_RESOURCES_MAPPING, + BTINTEL_PCIE_LAST_MESSAGE = 4095 +}; + #define BTINTEL_PCIE_MSIX_NON_AUTO_CLEAR_CAUSE BIT(7) /* Minimum and Maximum number of MSI-X Vector @@ -198,6 +210,12 @@ enum { /* RBD buffer size mapping */ #define BTINTEL_PCIE_RBD_SIZE_4K 0x04 +#define BTINTEL_PCIE_TLV_TYPE_EXCEPTION_DUMP_ADDRESS 0x04 +#define BTINTEL_PCIE_TLV_TYPE_DCCM_MEM_ADDRESS 0x05 +#define BTINTEL_PCIE_TLV_TYPE_SDS_MEM_ADDRESS 0x06 +#define BTINTEL_PCIE_TLV_TYPE_ECL_MEM_ADDRESS 0x07 +#define BTINTEL_PCIE_TLV_TYPE_SMEM_ADDRESS 0x08 + /* * Struct for Context Information (v2) * @@ -424,6 +442,32 @@ struct btintel_pcie_dbgc { struct data_buf *bufs; }; +struct btintel_pcie_dump_mem_info { + u32 exception_dump_addr; + u32 exception_dump_len; + u32 dccm_addr_start; + u32 dccm_addr_end; + u32 sds_fixed_rom_addr_start; + u32 sds_fixed_rom_addr_end; + u32 sds_start_addr_start; + u32 sds_start_addr_end; + u32 sds_iosf_data_addr_start; + u32 sds_iosf_data_addr_end; + u32 ecl_addr_start; + u32 ecl_addr_end; + u32 smem_addr_start; + u32 smem_addr_end; +}; + +struct btintel_pcie_mbox { + u32 mbox_flags; + u32 mbox_status; + u32 mbox1; + u32 mbox2; + u32 mbox3; + u32 mbox4; +}; + struct btintel_pcie_dump_header { const char *driver_name; u32 cnvi_top; @@ -431,7 +475,7 @@ struct btintel_pcie_dump_header { u16 fw_timestamp; u8 fw_build_type; u32 fw_build_num; - u32 fw_git_sha1; + u32 fw_sha; u32 cnvi_bt; u32 write_ptr; u32 wrap_ctr; @@ -522,6 +566,7 @@ struct btintel_pcie_data { struct work_struct coredump_work; struct work_struct hwexp_work; struct work_struct fwtrigger_work; + struct work_struct mbox_work; struct dma_pool *dma_pool; dma_addr_t dma_p_addr; @@ -539,6 +584,10 @@ struct btintel_pcie_data { u8 pm_sx_event; u32 debug_evt_addr; u32 debug_evt_size; + dma_addr_t debug_table_addr; + u32 debug_table_size; + struct btintel_pcie_dump_mem_info dump_info; + struct btintel_pcie_mbox mbox; }; static inline u32 btintel_pcie_rd_reg32(struct btintel_pcie_data *data, From 185b9b3869fa9eaae23a95e682ddecdaeb5a2988 Mon Sep 17 00:00:00 2001 From: Kiran K Date: Wed, 19 Aug 2026 20:02:18 +0530 Subject: [PATCH 163/857] Bluetooth: btintel_pcie: sync mbox tlv parsing with GP0 alive interrupt Performing a target access to read the mbox TLV table while the driver is concurrently posting RX buffers to the firmware causes the hardware to return 0 for the target address, resulting in an invalid/empty TLV parse. Add a synchronization handshake between the mbox TLV read operation performed by the mbox worker and the GP0 (alive) MSI-X interrupt (which signals completion of RX buffer posting). The worker now waits for the alive interrupt before initiating the target access, ensuring the hardware returns valid data. Assisted-by: Gemini:gemini-3.1-pro-preview Signed-off-by: Kiran K Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btintel_pcie.c | 42 +++++++++++++++++++++++++++++--- drivers/bluetooth/btintel_pcie.h | 10 +++++++- 2 files changed, 48 insertions(+), 4 deletions(-) diff --git a/drivers/bluetooth/btintel_pcie.c b/drivers/bluetooth/btintel_pcie.c index 4d1b6e9a2bdc7b..bbce41b5b3760b 100644 --- a/drivers/bluetooth/btintel_pcie.c +++ b/drivers/bluetooth/btintel_pcie.c @@ -987,6 +987,24 @@ static int btintel_parse_mbox_tlv(struct btintel_pcie_data *data) struct mbox_tlv *tlv; struct btintel_data *cnvi_data = hci_get_priv(data->hdev); u8 hw_variant = INTEL_HW_VARIANT(cnvi_data->cnvi_bt); + long t; + + /* Wait for GP0 alive interrupt to post RX buffers */ + t = wait_event_timeout(data->mbox_parse_wait_q, + test_bit(BTINTEL_PCIE_MBOX_PARSE_READY, &data->flags), + msecs_to_jiffies(BTINTEL_PCIE_MBOX_INTR_TIMEOUT_MS)); + if (!t) { + bt_dev_warn(data->hdev, + "Timeout (%u ms) waiting for alive interrupt before mbox TLV parse; skipping", + BTINTEL_PCIE_MBOX_INTR_TIMEOUT_MS); + return 0; + } + clear_bit(BTINTEL_PCIE_MBOX_PARSE_READY, &data->flags); + + bt_dev_info(data->hdev, + "mbox TLV parse started at %lld ns (%lld us after mbox interrupt)", + ktime_to_ns(ktime_get()), + ktime_to_us(ktime_sub(ktime_get(), data->mbox_intr_ts))); memset(&data->dump_info, 0, sizeof(data->dump_info)); @@ -1230,12 +1248,22 @@ static void btintel_pcie_msix_gp1_handler(struct btintel_pcie_data *data) if (target_access && !test_and_set_bit(BTINTEL_PCIE_MAIL_BOX_INTR, &data->flags)) { + /* Arm the mbox<->alive handshake */ + clear_bit(BTINTEL_PCIE_MBOX_PARSE_READY, &data->flags); + set_bit(BTINTEL_PCIE_MBOX_PARSE_PENDING, &data->flags); + data->mbox_intr_ts = ktime_get(); + + bt_dev_info(data->hdev, + "mbox interrupt received at %lld ns; queuing mbox_work", + ktime_to_ns(data->mbox_intr_ts)); + WRITE_ONCE(data->debug_table_addr, addr); WRITE_ONCE(data->debug_table_size, size); if (!queue_work(data->dump_workqueue, - &data->mbox_work)) - clear_bit(BTINTEL_PCIE_MAIL_BOX_INTR, - &data->flags); + &data->mbox_work)) { + clear_bit(BTINTEL_PCIE_MAIL_BOX_INTR, &data->flags); + clear_bit(BTINTEL_PCIE_MBOX_PARSE_PENDING, &data->flags); + } } /* Mailbox is read, ack to FW */ @@ -1345,6 +1373,12 @@ static void btintel_pcie_msix_gp0_handler(struct btintel_pcie_data *data) if (submit_rx) { btintel_pcie_reset_ia(data); btintel_pcie_start_rx(data); + + /* Complete the mbox<->alive handshake */ + if (test_and_clear_bit(BTINTEL_PCIE_MBOX_PARSE_PENDING, &data->flags)) { + set_bit(BTINTEL_PCIE_MBOX_PARSE_READY, &data->flags); + wake_up(&data->mbox_parse_wait_q); + } } if (signal_waitq) { @@ -3301,6 +3335,8 @@ static int btintel_pcie_probe(struct pci_dev *pdev, init_waitqueue_head(&data->tx_wait_q); data->tx_wait_done = false; + init_waitqueue_head(&data->mbox_parse_wait_q); + data->workqueue = alloc_ordered_workqueue(KBUILD_MODNAME, WQ_HIGHPRI); if (!data->workqueue) return -ENOMEM; diff --git a/drivers/bluetooth/btintel_pcie.h b/drivers/bluetooth/btintel_pcie.h index 5ceb2ba1276fd8..5aff1dfa888fa0 100644 --- a/drivers/bluetooth/btintel_pcie.h +++ b/drivers/bluetooth/btintel_pcie.h @@ -125,7 +125,9 @@ enum { BTINTEL_PCIE_FWTRIGGER_DUMP_INPROGRESS, BTINTEL_PCIE_RECOVERY_IN_PROGRESS, BTINTEL_PCIE_SETUP_DONE, - BTINTEL_PCIE_MAIL_BOX_INTR + BTINTEL_PCIE_MAIL_BOX_INTR, + BTINTEL_PCIE_MBOX_PARSE_PENDING, + BTINTEL_PCIE_MBOX_PARSE_READY }; enum btintel_pcie_tlv_type { @@ -178,6 +180,7 @@ enum btintel_pcie_mbox_msg { /* Default interrupt timeout in msec */ #define BTINTEL_DEFAULT_INTR_TIMEOUT_MS 3000 +#define BTINTEL_PCIE_MBOX_INTR_TIMEOUT_MS 500 #define BTINTEL_PCIE_DX_TRANSITION_MAX_RETRIES 3 @@ -588,6 +591,11 @@ struct btintel_pcie_data { u32 debug_table_size; struct btintel_pcie_dump_mem_info dump_info; struct btintel_pcie_mbox mbox; + + /* Wait queue for mbox_worker to wait for GP0 alive interrupt */ + wait_queue_head_t mbox_parse_wait_q; + /* Timestamp captured in GP1 handler when mbox interrupt is received */ + ktime_t mbox_intr_ts; }; static inline u32 btintel_pcie_rd_reg32(struct btintel_pcie_data *data, From 52eff0428bf2729c65572b3c990611dc7b7b1f4e Mon Sep 17 00:00:00 2001 From: Catherine L Date: Wed, 19 Aug 2026 20:02:19 +0530 Subject: [PATCH 164/857] Bluetooth: btintel_pcie: Route debug traces to WiFi DBGC by default Set dbg_output_mode to 0x06 (BTINTEL_PCIE_WIFI_DBGC) by default so firmware debug traces are forwarded to the WiFi DBGC. In this mode: - Host DBGC fragment/data buffers are NOT allocated. - Context info publishes dbgc_addr/size as 0. Add a small helper btintel_pcie_dbg_to_wifi() driven by a cached dbg_path_cache field in struct btintel_pcie_data, initialized to BTINTEL_PCIE_WIFI_DBGC in probe. Signed-off-by: Kiran K Signed-off-by: Catherine L Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btintel_pcie.c | 43 +++++++++++++++++++++++++++++--- drivers/bluetooth/btintel_pcie.h | 16 ++++++++++++ 2 files changed, 55 insertions(+), 4 deletions(-) diff --git a/drivers/bluetooth/btintel_pcie.c b/drivers/bluetooth/btintel_pcie.c index bbce41b5b3760b..baa621b3fef931 100644 --- a/drivers/bluetooth/btintel_pcie.c +++ b/drivers/bluetooth/btintel_pcie.c @@ -180,6 +180,15 @@ static inline char *btintel_pcie_alivectxt_state2str(u32 alive_intr_ctxt) } } +/* Returns true when firmware traces are routed to the WiFi DBGC. In that + * mode the host must not allocate DBGC buffers and must not publish their + * addresses in the context info. + */ +static inline bool btintel_pcie_dbg_to_wifi(struct btintel_pcie_data *data) +{ + return data->dbg_path_cache != BTINTEL_PCIE_DRAM; +} + /* This function initializes the memory for DBGC buffers and formats the * DBGC fragment which consists header info and DBGC buffer's LSB, MSB and * size as the payload @@ -1853,6 +1862,15 @@ static void btintel_pcie_coredump_worker(struct work_struct *work) if (!data->hdev) goto out; + /* When firmware routes debug traces to the WiFi DBGC, no host + * DBGC buffers were allocated, so there is nothing to dump here. + */ + if (btintel_pcie_dbg_to_wifi(data)) { + bt_dev_info(data->hdev, + "Skipping coredump: debug traces routed to WiFi DBGC"); + goto out; + } + btintel_pcie_dump_traces(data->hdev); out: /* Release guard last so a new trigger can run only after this @@ -2223,9 +2241,18 @@ static void btintel_pcie_init_ci(struct btintel_pcie_data *data, ci->num_urbdq1 = data->rxq.count; ci->urbdq_db_vec = BTINTEL_PCIE_RXQ_NUM; - ci->dbg_output_mode = 0x01; - ci->dbgc_addr = data->dbgc.frag_p_addr; - ci->dbgc_size = data->dbgc.frag_size; + ci->dbg_output_mode = btintel_pcie_dbg_to_wifi(data) ? + BTINTEL_PCIE_WIFI_DBGC : BTINTEL_PCIE_DRAM; + if (btintel_pcie_dbg_to_wifi(data)) { + /* Firmware forwards debug traces to the WiFi DBGC, so no + * host DBGC buffer is needed; leave dbgc_addr/size as 0. + */ + ci->dbgc_addr = 0; + ci->dbgc_size = 0; + } else { + ci->dbgc_addr = data->dbgc.frag_p_addr; + ci->dbgc_size = data->dbgc.frag_size; + } ci->dbg_preset = 0x00; } @@ -2453,7 +2480,14 @@ static int btintel_pcie_alloc(struct btintel_pcie_data *data) v_addr += ci_size; /* Setup data buffers for dbgc */ - err = btintel_pcie_setup_dbgc(data); + if (btintel_pcie_dbg_to_wifi(data)) { + /* Firmware routes traces to the WiFi DBGC; skip host DBGC + * buffer allocation entirely. + */ + err = 0; + } else { + err = btintel_pcie_setup_dbgc(data); + } if (err) goto exit_error_txq; @@ -3357,6 +3391,7 @@ static int btintel_pcie_probe(struct pci_dev *pdev, data->boot_stage_cache = 0x00; data->img_resp_cache = 0x00; + data->dbg_path_cache = BTINTEL_PCIE_WIFI_DBGC; /* FLR can be invoked by echoing to debugfs path, so explicitly * initialized */ diff --git a/drivers/bluetooth/btintel_pcie.h b/drivers/bluetooth/btintel_pcie.h index 5aff1dfa888fa0..9baa214d9bbee9 100644 --- a/drivers/bluetooth/btintel_pcie.h +++ b/drivers/bluetooth/btintel_pcie.h @@ -94,6 +94,20 @@ /* Num of alloc Dbg buff (4) + (LSB(4), MSB(4), Size(4)) for each buffer */ #define BTINTEL_PCIE_DBGC_FRAG_PAYLOAD_SIZE 196 +/* dbg_output_mode values for the context info. + * BTINTEL_PCIE_DRAM: firmware writes traces to host DRAM DBGC buffers. + * BTINTEL_PCIE_WIFI_DBGC: firmware forwards traces to the WiFi DBGC; the + * host does NOT need to allocate DBGC fragment/data buffers and must + * publish dbgc_addr/size as 0 in the context info. + * + * Encoding of BTINTEL_PCIE_WIFI_DBGC (0x06): + * Bit[0] DBGC O/P : 0 = SRAM (don't care, DBGI selected) + * Bit[1] DBGC I/P : 1 = DBGI + * Bits[2:3] DBGI O/P : 01 = WiFi DBGC + */ +#define BTINTEL_PCIE_DRAM 0x01 +#define BTINTEL_PCIE_WIFI_DBGC 0x06 + /* Causes for the FH register interrupts */ enum msix_fh_int_causes { BTINTEL_PCIE_MSIX_FH_INT_CAUSES_0 = BIT(0), /* cause 0 */ @@ -503,6 +517,7 @@ struct btintel_pcie_dump_header { * @hw_init_mask: initial unmaksed hw causes * @boot_stage_cache: cached value of boot stage register * @img_resp_cache: cached value of image response register + * @dbg_path_cache: cached debug output routing mode (BT DRAM or WiFi DBGC) * @cnvi: CNVi register value * @cnvr: CNVr register value * @gp0_received: condition for gp0 interrupt @@ -550,6 +565,7 @@ struct btintel_pcie_data { u32 boot_stage_cache; u32 img_resp_cache; + u32 dbg_path_cache; u32 cnvi; u32 cnvr; From 486f8908aa587ab2a213bbef39311743e4f8f57a Mon Sep 17 00:00:00 2001 From: Hang Nan <2122295973@qq.com> Date: Wed, 19 Aug 2026 08:57:58 +0800 Subject: [PATCH 165/857] Bluetooth: ISO: fix use-after-free of listener socket in iso_conn_ready iso_conn_ready() looks up the BIS listener socket with iso_get_sock(), which takes a reference, and then, without re-checking its state, creates a child socket from it: parent = iso_get_sock(hdev, ...); if (!parent) return; lock_sock(parent); sk = iso_sock_alloc(sock_net(parent), NULL, BTPROTO_ISO, ...); ... iso_chan_add(conn, sk, parent); ... release_sock(parent); sock_put(parent); If the listener socket is closed concurrently, between iso_get_sock() and lock_sock(), the reference taken by iso_get_sock() may be the last one: the close path drops the link-list reference, and once iso_conn_ready() drops its own reference at the end of the function the socket is freed. The child socket, however, is already linked to the freed parent, and a later disconnect of the child runs iso_chan_del() -> bt_accept_unlink(), which dereferences the dangling parent pointer into the freed accept queue (a use-after-free). The same dangling pointer is also dereferenced through parent->***() in iso_chan_del(). Fix it the same way the connected (non-BIS) path was fixed in commit 0d255e63fcf3 ("Bluetooth: ISO: hold sk properly in iso_conn_ready"): after taking the socket lock, re-check that the parent is still a listening, alive socket, and bail out otherwise. Fixes: ccf74f2390d60 ("Bluetooth: Add BTPROTO_ISO socket type") Cc: stable@vger.kernel.org Signed-off-by: Hang Nan <2122295973@qq.com> Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/iso.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/net/bluetooth/iso.c b/net/bluetooth/iso.c index aa2ce78f56a2ce..75bfd5938b2ea7 100644 --- a/net/bluetooth/iso.c +++ b/net/bluetooth/iso.c @@ -2277,6 +2277,14 @@ static void iso_conn_ready(struct iso_conn *conn) lock_sock(parent); + /* The listener may have been closed concurrently. */ + if (parent->sk_state != BT_LISTEN || + sock_flag(parent, SOCK_ZAPPED)) { + release_sock(parent); + sock_put(parent); + return; + } + sk = iso_sock_alloc(sock_net(parent), NULL, BTPROTO_ISO, GFP_ATOMIC, 0); if (!sk) { From c5400637ac253fd645aeceec907ff69bf9cf03af Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Thu, 20 Aug 2026 13:40:45 +0200 Subject: [PATCH 166/857] gfs2: Avoid ip->i_gl dereferences in gfs2_inode_lookup and gfs2_create_inode Store the inode glock in a separate variable instead of repeatedly retrieving it as ip->i_gl. Signed-off-by: Andreas Gruenbacher --- fs/gfs2/inode.c | 35 +++++++++++++++++++---------------- 1 file changed, 19 insertions(+), 16 deletions(-) diff --git a/fs/gfs2/inode.c b/fs/gfs2/inode.c index f361876c558335..438519bc06d496 100644 --- a/fs/gfs2/inode.c +++ b/fs/gfs2/inode.c @@ -130,6 +130,7 @@ struct inode *gfs2_inode_lookup(struct super_block *sb, unsigned int type, { struct inode *inode; struct gfs2_inode *ip; + struct gfs2_glock *gl = NULL; struct gfs2_holder i_gh; int error; @@ -147,9 +148,10 @@ struct inode *gfs2_inode_lookup(struct super_block *sb, unsigned int type, gfs2_setup_inode(inode); error = gfs2_glock_get(sdp, no_addr, &gfs2_inode_glops, CREATE, - &ip->i_gl); + &gl); if (unlikely(error)) goto fail; + ip->i_gl = gl; error = gfs2_glock_get(sdp, no_addr, &gfs2_iopen_glops, CREATE, &io_gl); @@ -178,14 +180,14 @@ struct inode *gfs2_inode_lookup(struct super_block *sb, unsigned int type, * block. We read the inode when instantiating it * after possibly checking the block type. */ - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, + error = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, GL_SKIP, &i_gh); if (error) goto fail; error = -ESTALE; if (no_formal_ino && - gfs2_inode_already_deleted(ip->i_gl, no_formal_ino)) + gfs2_inode_already_deleted(gl, no_formal_ino)) goto fail; if (blktype != GFS2_BLKST_FREE) { @@ -196,20 +198,20 @@ struct inode *gfs2_inode_lookup(struct super_block *sb, unsigned int type, } } - set_bit(GLF_INSTANTIATE_NEEDED, &ip->i_gl->gl_flags); + set_bit(GLF_INSTANTIATE_NEEDED, &gl->gl_flags); /* Lowest possible timestamp; will be overwritten in gfs2_dinode_in. */ inode_set_atime(inode, 1LL << (8 * sizeof(inode_get_atime_sec(inode)) - 1), 0); - glock_set_object(ip->i_gl, ip); + glock_set_object(gl, ip); if (type == DT_UNKNOWN) { /* Inode glock must be locked already */ error = gfs2_instantiate(&i_gh); if (error) { - glock_clear_object(ip->i_gl, ip); + glock_clear_object(gl, ip); goto fail; } } else { @@ -240,8 +242,8 @@ struct inode *gfs2_inode_lookup(struct super_block *sb, unsigned int type, gfs2_glock_dq_uninit(&ip->i_iopen_gh); if (gfs2_holder_initialized(&i_gh)) gfs2_glock_dq_uninit(&i_gh); - if (ip->i_gl) { - gfs2_glock_put(ip->i_gl); + if (gl) { + gfs2_glock_put(gl); ip->i_gl = NULL; } iget_failed(inode); @@ -708,7 +710,7 @@ static int gfs2_create_inode(struct inode *dir, struct dentry *dentry, struct inode *inode = NULL; struct gfs2_inode *dip = GFS2_I(dir), *ip; struct gfs2_sbd *sdp = GFS2_SB(&dip->i_inode); - struct gfs2_glock *io_gl; + struct gfs2_glock *gl = NULL, *io_gl; int error, dealloc_error; u32 aflags = 0; unsigned blocks = 1; @@ -832,9 +834,10 @@ static int gfs2_create_inode(struct inode *dir, struct dentry *dentry, gfs2_set_inode_blocks(inode, blocks); - error = gfs2_glock_get(sdp, ip->i_no_addr, &gfs2_inode_glops, CREATE, &ip->i_gl); + error = gfs2_glock_get(sdp, ip->i_no_addr, &gfs2_inode_glops, CREATE, &gl); if (error) goto fail_dealloc_inode; + ip->i_gl = gl; error = gfs2_glock_get(sdp, ip->i_no_addr, &gfs2_iopen_glops, CREATE, &io_gl); if (error) @@ -854,10 +857,10 @@ static int gfs2_create_inode(struct inode *dir, struct dentry *dentry, if (error) goto fail_gunlock2; - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, GL_SKIP, &gh); + error = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, GL_SKIP, &gh); if (error) goto fail_gunlock3; - clear_bit(GLF_INSTANTIATE_NEEDED, &ip->i_gl->gl_flags); + clear_bit(GLF_INSTANTIATE_NEEDED, &gl->gl_flags); error = gfs2_trans_begin(sdp, blocks, 0); if (error) @@ -870,7 +873,7 @@ static int gfs2_create_inode(struct inode *dir, struct dentry *dentry, init_dinode(dip, ip, symname); gfs2_trans_end(sdp); - glock_set_object(ip->i_gl, ip); + glock_set_object(gl, ip); glock_set_object(io_gl, ip); gfs2_set_iop(inode); @@ -914,7 +917,7 @@ static int gfs2_create_inode(struct inode *dir, struct dentry *dentry, return error; fail_gunlock4: - glock_clear_object(ip->i_gl, ip); + glock_clear_object(gl, ip); glock_clear_object(io_gl, ip); fail_gunlock3: gfs2_glock_dq_uninit(&ip->i_iopen_gh); @@ -932,8 +935,8 @@ static int gfs2_create_inode(struct inode *dir, struct dentry *dentry, fs_warn(sdp, "%s: %d\n", __func__, dealloc_error); ip->i_no_addr = 0; fail_free_inode: - if (ip->i_gl) { - gfs2_glock_put(ip->i_gl); + if (gl) { + gfs2_glock_put(gl); ip->i_gl = NULL; } gfs2_rs_deltree(&ip->i_res); From 50db51ca836191ddf4ea1f010f52cb66b21c63bd Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Thu, 20 Aug 2026 13:45:28 +0200 Subject: [PATCH 167/857] gfs2: Get to super block from inode in tracepoints When we have a gfs2 inode, use that to get to the super block instead of going via the glock. Signed-off-by: Andreas Gruenbacher --- fs/gfs2/trace_gfs2.h | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/gfs2/trace_gfs2.h b/fs/gfs2/trace_gfs2.h index bc40320ef239b1..0c60d755ba7627 100644 --- a/fs/gfs2/trace_gfs2.h +++ b/fs/gfs2/trace_gfs2.h @@ -457,7 +457,7 @@ TRACE_EVENT(gfs2_bmap, ), TP_fast_assign( - __entry->dev = glock_sbd(ip->i_gl)->sd_vfs->s_dev; + __entry->dev = ip->i_inode.i_sb->s_dev; __entry->lblock = lblock; __entry->pblock = buffer_mapped(bh) ? bh->b_blocknr : 0; __entry->inum = ip->i_no_addr; @@ -493,7 +493,7 @@ TRACE_EVENT(gfs2_iomap_start, ), TP_fast_assign( - __entry->dev = glock_sbd(ip->i_gl)->sd_vfs->s_dev; + __entry->dev = ip->i_inode.i_sb->s_dev; __entry->inum = ip->i_no_addr; __entry->pos = pos; __entry->length = length; @@ -525,7 +525,7 @@ TRACE_EVENT(gfs2_iomap_end, ), TP_fast_assign( - __entry->dev = glock_sbd(ip->i_gl)->sd_vfs->s_dev; + __entry->dev = ip->i_inode.i_sb->s_dev; __entry->inum = ip->i_no_addr; __entry->offset = iomap->offset; __entry->length = iomap->length; From 999c389377cc6712996d6310081d4600ff11b58f Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Thu, 20 Aug 2026 14:47:05 +0200 Subject: [PATCH 168/857] gfs2: Add inode variables Add inode variables in a few places to get rid of '&ip->i_inode'. This cleans up the code a little and prepares for the next step of adding an accessor function for ip->i_gl. Signed-off-by: Andreas Gruenbacher --- fs/gfs2/bmap.c | 37 ++++++++++++++++++++----------------- fs/gfs2/dir.c | 4 +++- fs/gfs2/file.c | 5 +++-- fs/gfs2/quota.c | 25 ++++++++++++++----------- fs/gfs2/recovery.c | 5 +++-- 5 files changed, 43 insertions(+), 33 deletions(-) diff --git a/fs/gfs2/bmap.c b/fs/gfs2/bmap.c index 73c62697116390..636139f463b062 100644 --- a/fs/gfs2/bmap.c +++ b/fs/gfs2/bmap.c @@ -89,6 +89,7 @@ static int gfs2_unstuffer_folio(struct gfs2_inode *ip, struct buffer_head *dibh, static int __gfs2_unstuff_inode(struct gfs2_inode *ip, struct folio *folio) { + struct inode *inode = &ip->i_inode; struct buffer_head *bh, *dibh; struct gfs2_dinode *di; u64 block = 0; @@ -99,7 +100,7 @@ static int __gfs2_unstuff_inode(struct gfs2_inode *ip, struct folio *folio) if (error) return error; - if (i_size_read(&ip->i_inode)) { + if (i_size_read(inode)) { /* Get a free block, fill it with the stuffed data, and write it out to disk */ @@ -108,7 +109,7 @@ static int __gfs2_unstuff_inode(struct gfs2_inode *ip, struct folio *folio) if (error) goto out_brelse; if (isdir) { - gfs2_trans_remove_revoke(GFS2_SB(&ip->i_inode), block, 1); + gfs2_trans_remove_revoke(GFS2_SB(inode), block, 1); error = gfs2_dir_get_new_buffer(ip, block, &bh); if (error) goto out_brelse; @@ -128,10 +129,10 @@ static int __gfs2_unstuff_inode(struct gfs2_inode *ip, struct folio *folio) di = (struct gfs2_dinode *)dibh->b_data; gfs2_buffer_clear_tail(dibh, sizeof(struct gfs2_dinode)); - if (i_size_read(&ip->i_inode)) { + if (i_size_read(inode)) { *(__be64 *)(di + 1) = cpu_to_be64(block); - gfs2_add_inode_blocks(&ip->i_inode, 1); - di->di_blocks = cpu_to_be64(gfs2_get_inode_blocks(&ip->i_inode)); + gfs2_add_inode_blocks(inode, 1); + di->di_blocks = cpu_to_be64(gfs2_get_inode_blocks(inode)); } ip->i_height = 1; @@ -1488,7 +1489,8 @@ static int sweep_bh_for_rgrps(struct gfs2_inode *ip, struct gfs2_holder *rd_gh, struct buffer_head *bh, __be64 *start, __be64 *end, bool meta, u32 *btotal) { - struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); + struct inode *inode = &ip->i_inode; + struct gfs2_sbd *sdp = GFS2_SB(inode); struct gfs2_rgrpd *rgd; struct gfs2_trans *tr; __be64 *p; @@ -1546,7 +1548,7 @@ static int sweep_bh_for_rgrps(struct gfs2_inode *ip, struct gfs2_holder *rd_gh, jblocks_rqsted = rgd->rd_length + RES_DINODE + RES_INDIRECT; - isize_blks = gfs2_get_inode_blocks(&ip->i_inode); + isize_blks = gfs2_get_inode_blocks(inode); if (isize_blks > atomic_read(&sdp->sd_log_thresh2)) jblocks_rqsted += atomic_read(&sdp->sd_log_thresh2); @@ -1597,7 +1599,7 @@ static int sweep_bh_for_rgrps(struct gfs2_inode *ip, struct gfs2_holder *rd_gh, if (bstart) { __gfs2_free_blocks(ip, rgd, bstart, (u32)blen, meta); (*btotal) += blen; - gfs2_add_inode_blocks(&ip->i_inode, -blen); + gfs2_add_inode_blocks(inode, -blen); } bstart = bn; blen = 1; @@ -1605,7 +1607,7 @@ static int sweep_bh_for_rgrps(struct gfs2_inode *ip, struct gfs2_holder *rd_gh, if (bstart) { __gfs2_free_blocks(ip, rgd, bstart, (u32)blen, meta); (*btotal) += blen; - gfs2_add_inode_blocks(&ip->i_inode, -blen); + gfs2_add_inode_blocks(inode, -blen); } out_unlock: if (!ret && blks_outside_rgrp) { /* If buffer still has non-zero blocks @@ -1620,7 +1622,7 @@ static int sweep_bh_for_rgrps(struct gfs2_inode *ip, struct gfs2_holder *rd_gh, /* Every transaction boundary, we rewrite the dinode to keep its di_blocks current in case of failure. */ - inode_set_mtime_to_ts(&ip->i_inode, inode_set_ctime_current(&ip->i_inode)); + inode_set_mtime_to_ts(inode, inode_set_ctime_current(inode)); gfs2_trans_add_meta(ip->i_gl, dibh); gfs2_dinode_out(ip, dibh->b_data); brelse(dibh); @@ -1746,7 +1748,8 @@ static inline bool walk_done(struct gfs2_sbd *sdp, */ static int punch_hole(struct gfs2_inode *ip, u64 offset, u64 length) { - struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); + struct inode *inode = &ip->i_inode; + struct gfs2_sbd *sdp = GFS2_SB(inode); u64 maxsize = sdp->sd_heightsize[ip->i_height]; struct metapath mp = {}; struct buffer_head *dibh, *bh; @@ -1985,9 +1988,8 @@ static int punch_hole(struct gfs2_inode *ip, u64 offset, u64 length) down_write(&ip->i_rw_mutex); } gfs2_statfs_change(sdp, 0, +btotal, 0); - gfs2_quota_change(ip, -(s64)btotal, ip->i_inode.i_uid, - ip->i_inode.i_gid); - inode_set_mtime_to_ts(&ip->i_inode, inode_set_ctime_current(&ip->i_inode)); + gfs2_quota_change(ip, -(s64)btotal, inode->i_uid, inode->i_gid); + inode_set_mtime_to_ts(inode, inode_set_ctime_current(inode)); gfs2_trans_add_meta(ip->i_gl, dibh); gfs2_dinode_out(ip, dibh->b_data); up_write(&ip->i_rw_mutex); @@ -2010,7 +2012,8 @@ static int punch_hole(struct gfs2_inode *ip, u64 offset, u64 length) static int trunc_end(struct gfs2_inode *ip) { - struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); + struct inode *inode = &ip->i_inode; + struct gfs2_sbd *sdp = GFS2_SB(inode); struct buffer_head *dibh; int error; @@ -2024,13 +2027,13 @@ static int trunc_end(struct gfs2_inode *ip) if (error) goto out; - if (!i_size_read(&ip->i_inode)) { + if (!i_size_read(inode)) { ip->i_height = 0; ip->i_goal = ip->i_no_addr; gfs2_buffer_clear_tail(dibh, sizeof(struct gfs2_dinode)); gfs2_ordered_del_inode(ip); } - inode_set_mtime_to_ts(&ip->i_inode, inode_set_ctime_current(&ip->i_inode)); + inode_set_mtime_to_ts(inode, inode_set_ctime_current(inode)); ip->i_diskflags &= ~GFS2_DIF_TRUNC_IN_PROG; gfs2_trans_add_meta(ip->i_gl, dibh); diff --git a/fs/gfs2/dir.c b/fs/gfs2/dir.c index 0237b36b9eb165..f6111276ebb072 100644 --- a/fs/gfs2/dir.c +++ b/fs/gfs2/dir.c @@ -759,10 +759,12 @@ static struct gfs2_dirent *gfs2_dirent_split_alloc(struct inode *inode, static int get_leaf(struct gfs2_inode *dip, u64 leaf_no, struct buffer_head **bhp) { + struct inode *inode = &dip->i_inode; + struct gfs2_sbd *sdp = GFS2_SB(inode); int error; error = gfs2_meta_read(dip->i_gl, leaf_no, DIO_WAIT, 0, bhp); - if (!error && gfs2_metatype_check(GFS2_SB(&dip->i_inode), *bhp, GFS2_METATYPE_LF)) { + if (!error && gfs2_metatype_check(sdp, *bhp, GFS2_METATYPE_LF)) { /* pr_info("block num=%llu\n", leaf_no); */ error = -EIO; } diff --git a/fs/gfs2/file.c b/fs/gfs2/file.c index b8c10de113ba7a..164e160e9064a9 100644 --- a/fs/gfs2/file.c +++ b/fs/gfs2/file.c @@ -812,7 +812,8 @@ static ssize_t gfs2_file_direct_read(struct kiocb *iocb, struct iov_iter *to, struct gfs2_holder *gh) { struct file *file = iocb->ki_filp; - struct gfs2_inode *ip = GFS2_I(file->f_mapping->host); + struct inode *inode = file->f_mapping->host; + struct gfs2_inode *ip = GFS2_I(inode); size_t prev_count = 0, window_size = 0; size_t read = 0; ssize_t ret; @@ -906,7 +907,7 @@ static ssize_t gfs2_file_direct_write(struct kiocb *iocb, struct iov_iter *from, if (ret) goto out_uninit; /* Silently fall back to buffered I/O when writing beyond EOF */ - if (iocb->ki_pos + iov_iter_count(from) > i_size_read(&ip->i_inode)) + if (iocb->ki_pos + iov_iter_count(from) > i_size_read(inode)) goto out_unlock; from->nofault = true; diff --git a/fs/gfs2/quota.c b/fs/gfs2/quota.c index 001c8b39ca5573..0cb2bb0aec7b23 100644 --- a/fs/gfs2/quota.c +++ b/fs/gfs2/quota.c @@ -740,8 +740,8 @@ static void do_qc(struct gfs2_quota_data *qd, s64 change) static int gfs2_write_buf_to_page(struct gfs2_sbd *sdp, unsigned long index, unsigned off, void *buf, unsigned bytes) { - struct gfs2_inode *ip = GFS2_I(sdp->sd_quota_inode); - struct inode *inode = &ip->i_inode; + struct inode *inode = sdp->sd_quota_inode; + struct gfs2_inode *ip = GFS2_I(inode); struct address_space *mapping = inode->i_mapping; struct folio *folio; struct buffer_head *bh; @@ -908,7 +908,8 @@ static int do_sync(unsigned int num_qd, struct gfs2_quota_data **qda, u64 sync_gen) { struct gfs2_sbd *sdp = (*qda)->qd_sbd; - struct gfs2_inode *ip = GFS2_I(sdp->sd_quota_inode); + struct inode *inode = sdp->sd_quota_inode; + struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_alloc_parms ap = {}; unsigned int data_blocks, ind_blocks; struct gfs2_holder *ghs, i_gh; @@ -927,7 +928,7 @@ static int do_sync(unsigned int num_qd, struct gfs2_quota_data **qda, return -ENOMEM; sort(qda, num_qd, sizeof(struct gfs2_quota_data *), sort_qd, NULL); - inode_lock(&ip->i_inode); + inode_lock(inode); for (qx = 0; qx < num_qd; qx++) { error = gfs2_glock_nq_init(qda[qx]->qd_gl, LM_ST_EXCLUSIVE, GL_NOCACHE, &ghs[qx]); @@ -991,9 +992,9 @@ static int do_sync(unsigned int num_qd, struct gfs2_quota_data **qda, out_dq: while (qx--) gfs2_glock_dq_uninit(&ghs[qx]); - inode_unlock(&ip->i_inode); + inode_unlock(inode); kfree(ghs); - gfs2_log_flush(glock_sbd(ip->i_gl), ip->i_gl, + gfs2_log_flush(sdp, ip->i_gl, GFS2_LOG_HEAD_FLUSH_NORMAL | GFS2_LFC_DO_SYNC); if (!error) { for (x = 0; x < num_qd; x++) { @@ -1402,7 +1403,8 @@ int gfs2_quota_refresh(struct gfs2_sbd *sdp, struct kqid qid) int gfs2_quota_init(struct gfs2_sbd *sdp) { - struct gfs2_inode *ip = GFS2_I(sdp->sd_qc_inode); + struct inode *inode = sdp->sd_qc_inode; + struct gfs2_inode *ip = GFS2_I(inode); u64 size = i_size_read(sdp->sd_qc_inode); unsigned int blocks = size >> sdp->sd_sb.sb_bsize_shift; unsigned int x, slot = 0; @@ -1434,7 +1436,7 @@ int gfs2_quota_init(struct gfs2_sbd *sdp) if (!extlen) { extlen = 32; - error = gfs2_get_extent(&ip->i_inode, x, &dblock, &extlen); + error = gfs2_get_extent(inode, x, &dblock, &extlen); if (error) goto fail; } @@ -1714,7 +1716,8 @@ static int gfs2_set_dqblk(struct super_block *sb, struct kqid qid, struct qc_dqblk *fdq) { struct gfs2_sbd *sdp = sb->s_fs_info; - struct gfs2_inode *ip = GFS2_I(sdp->sd_quota_inode); + struct inode *inode = sdp->sd_quota_inode; + struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_quota_data *qd; struct gfs2_holder q_gh, i_gh; unsigned int data_blocks, ind_blocks; @@ -1741,7 +1744,7 @@ static int gfs2_set_dqblk(struct super_block *sb, struct kqid qid, if (error) goto out_put; - inode_lock(&ip->i_inode); + inode_lock(inode); error = gfs2_glock_nq_init(qd->qd_gl, LM_ST_EXCLUSIVE, 0, &q_gh); if (error) goto out_unlockput; @@ -1807,7 +1810,7 @@ static int gfs2_set_dqblk(struct super_block *sb, struct kqid qid, gfs2_glock_dq_uninit(&q_gh); out_unlockput: gfs2_qa_put(ip); - inode_unlock(&ip->i_inode); + inode_unlock(inode); out_put: qd_put(qd); return error; diff --git a/fs/gfs2/recovery.c b/fs/gfs2/recovery.c index 616c46aa3434ae..b45aa3032ca2a9 100644 --- a/fs/gfs2/recovery.c +++ b/fs/gfs2/recovery.c @@ -32,14 +32,15 @@ struct workqueue_struct *gfs2_recovery_wq; int gfs2_replay_read_block(struct gfs2_jdesc *jd, unsigned int blk, struct buffer_head **bh) { - struct gfs2_inode *ip = GFS2_I(jd->jd_inode); + struct inode *inode = jd->jd_inode; + struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_glock *gl = ip->i_gl; u64 dblock; u32 extlen; int error; extlen = 32; - error = gfs2_get_extent(&ip->i_inode, blk, &dblock, &extlen); + error = gfs2_get_extent(inode, blk, &dblock, &extlen); if (error) return error; if (!dblock) { From bea09a05ee08d86c18a941e69fc6bfab6a918f87 Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Thu, 20 Aug 2026 12:55:19 +0200 Subject: [PATCH 169/857] gfs2: Introduce gfs2_inode_glock Introduce gfs2_inode_glock(inode) for getting from an inode to its inode glock. This obsoletes the local variable 'ip' in several places. The remaining direct accesses to ip->i_gl are for inode creation and distruction, and will be cleaned up in the next patch. Signed-off-by: Andreas Gruenbacher --- fs/gfs2/acl.c | 11 +++--- fs/gfs2/aops.c | 17 ++++----- fs/gfs2/bmap.c | 48 +++++++++++++++---------- fs/gfs2/dentry.c | 6 ++-- fs/gfs2/dir.c | 57 ++++++++++++++++++------------ fs/gfs2/export.c | 7 ++-- fs/gfs2/file.c | 69 ++++++++++++++++++++---------------- fs/gfs2/glock.c | 4 ++- fs/gfs2/glops.c | 3 +- fs/gfs2/incore.h | 5 ++- fs/gfs2/inode.c | 83 ++++++++++++++++++++++++-------------------- fs/gfs2/lops.c | 18 +++++----- fs/gfs2/meta_io.c | 7 ++-- fs/gfs2/ops_fstype.c | 25 +++++++------ fs/gfs2/quota.c | 29 ++++++++-------- fs/gfs2/recovery.c | 12 +++---- fs/gfs2/rgrp.c | 4 +-- fs/gfs2/super.c | 70 +++++++++++++++++++------------------ fs/gfs2/util.c | 8 ++--- fs/gfs2/xattr.c | 69 +++++++++++++++++++++--------------- 20 files changed, 307 insertions(+), 245 deletions(-) diff --git a/fs/gfs2/acl.c b/fs/gfs2/acl.c index a5b60778b91c90..49e489fe27ef59 100644 --- a/fs/gfs2/acl.c +++ b/fs/gfs2/acl.c @@ -59,7 +59,7 @@ static struct posix_acl *__gfs2_get_acl(struct inode *inode, int type) struct posix_acl *gfs2_get_acl(struct inode *inode, int type, bool rcu) { - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_holder gh; bool need_unlock = false; struct posix_acl *acl; @@ -67,8 +67,8 @@ struct posix_acl *gfs2_get_acl(struct inode *inode, int type, bool rcu) if (rcu) return ERR_PTR(-ECHILD); - if (!gfs2_glock_is_locked_by_me(ip->i_gl)) { - int ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, + if (!gfs2_glock_is_locked_by_me(gl)) { + int ret = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_ANY, &gh); if (ret) return ERR_PTR(ret); @@ -106,6 +106,7 @@ int gfs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct inode *inode = d_inode(dentry); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder gh; bool need_unlock = false; @@ -119,8 +120,8 @@ int gfs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, if (ret) return ret; - if (!gfs2_glock_is_locked_by_me(ip->i_gl)) { - ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh); + if (!gfs2_glock_is_locked_by_me(gl)) { + ret = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, 0, &gh); if (ret) goto out; need_unlock = true; diff --git a/fs/gfs2/aops.c b/fs/gfs2/aops.c index 0a7b8076af3a5e..ee11d494e47c52 100644 --- a/fs/gfs2/aops.c +++ b/fs/gfs2/aops.c @@ -102,7 +102,7 @@ static int __gfs2_jdata_write_folio(struct folio *folio, struct writeback_control *wbc) { struct inode *inode = folio->mapping->host; - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); if (folio_test_checked(folio)) { folio_clear_checked(folio); @@ -111,7 +111,7 @@ static int __gfs2_jdata_write_folio(struct folio *folio, inode->i_sb->s_blocksize, BIT(BH_Dirty)|BIT(BH_Uptodate)); } - gfs2_trans_add_databufs(ip->i_gl, folio, 0, folio_size(folio)); + gfs2_trans_add_databufs(gl, folio, 0, folio_size(folio)); } return gfs2_write_jdata_folio(folio, wbc); } @@ -126,13 +126,13 @@ static int __gfs2_jdata_write_folio(struct folio *folio, int gfs2_jdata_writeback(struct address_space *mapping, struct writeback_control *wbc) { struct inode *inode = mapping->host; - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_sbd *sdp = GFS2_SB(mapping->host); struct folio *folio = NULL; int error; BUG_ON(current->journal_info); - if (gfs2_assert_withdraw(sdp, ip->i_gl->gl_state == LM_ST_EXCLUSIVE)) + if (gfs2_assert_withdraw(sdp, gl->gl_state == LM_ST_EXCLUSIVE)) return 0; while ((folio = writeback_iter(mapping, wbc, folio, &error))) { @@ -362,14 +362,14 @@ static int gfs2_write_cache_jdata(struct address_space *mapping, static int gfs2_jdata_writepages(struct address_space *mapping, struct writeback_control *wbc) { - struct gfs2_inode *ip = GFS2_I(mapping->host); + struct gfs2_glock *gl = gfs2_inode_glock(mapping->host); struct gfs2_sbd *sdp = GFS2_SB(mapping->host); int ret; ret = gfs2_write_cache_jdata(mapping, wbc); if (ret == 0 && wbc->sync_mode == WB_SYNC_ALL) { - gfs2_log_flush(sdp, ip->i_gl, GFS2_LOG_HEAD_FLUSH_NORMAL | - GFS2_LFC_JDATA_WPAGES); + gfs2_log_flush(sdp, gl, GFS2_LOG_HEAD_FLUSH_NORMAL | + GFS2_LFC_JDATA_WPAGES); ret = gfs2_write_cache_jdata(mapping, wbc); } return ret; @@ -561,12 +561,13 @@ static bool gfs2_jdata_dirty_folio(struct address_space *mapping, static sector_t gfs2_bmap(struct address_space *mapping, sector_t lblock) { + struct gfs2_glock *gl = gfs2_inode_glock(mapping->host); struct gfs2_inode *ip = GFS2_I(mapping->host); struct gfs2_holder i_gh; sector_t dblock = 0; int error; - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY, &i_gh); + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_ANY, &i_gh); if (error) return 0; diff --git a/fs/gfs2/bmap.c b/fs/gfs2/bmap.c index 636139f463b062..b5ddd936f48925 100644 --- a/fs/gfs2/bmap.c +++ b/fs/gfs2/bmap.c @@ -55,6 +55,7 @@ static int gfs2_unstuffer_folio(struct gfs2_inode *ip, struct buffer_head *dibh, u64 block, struct folio *folio) { struct inode *inode = &ip->i_inode; + struct gfs2_glock *gl = gfs2_inode_glock(inode); if (!folio_test_uptodate(folio)) { void *kaddr = kmap_local_folio(folio, 0); @@ -78,7 +79,7 @@ static int gfs2_unstuffer_folio(struct gfs2_inode *ip, struct buffer_head *dibh, map_bh(bh, inode->i_sb, block); set_buffer_uptodate(bh); - gfs2_trans_add_data(ip->i_gl, bh); + gfs2_trans_add_data(gl, bh); } else { folio_mark_dirty(folio); gfs2_ordered_add_inode(ip); @@ -90,6 +91,7 @@ static int gfs2_unstuffer_folio(struct gfs2_inode *ip, struct buffer_head *dibh, static int __gfs2_unstuff_inode(struct gfs2_inode *ip, struct folio *folio) { struct inode *inode = &ip->i_inode; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct buffer_head *bh, *dibh; struct gfs2_dinode *di; u64 block = 0; @@ -125,7 +127,7 @@ static int __gfs2_unstuff_inode(struct gfs2_inode *ip, struct folio *folio) /* Set up the pointer to the new block */ - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); di = (struct gfs2_dinode *)dibh->b_data; gfs2_buffer_clear_tail(dibh, sizeof(struct gfs2_dinode)); @@ -663,6 +665,7 @@ enum alloc_state { static int __gfs2_iomap_alloc(struct inode *inode, struct iomap *iomap, struct metapath *mp) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct buffer_head *dibh = metapath_dibh(mp); @@ -679,7 +682,7 @@ static int __gfs2_iomap_alloc(struct inode *inode, struct iomap *iomap, BUG_ON(dibh == NULL); BUG_ON(dblks < 1); - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); down_write(&ip->i_rw_mutex); @@ -723,7 +726,7 @@ static int __gfs2_iomap_alloc(struct inode *inode, struct iomap *iomap, } for (; i - 1 < mp->mp_fheight - ip->i_height && n > 0; i++, n--) - gfs2_indirect_init(mp, ip->i_gl, i, 0, bn++); + gfs2_indirect_init(mp, gl, i, 0, bn++); if (i - 1 == mp->mp_fheight - ip->i_height) { i--; gfs2_buffer_copy_tail(mp->mp_bh[i], @@ -749,9 +752,9 @@ static int __gfs2_iomap_alloc(struct inode *inode, struct iomap *iomap, fallthrough; /* To branching from existing tree */ case ALLOC_GROW_DEPTH: if (i > 1 && i < mp->mp_fheight) - gfs2_trans_add_meta(ip->i_gl, mp->mp_bh[i-1]); + gfs2_trans_add_meta(gl, mp->mp_bh[i-1]); for (; i < mp->mp_fheight && n > 0; i++, n--) - gfs2_indirect_init(mp, ip->i_gl, i, + gfs2_indirect_init(mp, gl, i, mp->mp_list[i-1], bn++); if (i == mp->mp_fheight) state = ALLOC_DATA; @@ -761,7 +764,7 @@ static int __gfs2_iomap_alloc(struct inode *inode, struct iomap *iomap, case ALLOC_DATA: BUG_ON(n > dblks); BUG_ON(mp->mp_bh[end_of_metadata] == NULL); - gfs2_trans_add_meta(ip->i_gl, mp->mp_bh[end_of_metadata]); + gfs2_trans_add_meta(gl, mp->mp_bh[end_of_metadata]); dblks = n; ptr = metapointer(end_of_metadata, mp); iomap->addr = bn << inode->i_blkbits; @@ -990,12 +993,12 @@ static void gfs2_iomap_put_folio(struct inode *inode, loff_t pos, unsigned copied, struct folio *folio) { struct gfs2_trans *tr = current->journal_info; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); if (gfs2_is_jdata(ip) && !gfs2_is_stuffed(ip)) - gfs2_trans_add_databufs(ip->i_gl, folio, - offset_in_folio(folio, pos), + gfs2_trans_add_databufs(gl, folio, offset_in_folio(folio, pos), copied); folio_unlock(folio); @@ -1151,6 +1154,7 @@ static int gfs2_iomap_begin(struct inode *inode, loff_t pos, loff_t length, static int gfs2_iomap_end(struct inode *inode, loff_t pos, loff_t length, ssize_t written, unsigned flags, struct iomap *iomap) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); @@ -1197,7 +1201,7 @@ static int gfs2_iomap_end(struct inode *inode, loff_t pos, loff_t length, if (iomap->flags & IOMAP_F_SIZE_CHANGED) mark_inode_dirty(inode); - set_bit(GLF_DIRTY, &ip->i_gl->gl_flags); + set_bit(GLF_DIRTY, &gl->gl_flags); return 0; } @@ -1388,6 +1392,7 @@ static int gfs2_journaled_truncate(struct inode *inode, u64 oldsize, u64 newsize static int trunc_start(struct inode *inode, u64 newsize) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct buffer_head *dibh = NULL; @@ -1416,7 +1421,7 @@ static int trunc_start(struct inode *inode, u64 newsize) if (error) goto out; - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); if (gfs2_is_stuffed(ip)) gfs2_buffer_clear_tail(dibh, sizeof(struct gfs2_dinode) + newsize); @@ -1490,6 +1495,7 @@ static int sweep_bh_for_rgrps(struct gfs2_inode *ip, struct gfs2_holder *rd_gh, bool meta, u32 *btotal) { struct inode *inode = &ip->i_inode; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct gfs2_rgrpd *rgd; struct gfs2_trans *tr; @@ -1589,7 +1595,7 @@ static int sweep_bh_for_rgrps(struct gfs2_inode *ip, struct gfs2_holder *rd_gh, goto out_unlock; } - gfs2_trans_add_meta(ip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); buf_in_tr = true; *p = 0; if (bstart + blen == bn) { @@ -1623,7 +1629,7 @@ static int sweep_bh_for_rgrps(struct gfs2_inode *ip, struct gfs2_holder *rd_gh, /* Every transaction boundary, we rewrite the dinode to keep its di_blocks current in case of failure. */ inode_set_mtime_to_ts(inode, inode_set_ctime_current(inode)); - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); gfs2_dinode_out(ip, dibh->b_data); brelse(dibh); up_write(&ip->i_rw_mutex); @@ -1749,6 +1755,7 @@ static inline bool walk_done(struct gfs2_sbd *sdp, static int punch_hole(struct gfs2_inode *ip, u64 offset, u64 length) { struct inode *inode = &ip->i_inode; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); u64 maxsize = sdp->sd_heightsize[ip->i_height]; struct metapath mp = {}; @@ -1836,7 +1843,7 @@ static int punch_hole(struct gfs2_inode *ip, u64 offset, u64 length) for (mp_h = 0; mp_h < mp.mp_aheight - 1; mp_h++) { metapointer_range(&mp, mp_h, start_list, start_aligned, end_list, end_aligned, &start, &end); - gfs2_metapath_ra(ip->i_gl, start, end); + gfs2_metapath_ra(gl, start, end); } if (mp.mp_aheight == ip->i_height) @@ -1956,7 +1963,7 @@ static int punch_hole(struct gfs2_inode *ip, u64 offset, u64 length) start_list, start_aligned, end_list, end_aligned, &start, &end); - gfs2_metapath_ra(ip->i_gl, start, end); + gfs2_metapath_ra(gl, start, end); } } @@ -1990,7 +1997,7 @@ static int punch_hole(struct gfs2_inode *ip, u64 offset, u64 length) gfs2_statfs_change(sdp, 0, +btotal, 0); gfs2_quota_change(ip, -(s64)btotal, inode->i_uid, inode->i_gid); inode_set_mtime_to_ts(inode, inode_set_ctime_current(inode)); - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); gfs2_dinode_out(ip, dibh->b_data); up_write(&ip->i_rw_mutex); gfs2_trans_end(sdp); @@ -2013,6 +2020,7 @@ static int punch_hole(struct gfs2_inode *ip, u64 offset, u64 length) static int trunc_end(struct gfs2_inode *ip) { struct inode *inode = &ip->i_inode; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct buffer_head *dibh; int error; @@ -2036,7 +2044,7 @@ static int trunc_end(struct gfs2_inode *ip) inode_set_mtime_to_ts(inode, inode_set_ctime_current(inode)); ip->i_diskflags &= ~GFS2_DIF_TRUNC_IN_PROG; - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); gfs2_dinode_out(ip, dibh->b_data); brelse(dibh); @@ -2097,6 +2105,7 @@ static int do_shrink(struct inode *inode, u64 newsize) static int do_grow(struct inode *inode, u64 size) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct gfs2_alloc_parms ap = { .target = 1, }; @@ -2141,7 +2150,7 @@ static int do_grow(struct inode *inode, u64 size) truncate_setsize(inode, size); inode_set_mtime_to_ts(&ip->i_inode, inode_set_ctime_current(&ip->i_inode)); - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); gfs2_dinode_out(ip, dibh->b_data); brelse(dibh); @@ -2380,6 +2389,7 @@ int gfs2_write_alloc_required(struct gfs2_inode *ip, u64 offset, static int stuffed_zero_range(struct inode *inode, loff_t offset, loff_t length) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct buffer_head *dibh; int error; @@ -2392,7 +2402,7 @@ static int stuffed_zero_range(struct inode *inode, loff_t offset, loff_t length) error = gfs2_meta_inode_buffer(ip, &dibh); if (error) return error; - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); memset(dibh->b_data + sizeof(struct gfs2_dinode) + offset, 0, length); brelse(dibh); diff --git a/fs/gfs2/dentry.c b/fs/gfs2/dentry.c index 95050e719233db..7b344461658ec3 100644 --- a/fs/gfs2/dentry.c +++ b/fs/gfs2/dentry.c @@ -35,8 +35,8 @@ static int gfs2_drevalidate(struct inode *dir, const struct qstr *name, struct dentry *dentry, unsigned int flags) { + struct gfs2_glock *gl = gfs2_inode_glock(dir); struct gfs2_sbd *sdp = GFS2_SB(dir); - struct gfs2_inode *dip = GFS2_I(dir); struct inode *inode; struct gfs2_holder d_gh; struct gfs2_inode *ip = NULL; @@ -57,9 +57,9 @@ static int gfs2_drevalidate(struct inode *dir, const struct qstr *name, if (sdp->sd_lockstruct.ls_ops->lm_mount == NULL) return 1; - had_lock = (gfs2_glock_is_locked_by_me(dip->i_gl) != NULL); + had_lock = (gfs2_glock_is_locked_by_me(gl) != NULL); if (!had_lock) { - error = gfs2_glock_nq_init(dip->i_gl, LM_ST_SHARED, 0, &d_gh); + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &d_gh); if (error) return 0; } diff --git a/fs/gfs2/dir.c b/fs/gfs2/dir.c index f6111276ebb072..6cfe335fd5909f 100644 --- a/fs/gfs2/dir.c +++ b/fs/gfs2/dir.c @@ -90,10 +90,11 @@ typedef int (*gfs2_dscan_t)(const struct gfs2_dirent *dent, int gfs2_dir_get_new_buffer(struct gfs2_inode *ip, u64 block, struct buffer_head **bhp) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct buffer_head *bh; - bh = gfs2_meta_new(ip->i_gl, block); - gfs2_trans_add_meta(ip->i_gl, bh); + bh = gfs2_meta_new(gl, block); + gfs2_trans_add_meta(gl, bh); gfs2_metatype_set(bh, GFS2_METATYPE_JD, GFS2_FORMAT_JD); gfs2_buffer_clear_tail(bh, sizeof(struct gfs2_meta_header)); *bhp = bh; @@ -103,10 +104,11 @@ int gfs2_dir_get_new_buffer(struct gfs2_inode *ip, u64 block, static int gfs2_dir_get_existing_buffer(struct gfs2_inode *ip, u64 block, struct buffer_head **bhp) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct buffer_head *bh; int error; - error = gfs2_meta_read(ip->i_gl, block, DIO_WAIT, 0, &bh); + error = gfs2_meta_read(gl, block, DIO_WAIT, 0, &bh); if (error) return error; if (gfs2_metatype_check(GFS2_SB(&ip->i_inode), bh, GFS2_METATYPE_JD)) { @@ -120,6 +122,7 @@ static int gfs2_dir_get_existing_buffer(struct gfs2_inode *ip, u64 block, static int gfs2_dir_write_stuffed(struct gfs2_inode *ip, const char *buf, unsigned int offset, unsigned int size) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct buffer_head *dibh; int error; @@ -127,7 +130,7 @@ static int gfs2_dir_write_stuffed(struct gfs2_inode *ip, const char *buf, if (error) return error; - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); memcpy(dibh->b_data + offset + sizeof(struct gfs2_dinode), buf, size); if (ip->i_inode.i_size < offset + size) i_size_write(&ip->i_inode, offset + size); @@ -153,6 +156,7 @@ static int gfs2_dir_write_stuffed(struct gfs2_inode *ip, const char *buf, static int gfs2_dir_write_data(struct gfs2_inode *ip, const char *buf, u64 offset, unsigned int size) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); struct buffer_head *dibh; u64 lblock, dblock; @@ -208,7 +212,7 @@ static int gfs2_dir_write_data(struct gfs2_inode *ip, const char *buf, if (error) goto fail; - gfs2_trans_add_meta(ip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); memcpy(bh->b_data + o, buf, amount); brelse(bh); @@ -230,7 +234,7 @@ static int gfs2_dir_write_data(struct gfs2_inode *ip, const char *buf, i_size_write(&ip->i_inode, offset + copied); inode_set_mtime_to_ts(&ip->i_inode, inode_set_ctime_current(&ip->i_inode)); - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); gfs2_dinode_out(ip, dibh->b_data); brelse(dibh); @@ -268,6 +272,7 @@ static int gfs2_dir_read_stuffed(struct gfs2_inode *ip, __be64 *buf, static int gfs2_dir_read_data(struct gfs2_inode *ip, __be64 *buf, unsigned int size) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); u64 lblock, dblock; u32 extlen = 0; @@ -299,9 +304,9 @@ static int gfs2_dir_read_data(struct gfs2_inode *ip, __be64 *buf, if (error || !dblock) goto fail; BUG_ON(extlen < 1); - bh = gfs2_meta_ra(ip->i_gl, dblock, extlen); + bh = gfs2_meta_ra(gl, dblock, extlen); } else { - error = gfs2_meta_read(ip->i_gl, dblock, DIO_WAIT, 0, &bh); + error = gfs2_meta_read(gl, dblock, DIO_WAIT, 0, &bh); if (error) goto fail; } @@ -672,6 +677,7 @@ static int dirent_next(struct gfs2_inode *dip, struct buffer_head *bh, static void dirent_del(struct gfs2_inode *dip, struct buffer_head *bh, struct gfs2_dirent *prev, struct gfs2_dirent *cur) { + struct gfs2_glock *gl = gfs2_inode_glock(&dip->i_inode); u16 cur_rec_len, prev_rec_len; if (gfs2_dirent_sentinel(cur)) { @@ -679,7 +685,7 @@ static void dirent_del(struct gfs2_inode *dip, struct buffer_head *bh, return; } - gfs2_trans_add_meta(dip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); /* If there is no prev entry, this is the first entry in the block. The de_rec_len is already as big as it needs to be. Just zero @@ -712,13 +718,13 @@ static struct gfs2_dirent *do_init_dirent(struct inode *inode, struct buffer_head *bh, unsigned offset) { - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_dirent *ndent; unsigned totlen; totlen = be16_to_cpu(dent->de_rec_len); BUG_ON(offset + name->len > totlen); - gfs2_trans_add_meta(ip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); ndent = (struct gfs2_dirent *)((char *)dent + offset); dent->de_rec_len = cpu_to_be16(offset); gfs2_qstr2dirent(name, totlen - offset, ndent); @@ -760,10 +766,11 @@ static int get_leaf(struct gfs2_inode *dip, u64 leaf_no, struct buffer_head **bhp) { struct inode *inode = &dip->i_inode; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); int error; - error = gfs2_meta_read(dip->i_gl, leaf_no, DIO_WAIT, 0, bhp); + error = gfs2_meta_read(gl, leaf_no, DIO_WAIT, 0, bhp); if (!error && gfs2_metatype_check(sdp, *bhp, GFS2_METATYPE_LF)) { /* pr_info("block num=%llu\n", leaf_no); */ error = -EIO; @@ -865,6 +872,7 @@ static struct gfs2_dirent *gfs2_dirent_search(struct inode *inode, static struct gfs2_leaf *new_leaf(struct inode *inode, struct buffer_head **pbh, u16 depth) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); unsigned int n = 1; u64 bn; @@ -877,12 +885,12 @@ static struct gfs2_leaf *new_leaf(struct inode *inode, struct buffer_head **pbh, error = gfs2_alloc_blocks(ip, &bn, &n, 0); if (error) return NULL; - bh = gfs2_meta_new(ip->i_gl, bn); + bh = gfs2_meta_new(gl, bn); if (!bh) return NULL; gfs2_trans_remove_revoke(GFS2_SB(inode), bn, 1); - gfs2_trans_add_meta(ip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); gfs2_metatype_set(bh, GFS2_METATYPE_LF, GFS2_FORMAT_LF); leaf = (struct gfs2_leaf *)bh->b_data; leaf->lf_depth = cpu_to_be16(depth); @@ -909,6 +917,7 @@ static struct gfs2_leaf *new_leaf(struct inode *inode, struct buffer_head **pbh, static int dir_make_exhash(struct inode *inode) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *dip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct gfs2_dirent *dent; @@ -970,7 +979,7 @@ static int dir_make_exhash(struct inode *inode) /* We're done with the new leaf block, now setup the new hash table. */ - gfs2_trans_add_meta(dip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); gfs2_buffer_clear_tail(dibh, sizeof(struct gfs2_dinode)); lp = (__be64 *)(dibh->b_data + sizeof(struct gfs2_dinode)); @@ -1000,6 +1009,7 @@ static int dir_make_exhash(struct inode *inode) static int dir_split_leaf(struct inode *inode, const struct qstr *name) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *dip = GFS2_I(inode); struct buffer_head *nbh, *obh, *dibh; struct gfs2_leaf *nleaf, *oleaf; @@ -1027,7 +1037,7 @@ static int dir_split_leaf(struct inode *inode, const struct qstr *name) return 1; /* can't split */ } - gfs2_trans_add_meta(dip->i_gl, obh); + gfs2_trans_add_meta(gl, obh); nleaf = new_leaf(inode, &nbh, be16_to_cpu(oleaf->lf_depth) + 1); if (!nleaf) { @@ -1120,7 +1130,7 @@ static int dir_split_leaf(struct inode *inode, const struct qstr *name) error = gfs2_meta_inode_buffer(dip, &dibh); if (!gfs2_assert_withdraw(GFS2_SB(&dip->i_inode), !error)) { - gfs2_trans_add_meta(dip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); gfs2_add_inode_blocks(&dip->i_inode, 1); gfs2_dinode_out(dip, dibh->b_data); brelse(dibh); @@ -1482,8 +1492,8 @@ static int gfs2_dir_read_leaf(struct inode *inode, struct dir_context *ctx, static void gfs2_dir_readahead(struct inode *inode, unsigned hsize, u32 index, struct file_ra_state *f_ra) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); - struct gfs2_glock *gl = ip->i_gl; struct buffer_head *bh; u64 blocknr = 0, last; unsigned count; @@ -1724,6 +1734,7 @@ int gfs2_dir_check(struct inode *dir, const struct qstr *name, static int dir_new_leaf(struct inode *inode, const struct qstr *name) { struct buffer_head *bh, *obh; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_leaf *leaf, *oleaf; u32 dist = 1; @@ -1747,7 +1758,7 @@ static int dir_new_leaf(struct inode *inode, const struct qstr *name) return error; } while(1); - gfs2_trans_add_meta(ip->i_gl, obh); + gfs2_trans_add_meta(gl, obh); leaf = new_leaf(inode, &bh, be16_to_cpu(oleaf->lf_depth)); if (!leaf) { @@ -1762,7 +1773,7 @@ static int dir_new_leaf(struct inode *inode, const struct qstr *name) error = gfs2_meta_inode_buffer(ip, &bh); if (error) return error; - gfs2_trans_add_meta(ip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); gfs2_add_inode_blocks(&ip->i_inode, 1); gfs2_dinode_out(ip, bh->b_data); brelse(bh); @@ -1937,6 +1948,7 @@ int gfs2_dir_del(struct gfs2_inode *dip, const struct dentry *dentry) int gfs2_dir_mvino(struct gfs2_inode *dip, const struct qstr *filename, const struct gfs2_inode *nip, unsigned int new_type) { + struct gfs2_glock *gl = gfs2_inode_glock(&dip->i_inode); struct buffer_head *bh; struct gfs2_dirent *dent; @@ -1948,7 +1960,7 @@ int gfs2_dir_mvino(struct gfs2_inode *dip, const struct qstr *filename, if (IS_ERR(dent)) return PTR_ERR(dent); - gfs2_trans_add_meta(dip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); gfs2_inum_out(nip, dent); dent->de_type = cpu_to_be16(new_type); brelse(bh); @@ -1974,6 +1986,7 @@ static int leaf_dealloc(struct gfs2_inode *dip, u32 index, u32 len, u64 leaf_no, struct buffer_head *leaf_bh, int last_dealloc) { + struct gfs2_glock *gl = gfs2_inode_glock(&dip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&dip->i_inode); struct gfs2_leaf *tmp_leaf; struct gfs2_rgrp_list rlist; @@ -2068,7 +2081,7 @@ static int leaf_dealloc(struct gfs2_inode *dip, u32 index, u32 len, if (error) goto out_end_trans; - gfs2_trans_add_meta(dip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); /* On the last dealloc, make this a regular file in case we crash. (We don't want to free these blocks a second time.) */ if (last_dealloc) diff --git a/fs/gfs2/export.c b/fs/gfs2/export.c index 3334c394ce9cbe..970bf2d723884b 100644 --- a/fs/gfs2/export.c +++ b/fs/gfs2/export.c @@ -87,7 +87,8 @@ static int gfs2_get_name(struct dentry *parent, char *name, { struct inode *dir = d_inode(parent); struct inode *inode = d_inode(child); - struct gfs2_inode *dip, *ip; + struct gfs2_glock *gl; + struct gfs2_inode *ip; struct get_name_filldir gnfd = { .ctx.actor = get_name_filldir, .name = name @@ -102,14 +103,14 @@ static int gfs2_get_name(struct dentry *parent, char *name, if (!S_ISDIR(dir->i_mode) || !inode) return -EINVAL; - dip = GFS2_I(dir); + gl = gfs2_inode_glock(dir); ip = GFS2_I(inode); *name = 0; gnfd.inum.no_addr = ip->i_no_addr; gnfd.inum.no_formal_ino = ip->i_no_formal_ino; - error = gfs2_glock_nq_init(dip->i_gl, LM_ST_SHARED, 0, &gh); + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &gh); if (error) return error; diff --git a/fs/gfs2/file.c b/fs/gfs2/file.c index 164e160e9064a9..6fb2adeef2741b 100644 --- a/fs/gfs2/file.c +++ b/fs/gfs2/file.c @@ -57,13 +57,13 @@ static loff_t gfs2_llseek(struct file *file, loff_t offset, int whence) { - struct gfs2_inode *ip = GFS2_I(file->f_mapping->host); + struct gfs2_glock *gl = gfs2_inode_glock(file->f_mapping->host); struct gfs2_holder i_gh; loff_t error; switch (whence) { case SEEK_END: - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY, + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_ANY, &i_gh); if (!error) { error = generic_file_llseek(file, offset, whence); @@ -105,11 +105,11 @@ static loff_t gfs2_llseek(struct file *file, loff_t offset, int whence) static int gfs2_readdir(struct file *file, struct dir_context *ctx) { struct inode *dir = file->f_mapping->host; - struct gfs2_inode *dip = GFS2_I(dir); + struct gfs2_glock *gl = gfs2_inode_glock(dir); struct gfs2_holder d_gh; int error; - error = gfs2_glock_nq_init(dip->i_gl, LM_ST_SHARED, 0, &d_gh); + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &d_gh); if (error) return error; @@ -158,6 +158,7 @@ static inline u32 gfs2_gfsflags_to_fsflags(struct inode *inode, u32 gfsflags) int gfs2_fileattr_get(struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder gh; int error; @@ -166,7 +167,7 @@ int gfs2_fileattr_get(struct dentry *dentry, struct file_kattr *fa) if (d_is_special(dentry)) return -ENOTTY; - gfs2_holder_init(ip->i_gl, LM_ST_SHARED, 0, &gh); + gfs2_holder_init(gl, LM_ST_SHARED, 0, &gh); error = gfs2_glock_nq(&gh); if (error) goto out_uninit; @@ -218,6 +219,7 @@ void gfs2_set_inode_flags(struct inode *inode) */ static int do_gfs2_set_flags(struct inode *inode, u32 reqflags, u32 mask) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct buffer_head *bh; @@ -225,7 +227,7 @@ static int do_gfs2_set_flags(struct inode *inode, u32 reqflags, u32 mask) int error; u32 new_flags, flags; - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh); + error = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, 0, &gh); if (error) return error; @@ -242,7 +244,7 @@ static int do_gfs2_set_flags(struct inode *inode, u32 reqflags, u32 mask) } if ((flags ^ new_flags) & GFS2_DIF_JDATA) { if (new_flags & GFS2_DIF_JDATA) - gfs2_log_flush(sdp, ip->i_gl, + gfs2_log_flush(sdp, gl, GFS2_LOG_HEAD_FLUSH_NORMAL | GFS2_LFC_SET_FLAGS); error = filemap_fdatawrite(inode->i_mapping); @@ -262,7 +264,7 @@ static int do_gfs2_set_flags(struct inode *inode, u32 reqflags, u32 mask) if (error) goto out_trans_end; inode_set_ctime_current(inode); - gfs2_trans_add_meta(ip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); ip->i_diskflags = new_flags; gfs2_dinode_out(ip, bh->b_data); brelse(bh); @@ -417,6 +419,7 @@ static vm_fault_t gfs2_page_mkwrite(struct vm_fault *vmf) { struct folio *folio = page_folio(vmf->page); struct inode *inode = file_inode(vmf->vma->vm_file); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct gfs2_alloc_parms ap = {}; @@ -430,7 +433,7 @@ static vm_fault_t gfs2_page_mkwrite(struct vm_fault *vmf) sb_start_pagefault(inode->i_sb); - gfs2_holder_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh); + gfs2_holder_init(gl, LM_ST_EXCLUSIVE, 0, &gh); err = gfs2_glock_nq(&gh); if (err) { ret = vmf_fs_error(err); @@ -455,7 +458,7 @@ static vm_fault_t gfs2_page_mkwrite(struct vm_fault *vmf) gfs2_size_hint(vmf->vma->vm_file, pos, length); - set_bit(GLF_DIRTY, &ip->i_gl->gl_flags); + set_bit(GLF_DIRTY, &gl->gl_flags); set_bit(GIF_SW_PAGED, &ip->i_flags); /* @@ -552,12 +555,12 @@ static vm_fault_t gfs2_page_mkwrite(struct vm_fault *vmf) static vm_fault_t gfs2_fault(struct vm_fault *vmf) { struct inode *inode = file_inode(vmf->vma->vm_file); - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_holder gh; vm_fault_t ret; int err; - gfs2_holder_init(ip->i_gl, LM_ST_SHARED, 0, &gh); + gfs2_holder_init(gl, LM_ST_SHARED, 0, &gh); err = gfs2_glock_nq(&gh); if (err) { ret = vmf_fs_error(err); @@ -590,6 +593,7 @@ static const struct vm_operations_struct gfs2_vm_ops = { static int gfs2_mmap(struct file *file, struct vm_area_struct *vma) { + struct gfs2_glock *gl = gfs2_inode_glock(file->f_mapping->host); struct gfs2_inode *ip = GFS2_I(file->f_mapping->host); if (!(file->f_flags & O_NOATIME) && @@ -597,7 +601,7 @@ static int gfs2_mmap(struct file *file, struct vm_area_struct *vma) struct gfs2_holder i_gh; int error; - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY, + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_ANY, &i_gh); if (error) return error; @@ -674,13 +678,14 @@ int gfs2_open_common(struct inode *inode, struct file *file) static int gfs2_open(struct inode *inode, struct file *file) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder i_gh; int error; bool need_unlock = false; if (S_ISREG(ip->i_inode.i_mode)) { - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY, + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_ANY, &i_gh); if (error) return error; @@ -745,6 +750,7 @@ static int gfs2_fsync(struct file *file, loff_t start, loff_t end, struct address_space *mapping = file->f_mapping; struct inode *inode = mapping->host; int sync_state = inode_state_read_once(inode) & I_DIRTY; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); int ret = 0, ret1 = 0; @@ -767,7 +773,7 @@ static int gfs2_fsync(struct file *file, loff_t start, loff_t end, ret = file_write_and_wait(file); if (ret) return ret; - gfs2_ail_flush(ip->i_gl, 1); + gfs2_ail_flush(gl, 1); } if (mapping->nrpages) @@ -813,7 +819,7 @@ static ssize_t gfs2_file_direct_read(struct kiocb *iocb, struct iov_iter *to, { struct file *file = iocb->ki_filp; struct inode *inode = file->f_mapping->host; - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); size_t prev_count = 0, window_size = 0; size_t read = 0; ssize_t ret; @@ -838,7 +844,7 @@ static ssize_t gfs2_file_direct_read(struct kiocb *iocb, struct iov_iter *to, if (!iov_iter_count(to)) return 0; /* skip atime */ - gfs2_holder_init(ip->i_gl, LM_ST_DEFERRED, 0, gh); + gfs2_holder_init(gl, LM_ST_DEFERRED, 0, gh); retry: ret = gfs2_glock_nq(gh); if (ret) @@ -877,7 +883,7 @@ static ssize_t gfs2_file_direct_write(struct kiocb *iocb, struct iov_iter *from, { struct file *file = iocb->ki_filp; struct inode *inode = file->f_mapping->host; - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); size_t prev_count = 0, window_size = 0; size_t written = 0; bool enough_retries; @@ -901,7 +907,7 @@ static ssize_t gfs2_file_direct_write(struct kiocb *iocb, struct iov_iter *from, * unfortunately, have the option of only flushing a range like the * VFS does. */ - gfs2_holder_init(ip->i_gl, LM_ST_DEFERRED, 0, gh); + gfs2_holder_init(gl, LM_ST_DEFERRED, 0, gh); retry: ret = gfs2_glock_nq(gh); if (ret) @@ -949,7 +955,7 @@ static ssize_t gfs2_file_direct_write(struct kiocb *iocb, struct iov_iter *from, static ssize_t gfs2_file_read_iter(struct kiocb *iocb, struct iov_iter *to) { - struct gfs2_inode *ip; + struct gfs2_glock *gl; struct gfs2_holder gh; size_t prev_count = 0, window_size = 0; size_t read = 0; @@ -980,8 +986,8 @@ static ssize_t gfs2_file_read_iter(struct kiocb *iocb, struct iov_iter *to) if (iocb->ki_flags & IOCB_NOWAIT) return ret; } - ip = GFS2_I(iocb->ki_filp->f_mapping->host); - gfs2_holder_init(ip->i_gl, LM_ST_SHARED, 0, &gh); + gl = gfs2_inode_glock(iocb->ki_filp->f_mapping->host); + gfs2_holder_init(gl, LM_ST_SHARED, 0, &gh); retry: ret = gfs2_glock_nq(&gh); if (ret) @@ -1014,7 +1020,7 @@ static ssize_t gfs2_file_buffered_write(struct kiocb *iocb, { struct file *file = iocb->ki_filp; struct inode *inode = file_inode(file); - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct gfs2_holder *statfs_gh = NULL; size_t prev_count = 0, window_size = 0; @@ -1035,7 +1041,7 @@ static ssize_t gfs2_file_buffered_write(struct kiocb *iocb, return -ENOMEM; } - gfs2_holder_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, gh); + gfs2_holder_init(gl, LM_ST_EXCLUSIVE, 0, gh); if (should_fault_in_pages(from, iocb, &prev_count, &window_size)) { retry: window_size -= fault_in_iov_iter_readable(from, window_size); @@ -1050,9 +1056,9 @@ static ssize_t gfs2_file_buffered_write(struct kiocb *iocb, goto out_uninit; if (inode == sdp->sd_rindex) { - struct gfs2_inode *m_ip = GFS2_I(sdp->sd_statfs_inode); + struct gfs2_glock *m_gl = gfs2_inode_glock(sdp->sd_statfs_inode); - ret = gfs2_glock_nq_init(m_ip->i_gl, LM_ST_EXCLUSIVE, + ret = gfs2_glock_nq_init(m_gl, LM_ST_EXCLUSIVE, GL_NOCACHE, statfs_gh); if (ret) goto out_unlock; @@ -1106,14 +1112,15 @@ static ssize_t gfs2_file_write_iter(struct kiocb *iocb, struct iov_iter *from) { struct file *file = iocb->ki_filp; struct inode *inode = file_inode(file); - struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder gh; ssize_t ret; gfs2_size_hint(file, iocb->ki_pos, iov_iter_count(from)); if (iocb->ki_flags & IOCB_APPEND) { - ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, 0, &gh); + struct gfs2_glock *gl = gfs2_inode_glock(inode); + + ret = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &gh); if (ret) return ret; gfs2_glock_dq_uninit(&gh); @@ -1181,6 +1188,7 @@ static ssize_t gfs2_file_write_iter(struct kiocb *iocb, struct iov_iter *from) static int fallocate_chunk(struct inode *inode, loff_t offset, loff_t len) { struct super_block *sb = inode->i_sb; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); loff_t end = offset + len; struct buffer_head *dibh; @@ -1190,7 +1198,7 @@ static int fallocate_chunk(struct inode *inode, loff_t offset, loff_t len) if (unlikely(error)) return error; - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); if (gfs2_is_stuffed(ip)) { error = gfs2_unstuff_dinode(ip); @@ -1379,6 +1387,7 @@ static long gfs2_fallocate(struct file *file, int mode, loff_t offset, loff_t le { struct inode *inode = file_inode(file); struct gfs2_sbd *sdp = GFS2_SB(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder gh; int ret; @@ -1391,7 +1400,7 @@ static long gfs2_fallocate(struct file *file, int mode, loff_t offset, loff_t le inode_lock(inode); - gfs2_holder_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh); + gfs2_holder_init(gl, LM_ST_EXCLUSIVE, 0, &gh); ret = gfs2_glock_nq(&gh); if (ret) goto out_uninit; diff --git a/fs/gfs2/glock.c b/fs/gfs2/glock.c index d59ea71a84db06..d0612014408e43 100644 --- a/fs/gfs2/glock.c +++ b/fs/gfs2/glock.c @@ -891,7 +891,9 @@ static void gfs2_try_to_evict(struct gfs2_glock *gl) /* If the inode was evicted, gl->gl_object will now be NULL. */ ip = gfs2_grab_existing_inode(gl); if (ip) { - gfs2_glock_poke(ip->i_gl); + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); + + gfs2_glock_poke(gl); iput(&ip->i_inode); } } diff --git a/fs/gfs2/glops.c b/fs/gfs2/glops.c index 28f32424ee64bf..662d033fd2cafa 100644 --- a/fs/gfs2/glops.c +++ b/fs/gfs2/glops.c @@ -602,8 +602,7 @@ static void freeze_go_callback(struct gfs2_glock *gl, bool remote) static int freeze_go_xmote_bh(struct gfs2_glock *gl) { struct gfs2_sbd *sdp = glock_sbd(gl); - struct gfs2_inode *ip = GFS2_I(sdp->sd_jdesc->jd_inode); - struct gfs2_glock *j_gl = ip->i_gl; + struct gfs2_glock *j_gl = gfs2_inode_glock(sdp->sd_jdesc->jd_inode); struct gfs2_log_header_host head; int error; diff --git a/fs/gfs2/incore.h b/fs/gfs2/incore.h index dadb4d3c9d3d61..3ae8e2be486c90 100644 --- a/fs/gfs2/incore.h +++ b/fs/gfs2/incore.h @@ -879,5 +879,8 @@ static inline unsigned gfs2_max_stuffed_size(const struct gfs2_inode *ip) return GFS2_SB(&ip->i_inode)->sd_sb.sb_bsize - sizeof(struct gfs2_dinode); } +static inline struct gfs2_glock *gfs2_inode_glock(struct inode *inode) +{ + return GFS2_I(inode)->i_gl; +} #endif /* __INCORE_DOT_H__ */ - diff --git a/fs/gfs2/inode.c b/fs/gfs2/inode.c index 438519bc06d496..5aaed0018fb338 100644 --- a/fs/gfs2/inode.c +++ b/fs/gfs2/inode.c @@ -324,8 +324,8 @@ struct inode *gfs2_lookup_meta(struct inode *dip, const char *name) struct inode *gfs2_lookupi(struct inode *dir, const struct qstr *name, int is_root) { + struct gfs2_glock *gl = gfs2_inode_glock(dir); struct super_block *sb = dir->i_sb; - struct gfs2_inode *dip = GFS2_I(dir); struct gfs2_holder d_gh; int error = 0; struct inode *inode = NULL; @@ -341,8 +341,8 @@ struct inode *gfs2_lookupi(struct inode *dir, const struct qstr *name, return dir; } - if (gfs2_glock_is_locked_by_me(dip->i_gl) == NULL) { - error = gfs2_glock_nq_init(dip->i_gl, LM_ST_SHARED, 0, &d_gh); + if (gfs2_glock_is_locked_by_me(gl) == NULL) { + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &d_gh); if (error) return ERR_PTR(error); } @@ -458,7 +458,7 @@ static int alloc_dinode(struct gfs2_inode *ip, u32 flags, unsigned *dblocks) static void gfs2_final_release_pages(struct gfs2_inode *ip) { struct inode *inode = &ip->i_inode; - struct gfs2_glock *gl = ip->i_gl; + struct gfs2_glock *gl = gfs2_inode_glock(inode); /* This can only happen during incomplete inode creation. */ if (unlikely(!gl)) @@ -548,12 +548,13 @@ static void gfs2_init_dir(struct buffer_head *dibh, static void gfs2_init_xattr(struct gfs2_inode *ip) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); struct buffer_head *bh; struct gfs2_ea_header *ea; - bh = gfs2_meta_new(ip->i_gl, ip->i_eattr); - gfs2_trans_add_meta(ip->i_gl, bh); + bh = gfs2_meta_new(gl, ip->i_eattr); + gfs2_trans_add_meta(gl, bh); gfs2_metatype_set(bh, GFS2_METATYPE_EA, GFS2_FORMAT_EA); gfs2_buffer_clear_tail(bh, sizeof(struct gfs2_meta_header)); @@ -576,11 +577,12 @@ static void gfs2_init_xattr(struct gfs2_inode *ip) static void init_dinode(struct gfs2_inode *dip, struct gfs2_inode *ip, const char *symname) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_dinode *di; struct buffer_head *dibh; - dibh = gfs2_meta_new(ip->i_gl, ip->i_no_addr); - gfs2_trans_add_meta(ip->i_gl, dibh); + dibh = gfs2_meta_new(gl, ip->i_no_addr); + gfs2_trans_add_meta(gl, dibh); di = (struct gfs2_dinode *)dibh->b_data; gfs2_dinode_out(ip, di); @@ -728,7 +730,8 @@ static int gfs2_create_inode(struct inode *dir, struct dentry *dentry, if (error) goto fail; - error = gfs2_glock_nq_init(dip->i_gl, LM_ST_EXCLUSIVE, 0, &d_gh); + error = gfs2_glock_nq_init(gfs2_inode_glock(dir), LM_ST_EXCLUSIVE, 0, + &d_gh); if (error) goto fail; gfs2_holder_mark_uninitialized(&gh); @@ -999,7 +1002,7 @@ static struct dentry *__gfs2_lookup(struct inode *dir, struct dentry *dentry, if (inode == NULL || IS_ERR(inode)) return d_splice_alias(inode, dentry); - gl = GFS2_I(inode)->i_gl; + gl = gfs2_inode_glock(inode); error = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_ANY, &gh); if (error) { iput(inode); @@ -1059,8 +1062,8 @@ static int gfs2_link(struct dentry *old_dentry, struct inode *dir, if (error) return error; - gfs2_holder_init(dip->i_gl, LM_ST_EXCLUSIVE, 0, &d_gh); - gfs2_holder_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh); + gfs2_holder_init(gfs2_inode_glock(dir), LM_ST_EXCLUSIVE, 0, &d_gh); + gfs2_holder_init(gfs2_inode_glock(inode), LM_ST_EXCLUSIVE, 0, &gh); error = gfs2_glock_nq(&d_gh); if (error) @@ -1133,7 +1136,7 @@ static int gfs2_link(struct dentry *old_dentry, struct inode *dir, if (error) goto out_brelse; - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gfs2_inode_glock(inode), dibh); inc_nlink(&ip->i_inode); inode_set_ctime_current(&ip->i_inode); ihold(inode); @@ -1259,8 +1262,8 @@ static int gfs2_unlink(struct inode *dir, struct dentry *dentry) error = -EROFS; - gfs2_holder_init(dip->i_gl, LM_ST_EXCLUSIVE, 0, &d_gh); - gfs2_holder_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh); + gfs2_holder_init(gfs2_inode_glock(dir), LM_ST_EXCLUSIVE, 0, &d_gh); + gfs2_holder_init(gfs2_inode_glock(inode), LM_ST_EXCLUSIVE, 0, &gh); rgd = gfs2_blk2rgrpd(sdp, ip->i_no_addr, 1); if (!rgd) @@ -1534,18 +1537,19 @@ static int gfs2_rename(struct inode *odir, struct dentry *odentry, } num_gh = 1; - gfs2_holder_init(odip->i_gl, LM_ST_EXCLUSIVE, GL_ASYNC, ghs); + gfs2_holder_init(gfs2_inode_glock(odir), LM_ST_EXCLUSIVE, GL_ASYNC, ghs); if (odip != ndip) { - gfs2_holder_init(ndip->i_gl, LM_ST_EXCLUSIVE,GL_ASYNC, + gfs2_holder_init(gfs2_inode_glock(ndir), LM_ST_EXCLUSIVE,GL_ASYNC, ghs + num_gh); num_gh++; } - gfs2_holder_init(ip->i_gl, LM_ST_EXCLUSIVE, GL_ASYNC, ghs + num_gh); + gfs2_holder_init(gfs2_inode_glock(&ip->i_inode), LM_ST_EXCLUSIVE, + GL_ASYNC, ghs + num_gh); num_gh++; if (nip) { - gfs2_holder_init(nip->i_gl, LM_ST_EXCLUSIVE, GL_ASYNC, - ghs + num_gh); + gfs2_holder_init(gfs2_inode_glock(&nip->i_inode), + LM_ST_EXCLUSIVE, GL_ASYNC, ghs + num_gh); num_gh++; } @@ -1780,16 +1784,19 @@ static int gfs2_exchange(struct inode *odir, struct dentry *odentry, } num_gh = 1; - gfs2_holder_init(odip->i_gl, LM_ST_EXCLUSIVE, GL_ASYNC, ghs); + gfs2_holder_init(gfs2_inode_glock(odir), LM_ST_EXCLUSIVE, GL_ASYNC, + ghs); if (odip != ndip) { - gfs2_holder_init(ndip->i_gl, LM_ST_EXCLUSIVE, GL_ASYNC, - ghs + num_gh); + gfs2_holder_init(gfs2_inode_glock(ndir), LM_ST_EXCLUSIVE, + GL_ASYNC, ghs + num_gh); num_gh++; } - gfs2_holder_init(oip->i_gl, LM_ST_EXCLUSIVE, GL_ASYNC, ghs + num_gh); + gfs2_holder_init(gfs2_inode_glock(&oip->i_inode), LM_ST_EXCLUSIVE, + GL_ASYNC, ghs + num_gh); num_gh++; - gfs2_holder_init(nip->i_gl, LM_ST_EXCLUSIVE, GL_ASYNC, ghs + num_gh); + gfs2_holder_init(gfs2_inode_glock(&nip->i_inode), LM_ST_EXCLUSIVE, + GL_ASYNC, ghs + num_gh); num_gh++; again: @@ -1910,6 +1917,7 @@ static const char *gfs2_get_link(struct dentry *dentry, struct inode *inode, struct delayed_call *done) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder i_gh; struct buffer_head *dibh; @@ -1920,7 +1928,7 @@ static const char *gfs2_get_link(struct dentry *dentry, if (!dentry) return ERR_PTR(-ECHILD); - gfs2_holder_init(ip->i_gl, LM_ST_SHARED, 0, &i_gh); + gfs2_holder_init(gl, LM_ST_SHARED, 0, &i_gh); error = gfs2_glock_nq(&i_gh); if (error) { gfs2_holder_uninit(&i_gh); @@ -2102,6 +2110,7 @@ static int gfs2_setattr(struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder i_gh; int error; @@ -2110,7 +2119,7 @@ static int gfs2_setattr(struct mnt_idmap *idmap, if (error) return error; - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &i_gh); + error = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, 0, &i_gh); if (error) goto out; @@ -2164,14 +2173,15 @@ static int gfs2_getattr(struct mnt_idmap *idmap, u32 request_mask, unsigned int flags) { struct inode *inode = d_inode(path->dentry); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder gh; u32 gfsflags; int error; gfs2_holder_mark_uninitialized(&gh); - if (gfs2_glock_is_locked_by_me(ip->i_gl) == NULL) { - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY, &gh); + if (gfs2_glock_is_locked_by_me(gl) == NULL) { + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_ANY, &gh); if (error) return error; } @@ -2207,14 +2217,14 @@ static bool fault_in_fiemap(struct fiemap_extent_info *fi) static int gfs2_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo, u64 start, u64 len) { - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_holder gh; int ret; inode_lock_shared(inode); retry: - ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, 0, &gh); + ret = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &gh); if (ret) goto out; @@ -2237,12 +2247,12 @@ static int gfs2_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo, loff_t gfs2_seek_data(struct file *file, loff_t offset) { struct inode *inode = file->f_mapping->host; - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_holder gh; loff_t ret; inode_lock_shared(inode); - ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, 0, &gh); + ret = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &gh); if (!ret) ret = iomap_seek_data(inode, offset, &gfs2_iomap_ops); gfs2_glock_dq_uninit(&gh); @@ -2256,12 +2266,12 @@ loff_t gfs2_seek_data(struct file *file, loff_t offset) loff_t gfs2_seek_hole(struct file *file, loff_t offset) { struct inode *inode = file->f_mapping->host; - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_holder gh; loff_t ret; inode_lock_shared(inode); - ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, 0, &gh); + ret = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &gh); if (!ret) ret = iomap_seek_hole(inode, offset, &gfs2_iomap_ops); gfs2_glock_dq_uninit(&gh); @@ -2275,8 +2285,7 @@ loff_t gfs2_seek_hole(struct file *file, loff_t offset) static int gfs2_update_time(struct inode *inode, enum fs_update_time type, unsigned int flags) { - struct gfs2_inode *ip = GFS2_I(inode); - struct gfs2_glock *gl = ip->i_gl; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_holder *gh; int error; diff --git a/fs/gfs2/lops.c b/fs/gfs2/lops.c index 6dabe73ad790d9..77ef22eab36844 100644 --- a/fs/gfs2/lops.c +++ b/fs/gfs2/lops.c @@ -775,9 +775,8 @@ static int buf_lo_scan_elements(struct gfs2_jdesc *jd, u32 start, struct gfs2_log_descriptor *ld, __be64 *ptr, int pass) { - struct gfs2_inode *ip = GFS2_I(jd->jd_inode); + struct gfs2_glock *gl = gfs2_inode_glock(jd->jd_inode); struct gfs2_sbd *sdp = GFS2_SB(jd->jd_inode); - struct gfs2_glock *gl = ip->i_gl; unsigned int blks = be32_to_cpu(ld->ld_data1); struct buffer_head *bh_log, *bh_ip; u64 blkno; @@ -828,17 +827,17 @@ static int buf_lo_scan_elements(struct gfs2_jdesc *jd, u32 start, static void buf_lo_after_scan(struct gfs2_jdesc *jd, int error, int pass) { - struct gfs2_inode *ip = GFS2_I(jd->jd_inode); + struct gfs2_glock *gl = gfs2_inode_glock(jd->jd_inode); struct gfs2_sbd *sdp = GFS2_SB(jd->jd_inode); if (error) { - gfs2_inode_metasync(ip->i_gl); + gfs2_inode_metasync(gl); return; } if (pass != 1) return; - gfs2_inode_metasync(ip->i_gl); + gfs2_inode_metasync(gl); fs_info(sdp, "jid=%u: Replayed %u of %u blocks\n", jd->jd_jid, jd->jd_replayed_blocks, jd->jd_found_blocks); @@ -1000,8 +999,7 @@ static int databuf_lo_scan_elements(struct gfs2_jdesc *jd, u32 start, struct gfs2_log_descriptor *ld, __be64 *ptr, int pass) { - struct gfs2_inode *ip = GFS2_I(jd->jd_inode); - struct gfs2_glock *gl = ip->i_gl; + struct gfs2_glock *gl = gfs2_inode_glock(jd->jd_inode); unsigned int blks = be32_to_cpu(ld->ld_data1); struct buffer_head *bh_log, *bh_ip; u64 blkno; @@ -1048,18 +1046,18 @@ static int databuf_lo_scan_elements(struct gfs2_jdesc *jd, u32 start, static void databuf_lo_after_scan(struct gfs2_jdesc *jd, int error, int pass) { - struct gfs2_inode *ip = GFS2_I(jd->jd_inode); + struct gfs2_glock *gl = gfs2_inode_glock(jd->jd_inode); struct gfs2_sbd *sdp = GFS2_SB(jd->jd_inode); if (error) { - gfs2_inode_metasync(ip->i_gl); + gfs2_inode_metasync(gl); return; } if (pass != 1) return; /* data sync? */ - gfs2_inode_metasync(ip->i_gl); + gfs2_inode_metasync(gl); fs_info(sdp, "jid=%u: Replayed %u of %u data blocks\n", jd->jd_jid, jd->jd_replayed_blocks, jd->jd_found_blocks); diff --git a/fs/gfs2/meta_io.c b/fs/gfs2/meta_io.c index a87cfbf0df3870..63591a4be10a01 100644 --- a/fs/gfs2/meta_io.c +++ b/fs/gfs2/meta_io.c @@ -402,18 +402,19 @@ static struct buffer_head *gfs2_getjdatabuf(struct gfs2_inode *ip, u64 blkno) void gfs2_journal_wipe(struct gfs2_inode *ip, u64 bstart, u32 blen) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); struct buffer_head *bh; int ty; /* This can only happen during incomplete inode creation. */ - if (!ip->i_gl) + if (!gl) return; gfs2_ail1_wipe(sdp, bstart, blen); while (blen) { ty = REMOVE_META; - bh = gfs2_getbuf(ip->i_gl, bstart, NO_CREATE); + bh = gfs2_getbuf(gl, bstart, NO_CREATE); if (!bh && gfs2_is_jdata(ip)) { bh = gfs2_getjdatabuf(ip, bstart); ty = REMOVE_JDATA; @@ -447,8 +448,8 @@ void gfs2_journal_wipe(struct gfs2_inode *ip, u64 bstart, u32 blen) int gfs2_meta_buffer(struct gfs2_inode *ip, u32 mtype, u64 num, struct buffer_head **bhp) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); - struct gfs2_glock *gl = ip->i_gl; struct buffer_head *bh; int ret = 0; int rahead = 0; diff --git a/fs/gfs2/ops_fstype.c b/fs/gfs2/ops_fstype.c index 718e0da7dfce02..188b3e67f2d1eb 100644 --- a/fs/gfs2/ops_fstype.c +++ b/fs/gfs2/ops_fstype.c @@ -534,7 +534,7 @@ static void gfs2_others_may_mount(struct gfs2_sbd *sdp) static int gfs2_jindex_hold(struct gfs2_sbd *sdp, struct gfs2_holder *ji_gh) { - struct gfs2_inode *dip = GFS2_I(sdp->sd_jindex); + struct gfs2_glock *gl = gfs2_inode_glock(sdp->sd_jindex); struct qstr name; char buf[20]; struct gfs2_jdesc *jd; @@ -545,7 +545,7 @@ static int gfs2_jindex_hold(struct gfs2_sbd *sdp, struct gfs2_holder *ji_gh) mutex_lock(&sdp->sd_jindex_mutex); for (;;) { - error = gfs2_glock_nq_init(dip->i_gl, LM_ST_SHARED, 0, ji_gh); + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, ji_gh); if (error) break; @@ -611,7 +611,6 @@ static int init_statfs(struct gfs2_sbd *sdp) struct inode *pn = NULL; char buf[30]; struct gfs2_jdesc *jd; - struct gfs2_inode *ip; sdp->sd_statfs_inode = gfs2_lookup_meta(master, "statfs"); if (IS_ERR(sdp->sd_statfs_inode)) { @@ -656,15 +655,15 @@ static int init_statfs(struct gfs2_sbd *sdp) iput(pn); pn = NULL; - ip = GFS2_I(sdp->sd_sc_inode); - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, GL_NOPID, - &sdp->sd_sc_gh); + error = gfs2_glock_nq_init(gfs2_inode_glock(sdp->sd_sc_inode), + LM_ST_EXCLUSIVE, GL_NOPID, &sdp->sd_sc_gh); if (error) { fs_err(sdp, "can't lock local \"sc\" file: %d\n", error); goto free_local; } /* read in the local statfs buffer - other nodes don't change it. */ - error = gfs2_meta_inode_buffer(ip, &sdp->sd_sc_bh); + error = gfs2_meta_inode_buffer(GFS2_I(sdp->sd_sc_inode), + &sdp->sd_sc_bh); if (error) { fs_err(sdp, "Cannot read in local statfs: %d\n", error); goto unlock_sd_gh; @@ -697,7 +696,7 @@ static int init_journal(struct gfs2_sbd *sdp, int undo) { struct inode *master = d_inode(sdp->sd_master_dir); struct gfs2_holder ji_gh; - struct gfs2_inode *ip; + struct gfs2_glock *gl; int error = 0; gfs2_holder_mark_uninitialized(&ji_gh); @@ -751,8 +750,8 @@ static int init_journal(struct gfs2_sbd *sdp, int undo) goto fail_jindex; } - ip = GFS2_I(sdp->sd_jdesc->jd_inode); - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, + gl = gfs2_inode_glock(sdp->sd_jdesc->jd_inode); + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_RECOVER | GL_EXACT | GL_NOCACHE | GL_NOPID, &sdp->sd_jinode_gh); @@ -893,7 +892,7 @@ static int init_per_node(struct gfs2_sbd *sdp, int undo) struct inode *pn = NULL; char buf[30]; int error = 0; - struct gfs2_inode *ip; + struct gfs2_glock *gl; struct inode *master = d_inode(sdp->sd_master_dir); if (sdp->sd_args.ar_spectator) @@ -920,8 +919,8 @@ static int init_per_node(struct gfs2_sbd *sdp, int undo) iput(pn); pn = NULL; - ip = GFS2_I(sdp->sd_qc_inode); - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, GL_NOPID, + gl = gfs2_inode_glock(sdp->sd_qc_inode); + error = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, GL_NOPID, &sdp->sd_qc_gh); if (error) { fs_err(sdp, "can't lock local \"qc\" file: %d\n", error); diff --git a/fs/gfs2/quota.c b/fs/gfs2/quota.c index 0cb2bb0aec7b23..b62431724eea88 100644 --- a/fs/gfs2/quota.c +++ b/fs/gfs2/quota.c @@ -408,7 +408,7 @@ static int bh_get(struct gfs2_quota_data *qd) { struct gfs2_sbd *sdp = qd->qd_sbd; struct inode *inode = sdp->sd_qc_inode; - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); unsigned int block, offset; struct buffer_head *bh = NULL; struct iomap iomap = { }; @@ -434,7 +434,7 @@ static int bh_get(struct gfs2_quota_data *qd) if (iomap.type != IOMAP_MAPPED) return error; - error = gfs2_meta_read(ip->i_gl, iomap.addr >> inode->i_blkbits, + error = gfs2_meta_read(gl, iomap.addr >> inode->i_blkbits, DIO_WAIT, 0, &bh); if (error) return error; @@ -687,12 +687,12 @@ static int sort_qd(const void *a, const void *b) static void do_qc(struct gfs2_quota_data *qd, s64 change) { struct gfs2_sbd *sdp = qd->qd_sbd; - struct gfs2_inode *ip = GFS2_I(sdp->sd_qc_inode); + struct gfs2_glock *gl = gfs2_inode_glock(sdp->sd_qc_inode); struct gfs2_quota_change *qc = qd->qd_bh_qc; bool needs_put = false; s64 x; - gfs2_trans_add_meta(ip->i_gl, qd->qd_bh); + gfs2_trans_add_meta(gl, qd->qd_bh); /* * The QDF_CHANGE flag indicates that the slot in the quota change file @@ -741,7 +741,7 @@ static int gfs2_write_buf_to_page(struct gfs2_sbd *sdp, unsigned long index, unsigned off, void *buf, unsigned bytes) { struct inode *inode = sdp->sd_quota_inode; - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct address_space *mapping = inode->i_mapping; struct folio *folio; struct buffer_head *bh; @@ -780,7 +780,7 @@ static int gfs2_write_buf_to_page(struct gfs2_sbd *sdp, unsigned long index, set_buffer_uptodate(bh); if (bh_read(bh, REQ_META | REQ_PRIO) < 0) goto unlock_out; - gfs2_trans_add_data(ip->i_gl, bh); + gfs2_trans_add_data(gl, bh); /* If we need to write to the next block as well */ if (to_write > (bsize - boff)) { @@ -909,6 +909,7 @@ static int do_sync(unsigned int num_qd, struct gfs2_quota_data **qda, { struct gfs2_sbd *sdp = (*qda)->qd_sbd; struct inode *inode = sdp->sd_quota_inode; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_alloc_parms ap = {}; unsigned int data_blocks, ind_blocks; @@ -936,7 +937,7 @@ static int do_sync(unsigned int num_qd, struct gfs2_quota_data **qda, goto out_dq; } - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &i_gh); + error = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, 0, &i_gh); if (error) goto out_dq; @@ -994,8 +995,7 @@ static int do_sync(unsigned int num_qd, struct gfs2_quota_data **qda, gfs2_glock_dq_uninit(&ghs[qx]); inode_unlock(inode); kfree(ghs); - gfs2_log_flush(sdp, ip->i_gl, - GFS2_LOG_HEAD_FLUSH_NORMAL | GFS2_LFC_DO_SYNC); + gfs2_log_flush(sdp, gl, GFS2_LOG_HEAD_FLUSH_NORMAL | GFS2_LFC_DO_SYNC); if (!error) { for (x = 0; x < num_qd; x++) { qd = qda[x]; @@ -1039,7 +1039,7 @@ static int do_glock(struct gfs2_quota_data *qd, int force_refresh, struct gfs2_holder *q_gh) { struct gfs2_sbd *sdp = qd->qd_sbd; - struct gfs2_inode *ip = GFS2_I(sdp->sd_quota_inode); + struct gfs2_glock *gl = gfs2_inode_glock(sdp->sd_quota_inode); struct gfs2_holder i_gh; int error; @@ -1063,7 +1063,7 @@ static int do_glock(struct gfs2_quota_data *qd, int force_refresh, if (error) return error; - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, 0, &i_gh); + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &i_gh); if (error) goto fail; @@ -1404,7 +1404,7 @@ int gfs2_quota_refresh(struct gfs2_sbd *sdp, struct kqid qid) int gfs2_quota_init(struct gfs2_sbd *sdp) { struct inode *inode = sdp->sd_qc_inode; - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); u64 size = i_size_read(sdp->sd_qc_inode); unsigned int blocks = size >> sdp->sd_sb.sb_bsize_shift; unsigned int x, slot = 0; @@ -1441,7 +1441,7 @@ int gfs2_quota_init(struct gfs2_sbd *sdp) goto fail; } error = -EIO; - bh = gfs2_meta_ra(ip->i_gl, dblock, extlen); + bh = gfs2_meta_ra(gl, dblock, extlen); if (!bh) goto fail; if (gfs2_metatype_check(sdp, bh, GFS2_METATYPE_QC)) @@ -1717,6 +1717,7 @@ static int gfs2_set_dqblk(struct super_block *sb, struct kqid qid, { struct gfs2_sbd *sdp = sb->s_fs_info; struct inode *inode = sdp->sd_quota_inode; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_quota_data *qd; struct gfs2_holder q_gh, i_gh; @@ -1748,7 +1749,7 @@ static int gfs2_set_dqblk(struct super_block *sb, struct kqid qid, error = gfs2_glock_nq_init(qd->qd_gl, LM_ST_EXCLUSIVE, 0, &q_gh); if (error) goto out_unlockput; - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &i_gh); + error = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, 0, &i_gh); if (error) goto out_q; diff --git a/fs/gfs2/recovery.c b/fs/gfs2/recovery.c index b45aa3032ca2a9..5b1f494dc3c054 100644 --- a/fs/gfs2/recovery.c +++ b/fs/gfs2/recovery.c @@ -33,8 +33,8 @@ int gfs2_replay_read_block(struct gfs2_jdesc *jd, unsigned int blk, struct buffer_head **bh) { struct inode *inode = jd->jd_inode; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); - struct gfs2_glock *gl = ip->i_gl; u64 dblock; u32 extlen; int error; @@ -307,15 +307,13 @@ static int update_statfs_inode(struct gfs2_jdesc *jd, struct inode *inode) { struct gfs2_sbd *sdp = GFS2_SB(jd->jd_inode); - struct gfs2_inode *ip; struct buffer_head *bh; struct gfs2_statfs_change_host sc; int error = 0; BUG_ON(!inode); - ip = GFS2_I(inode); - error = gfs2_meta_inode_buffer(ip, &bh); + error = gfs2_meta_inode_buffer(GFS2_I(inode), &bh); if (error) goto out; @@ -346,7 +344,7 @@ static int update_statfs_inode(struct gfs2_jdesc *jd, mark_buffer_dirty(bh); brelse(bh); - gfs2_inode_metasync(ip->i_gl); + gfs2_inode_metasync(gfs2_inode_glock(inode)); out: return error; @@ -399,7 +397,7 @@ static void recover_local_statfs(struct gfs2_jdesc *jd, void gfs2_recover_func(struct work_struct *work) { struct gfs2_jdesc *jd = container_of(work, struct gfs2_jdesc, jd_work); - struct gfs2_inode *ip = GFS2_I(jd->jd_inode); + struct gfs2_glock *gl = gfs2_inode_glock(jd->jd_inode); struct gfs2_sbd *sdp = GFS2_SB(jd->jd_inode); struct gfs2_log_header_host head; struct gfs2_holder j_gh, ji_gh; @@ -441,7 +439,7 @@ void gfs2_recover_func(struct work_struct *work) goto fail; } - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_RECOVER | GL_NOCACHE, &ji_gh); if (error) diff --git a/fs/gfs2/rgrp.c b/fs/gfs2/rgrp.c index 5988a165a8301d..f6048a73e5c33a 100644 --- a/fs/gfs2/rgrp.c +++ b/fs/gfs2/rgrp.c @@ -1033,8 +1033,8 @@ static int gfs2_ri_update(struct gfs2_inode *ip) int gfs2_rindex_update(struct gfs2_sbd *sdp) { + struct gfs2_glock *gl = gfs2_inode_glock(sdp->sd_rindex); struct gfs2_inode *ip = GFS2_I(sdp->sd_rindex); - struct gfs2_glock *gl = ip->i_gl; struct gfs2_holder ri_gh; int error = 0; int unlock_required = 0; @@ -2453,7 +2453,7 @@ int gfs2_alloc_blocks(struct gfs2_inode *ip, u64 *bn, unsigned int *nblocks, if (error == 0) { struct gfs2_dinode *di = (struct gfs2_dinode *)dibh->b_data; - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gfs2_inode_glock(&ip->i_inode), dibh); di->di_goal_meta = di->di_goal_data = cpu_to_be64(ip->i_goal); brelse(dibh); diff --git a/fs/gfs2/super.c b/fs/gfs2/super.c index 06302c29340f2b..af8c715768336a 100644 --- a/fs/gfs2/super.c +++ b/fs/gfs2/super.c @@ -132,8 +132,7 @@ int gfs2_jdesc_check(struct gfs2_jdesc *jd) int gfs2_make_fs_rw(struct gfs2_sbd *sdp) { - struct gfs2_inode *ip = GFS2_I(sdp->sd_jdesc->jd_inode); - struct gfs2_glock *j_gl = ip->i_gl; + struct gfs2_glock *j_gl = gfs2_inode_glock(sdp->sd_jdesc->jd_inode); int error; j_gl->gl_ops->go_inval(j_gl, DIO_METADATA); @@ -176,6 +175,7 @@ void gfs2_statfs_change_out(const struct gfs2_statfs_change_host *sc, void *buf) int gfs2_statfs_init(struct gfs2_sbd *sdp) { + struct gfs2_glock *gl = gfs2_inode_glock(sdp->sd_statfs_inode); struct gfs2_inode *m_ip = GFS2_I(sdp->sd_statfs_inode); struct gfs2_statfs_change_host *m_sc = &sdp->sd_statfs_master; struct gfs2_statfs_change_host *l_sc = &sdp->sd_statfs_local; @@ -183,7 +183,7 @@ int gfs2_statfs_init(struct gfs2_sbd *sdp) struct gfs2_holder gh; int error; - error = gfs2_glock_nq_init(m_ip->i_gl, LM_ST_EXCLUSIVE, GL_NOCACHE, + error = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, GL_NOCACHE, &gh); if (error) return error; @@ -216,13 +216,13 @@ int gfs2_statfs_init(struct gfs2_sbd *sdp) void gfs2_statfs_change(struct gfs2_sbd *sdp, s64 total, s64 free, s64 dinodes) { - struct gfs2_inode *l_ip = GFS2_I(sdp->sd_sc_inode); + struct gfs2_glock *gl = gfs2_inode_glock(sdp->sd_sc_inode); struct gfs2_statfs_change_host *l_sc = &sdp->sd_statfs_local; struct gfs2_statfs_change_host *m_sc = &sdp->sd_statfs_master; s64 x, y; int need_sync = 0; - gfs2_trans_add_meta(l_ip->i_gl, sdp->sd_sc_bh); + gfs2_trans_add_meta(gl, sdp->sd_sc_bh); spin_lock(&sdp->sd_statfs_spin); l_sc->sc_total += total; @@ -244,13 +244,13 @@ void gfs2_statfs_change(struct gfs2_sbd *sdp, s64 total, s64 free, void update_statfs(struct gfs2_sbd *sdp, struct buffer_head *m_bh) { - struct gfs2_inode *m_ip = GFS2_I(sdp->sd_statfs_inode); - struct gfs2_inode *l_ip = GFS2_I(sdp->sd_sc_inode); + struct gfs2_glock *m_gl = gfs2_inode_glock(sdp->sd_statfs_inode); + struct gfs2_glock *l_gl = gfs2_inode_glock(sdp->sd_sc_inode); struct gfs2_statfs_change_host *m_sc = &sdp->sd_statfs_master; struct gfs2_statfs_change_host *l_sc = &sdp->sd_statfs_local; - gfs2_trans_add_meta(l_ip->i_gl, sdp->sd_sc_bh); - gfs2_trans_add_meta(m_ip->i_gl, m_bh); + gfs2_trans_add_meta(l_gl, sdp->sd_sc_bh); + gfs2_trans_add_meta(m_gl, m_bh); spin_lock(&sdp->sd_statfs_spin); m_sc->sc_total += l_sc->sc_total; @@ -273,8 +273,8 @@ int gfs2_statfs_sync(struct super_block *sb, int type) struct buffer_head *m_bh; int error; - error = gfs2_glock_nq_init(m_ip->i_gl, LM_ST_EXCLUSIVE, GL_NOCACHE, - &gh); + error = gfs2_glock_nq_init(gfs2_inode_glock(&m_ip->i_inode), + LM_ST_EXCLUSIVE, GL_NOCACHE, &gh); if (error) goto out; @@ -323,7 +323,6 @@ struct lfcc { static int gfs2_lock_fs_check_clean(struct gfs2_sbd *sdp) { - struct gfs2_inode *ip; struct gfs2_jdesc *jd; struct lfcc *lfcc; LIST_HEAD(list); @@ -336,13 +335,14 @@ static int gfs2_lock_fs_check_clean(struct gfs2_sbd *sdp) */ list_for_each_entry(jd, &sdp->sd_jindex_list, jd_list) { + struct gfs2_glock *gl = gfs2_inode_glock(jd->jd_inode); + lfcc = kmalloc_obj(struct lfcc); if (!lfcc) { error = -ENOMEM; goto out; } - ip = GFS2_I(jd->jd_inode); - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, 0, &lfcc->gh); + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, 0, &lfcc->gh); if (error) { kfree(lfcc); goto out; @@ -438,15 +438,16 @@ void gfs2_dinode_out(const struct gfs2_inode *ip, void *buf) static int gfs2_write_inode(struct inode *inode, struct writeback_control *wbc) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); - struct address_space *metamapping = gfs2_glock2aspace(ip->i_gl); + struct address_space *metamapping = gfs2_glock2aspace(gl); struct backing_dev_info *bdi = inode_to_bdi(metamapping->host); int ret = 0; bool flush_all = (wbc->sync_mode == WB_SYNC_ALL || gfs2_is_jdata(ip)); if (flush_all) - gfs2_log_flush(GFS2_SB(inode), ip->i_gl, + gfs2_log_flush(GFS2_SB(inode), gl, GFS2_LOG_HEAD_FLUSH_NORMAL | GFS2_LFC_WRITE_INODE); if (bdi_wb_dirty_exceeded(bdi)) @@ -481,6 +482,7 @@ static int gfs2_write_inode(struct inode *inode, struct writeback_control *wbc) static void gfs2_dirty_inode(struct inode *inode, int flags) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct buffer_head *bh; @@ -490,20 +492,20 @@ static void gfs2_dirty_inode(struct inode *inode, int flags) int ret; /* This can only happen during incomplete inode creation. */ - if (unlikely(!ip->i_gl)) + if (unlikely(!gl)) return; if (gfs2_withdrawn(sdp)) return; - if (!gfs2_glock_is_locked_by_me(ip->i_gl)) { - ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh); + if (!gfs2_glock_is_locked_by_me(gl)) { + ret = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, 0, &gh); if (ret) { fs_err(sdp, "dirty_inode: glock %d\n", ret); - gfs2_dump_glock(NULL, ip->i_gl, true); + gfs2_dump_glock(NULL, gl, true); return; } need_unlock = 1; - } else if (WARN_ON_ONCE(ip->i_gl->gl_state != LM_ST_EXCLUSIVE)) + } else if (WARN_ON_ONCE(gl->gl_state != LM_ST_EXCLUSIVE)) return; if (current->journal_info == NULL) { @@ -517,7 +519,7 @@ static void gfs2_dirty_inode(struct inode *inode, int flags) ret = gfs2_meta_inode_buffer(ip, &bh); if (ret == 0) { - gfs2_trans_add_meta(ip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); gfs2_dinode_out(ip, bh->b_data); brelse(bh); } @@ -1176,6 +1178,7 @@ static void gfs2_glock_put_eventually(struct gfs2_glock *gl) static enum evict_behavior gfs2_upgrade_iopen_glock(struct inode *inode) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_sbd *sdp = GFS2_SB(inode); struct gfs2_holder *gh = &ip->i_iopen_gh; @@ -1211,11 +1214,11 @@ static enum evict_behavior gfs2_upgrade_iopen_glock(struct inode *inode) wait_event_interruptible_timeout(sdp->sd_async_glock_wait, !test_bit(HIF_WAIT, &gh->gh_iflags) || - glock_needs_demote(ip->i_gl), + glock_needs_demote(gl), 5 * HZ); if (!test_bit(HIF_HOLDER, &gh->gh_iflags)) { gfs2_glock_dq(gh); - if (glock_needs_demote(ip->i_gl)) + if (glock_needs_demote(gl)) return EVICT_SHOULD_SKIP_DELETE; return EVICT_SHOULD_DEFER_DELETE; } @@ -1238,6 +1241,7 @@ static enum evict_behavior gfs2_upgrade_iopen_glock(struct inode *inode) static enum evict_behavior evict_should_delete(struct inode *inode, struct gfs2_holder *gh) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct super_block *sb = inode->i_sb; struct gfs2_sbd *sdp = sb->s_fs_info; @@ -1255,11 +1259,11 @@ static enum evict_behavior evict_should_delete(struct inode *inode, return EVICT_SHOULD_DEFER_DELETE; /* Must not read inode block until block type has been verified */ - ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, GL_SKIP, gh); + ret = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, GL_SKIP, gh); if (unlikely(ret)) return EVICT_SHOULD_SKIP_DELETE; - if (gfs2_inode_already_deleted(ip->i_gl, ip->i_no_formal_ino)) + if (gfs2_inode_already_deleted(gl, ip->i_no_formal_ino)) return EVICT_SHOULD_SKIP_DELETE; ret = gfs2_check_blk_type(sdp, ip->i_no_addr, GFS2_BLKST_UNLINKED); if (ret) @@ -1288,8 +1292,8 @@ static enum evict_behavior evict_should_delete(struct inode *inode, */ static int evict_unlinked_inode(struct inode *inode, struct gfs2_holder *gh) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); - struct gfs2_glock *gl = ip->i_gl; int ret; /* The inode glock must be held exclusively and be instantiated. */ @@ -1390,8 +1394,7 @@ static int evict_linked_inode(struct inode *inode, struct gfs2_holder *gh) { struct super_block *sb = inode->i_sb; struct gfs2_sbd *sdp = sb->s_fs_info; - struct gfs2_inode *ip = GFS2_I(inode); - struct gfs2_glock *gl = ip->i_gl; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct address_space *metamapping = gfs2_glock2aspace(gl); int ret; @@ -1446,13 +1449,14 @@ static void gfs2_evict_inode(struct inode *inode) { struct super_block *sb = inode->i_sb; struct gfs2_sbd *sdp = sb->s_fs_info; + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder gh; enum evict_behavior behavior; int ret; gfs2_holder_mark_uninitialized(&gh); - if (sb_rdonly(sb) || !ip->i_no_addr || !ip->i_gl) + if (sb_rdonly(sb) || !ip->i_no_addr || !gl) goto out; /* @@ -1505,10 +1509,10 @@ static void gfs2_evict_inode(struct inode *inode) gfs2_glock_dq_uninit(&ip->i_iopen_gh); gfs2_glock_put_eventually(gl); } - if (ip->i_gl) { - glock_clear_object(ip->i_gl, ip); + if (gl) { + glock_clear_object(gl, ip); wait_on_bit_io(&ip->i_flags, GIF_GLOP_PENDING, TASK_UNINTERRUPTIBLE); - gfs2_glock_put_eventually(ip->i_gl); + gfs2_glock_put_eventually(gl); rcu_assign_pointer(ip->i_gl, NULL); } } diff --git a/fs/gfs2/util.c b/fs/gfs2/util.c index 83b8bb6446e57b..61b0668ecabc08 100644 --- a/fs/gfs2/util.c +++ b/fs/gfs2/util.c @@ -55,10 +55,9 @@ int check_journal_clean(struct gfs2_sbd *sdp, struct gfs2_jdesc *jd, int error; struct gfs2_holder j_gh; struct gfs2_log_header_host head; - struct gfs2_inode *ip; + struct gfs2_glock *gl = gfs2_inode_glock(jd->jd_inode); - ip = GFS2_I(jd->jd_inode); - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_RECOVER | + error = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_RECOVER | GL_EXACT | GL_NOCACHE, &j_gh); if (error) { if (verbose) @@ -333,6 +332,7 @@ void gfs2_consist_i(struct gfs2_sbd *sdp, const char *function, void gfs2_consist_inode_i(struct gfs2_inode *ip, const char *function, char *file, unsigned int line) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); gfs2_lm(sdp, @@ -342,7 +342,7 @@ void gfs2_consist_inode_i(struct gfs2_inode *ip, (unsigned long long)ip->i_no_formal_ino, (unsigned long long)ip->i_no_addr, function, file, line); - gfs2_dump_glock(NULL, ip->i_gl, 1); + gfs2_dump_glock(NULL, gl, 1); gfs2_withdraw(sdp); } diff --git a/fs/gfs2/xattr.c b/fs/gfs2/xattr.c index b9f48d6f10a97a..db38d972debd04 100644 --- a/fs/gfs2/xattr.c +++ b/fs/gfs2/xattr.c @@ -128,11 +128,12 @@ static int ea_foreach_i(struct gfs2_inode *ip, struct buffer_head *bh, static int ea_foreach(struct gfs2_inode *ip, ea_call_t ea_call, void *data) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct buffer_head *bh, *eabh; __be64 *eablk, *end; int error; - error = gfs2_meta_read(ip->i_gl, ip->i_eattr, DIO_WAIT, 0, &bh); + error = gfs2_meta_read(gl, ip->i_eattr, DIO_WAIT, 0, &bh); if (error) return error; @@ -156,7 +157,7 @@ static int ea_foreach(struct gfs2_inode *ip, ea_call_t ea_call, void *data) break; bn = be64_to_cpu(*eablk); - error = gfs2_meta_read(ip->i_gl, bn, DIO_WAIT, 0, &eabh); + error = gfs2_meta_read(gl, bn, DIO_WAIT, 0, &eabh); if (error) break; error = ea_foreach_i(ip, eabh, ea_call, data); @@ -279,7 +280,7 @@ static int ea_dealloc_unstuffed(struct gfs2_inode *ip, struct buffer_head *bh, if (error) goto out_gunlock; - gfs2_trans_add_meta(ip->i_gl, bh); + gfs2_trans_add_meta(gfs2_inode_glock(&ip->i_inode), bh); dataptrs = GFS2_EA2DATAPTRS(ea); for (x = 0; x < ea->ea_num_ptrs; x++, dataptrs++) { @@ -426,7 +427,8 @@ ssize_t gfs2_listxattr(struct dentry *dentry, char *buffer, size_t size) er.er_data_len = size; } - error = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY, &i_gh); + error = gfs2_glock_nq_init(gfs2_inode_glock(&ip->i_inode), LM_ST_SHARED, + LM_FLAG_ANY, &i_gh); if (error) return error; @@ -457,6 +459,7 @@ ssize_t gfs2_listxattr(struct dentry *dentry, char *buffer, size_t size) static int gfs2_iter_unstuffed(struct gfs2_inode *ip, struct gfs2_ea_header *ea, const char *din, char *dout) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); struct buffer_head **bh; unsigned int amount = GFS2_EA_DATA_LEN(ea); @@ -472,7 +475,7 @@ static int gfs2_iter_unstuffed(struct gfs2_inode *ip, struct gfs2_ea_header *ea, return -ENOMEM; for (x = 0; x < nptrs; x++) { - error = gfs2_meta_read(ip->i_gl, be64_to_cpu(*dataptrs), 0, 0, + error = gfs2_meta_read(gl, be64_to_cpu(*dataptrs), 0, 0, bh + x); if (error) { while (x--) @@ -505,7 +508,7 @@ static int gfs2_iter_unstuffed(struct gfs2_inode *ip, struct gfs2_ea_header *ea, } if (din) { - gfs2_trans_add_meta(ip->i_gl, bh[x]); + gfs2_trans_add_meta(gl, bh[x]); memcpy(pos, din, cp_size); din += sdp->sd_jbsize; } @@ -608,14 +611,14 @@ static int gfs2_xattr_get(const struct xattr_handler *handler, struct dentry *unused, struct inode *inode, const char *name, void *buffer, size_t size) { - struct gfs2_inode *ip = GFS2_I(inode); + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_holder gh; int ret; /* During lookup, SELinux calls this function with the glock locked. */ - if (!gfs2_glock_is_locked_by_me(ip->i_gl)) { - ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_SHARED, LM_FLAG_ANY, &gh); + if (!gfs2_glock_is_locked_by_me(gl)) { + ret = gfs2_glock_nq_init(gl, LM_ST_SHARED, LM_FLAG_ANY, &gh); if (ret) return ret; } else { @@ -637,6 +640,7 @@ static int gfs2_xattr_get(const struct xattr_handler *handler, static int ea_alloc_blk(struct gfs2_inode *ip, struct buffer_head **bhp) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); struct gfs2_ea_header *ea; unsigned int n = 1; @@ -647,8 +651,8 @@ static int ea_alloc_blk(struct gfs2_inode *ip, struct buffer_head **bhp) if (error) return error; gfs2_trans_remove_revoke(sdp, block, 1); - *bhp = gfs2_meta_new(ip->i_gl, block); - gfs2_trans_add_meta(ip->i_gl, *bhp); + *bhp = gfs2_meta_new(gl, block); + gfs2_trans_add_meta(gl, *bhp); gfs2_metatype_set(*bhp, GFS2_METATYPE_EA, GFS2_FORMAT_EA); gfs2_buffer_clear_tail(*bhp, sizeof(struct gfs2_meta_header)); @@ -678,6 +682,7 @@ static int ea_alloc_blk(struct gfs2_inode *ip, struct buffer_head **bhp) static int ea_write(struct gfs2_inode *ip, struct gfs2_ea_header *ea, struct gfs2_ea_request *er) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); int error; @@ -709,8 +714,8 @@ static int ea_write(struct gfs2_inode *ip, struct gfs2_ea_header *ea, if (error) return error; gfs2_trans_remove_revoke(sdp, block, 1); - bh = gfs2_meta_new(ip->i_gl, block); - gfs2_trans_add_meta(ip->i_gl, bh); + bh = gfs2_meta_new(gl, block); + gfs2_trans_add_meta(gl, bh); gfs2_metatype_set(bh, GFS2_METATYPE_ED, GFS2_FORMAT_ED); gfs2_add_inode_blocks(&ip->i_inode, 1); @@ -841,11 +846,12 @@ static struct gfs2_ea_header *ea_split_ea(struct gfs2_ea_header *ea) static void ea_set_remove_stuffed(struct gfs2_inode *ip, struct gfs2_ea_location *el) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_ea_header *ea = el->el_ea; struct gfs2_ea_header *prev = el->el_prev; u32 len; - gfs2_trans_add_meta(ip->i_gl, el->el_bh); + gfs2_trans_add_meta(gl, el->el_bh); if (!prev || !GFS2_EA_IS_STUFFED(ea)) { ea->ea_type = GFS2_EATYPE_UNUSED; @@ -875,6 +881,7 @@ struct ea_set { static int ea_set_simple_noalloc(struct gfs2_inode *ip, struct buffer_head *bh, struct gfs2_ea_header *ea, struct ea_set *es) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_ea_request *er = es->es_er; int error; @@ -882,7 +889,7 @@ static int ea_set_simple_noalloc(struct gfs2_inode *ip, struct buffer_head *bh, if (error) return error; - gfs2_trans_add_meta(ip->i_gl, bh); + gfs2_trans_add_meta(gl, bh); if (es->ea_split) ea = ea_split_ea(ea); @@ -902,11 +909,12 @@ static int ea_set_simple_noalloc(struct gfs2_inode *ip, struct buffer_head *bh, static int ea_set_simple_alloc(struct gfs2_inode *ip, struct gfs2_ea_request *er, void *private) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct ea_set *es = private; struct gfs2_ea_header *ea = es->es_ea; int error; - gfs2_trans_add_meta(ip->i_gl, es->es_bh); + gfs2_trans_add_meta(gl, es->es_bh); if (es->ea_split) ea = ea_split_ea(ea); @@ -971,6 +979,7 @@ static int ea_set_simple(struct gfs2_inode *ip, struct buffer_head *bh, static int ea_set_block(struct gfs2_inode *ip, struct gfs2_ea_request *er, void *private) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); struct buffer_head *indbh, *newbh; __be64 *eablk; @@ -980,7 +989,7 @@ static int ea_set_block(struct gfs2_inode *ip, struct gfs2_ea_request *er, if (ip->i_diskflags & GFS2_DIF_EA_INDIRECT) { __be64 *end; - error = gfs2_meta_read(ip->i_gl, ip->i_eattr, DIO_WAIT, 0, + error = gfs2_meta_read(gl, ip->i_eattr, DIO_WAIT, 0, &indbh); if (error) return error; @@ -1002,7 +1011,7 @@ static int ea_set_block(struct gfs2_inode *ip, struct gfs2_ea_request *er, goto out; } - gfs2_trans_add_meta(ip->i_gl, indbh); + gfs2_trans_add_meta(gl, indbh); } else { u64 blk; unsigned int n = 1; @@ -1010,8 +1019,8 @@ static int ea_set_block(struct gfs2_inode *ip, struct gfs2_ea_request *er, if (error) return error; gfs2_trans_remove_revoke(sdp, blk, 1); - indbh = gfs2_meta_new(ip->i_gl, blk); - gfs2_trans_add_meta(ip->i_gl, indbh); + indbh = gfs2_meta_new(gl, blk); + gfs2_trans_add_meta(gl, indbh); gfs2_metatype_set(indbh, GFS2_METATYPE_IN, GFS2_FORMAT_IN); gfs2_buffer_clear_tail(indbh, mh_size); @@ -1088,6 +1097,7 @@ static int ea_set_remove_unstuffed(struct gfs2_inode *ip, static int ea_remove_stuffed(struct gfs2_inode *ip, struct gfs2_ea_location *el) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_ea_header *ea = el->el_ea; struct gfs2_ea_header *prev = el->el_prev; int error; @@ -1096,7 +1106,7 @@ static int ea_remove_stuffed(struct gfs2_inode *ip, struct gfs2_ea_location *el) if (error) return error; - gfs2_trans_add_meta(ip->i_gl, el->el_bh); + gfs2_trans_add_meta(gl, el->el_bh); if (prev) { u32 len; @@ -1234,6 +1244,7 @@ static int gfs2_xattr_set(const struct xattr_handler *handler, const char *name, const void *value, size_t size, int flags) { + struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); struct gfs2_holder gh; int ret; @@ -1244,12 +1255,12 @@ static int gfs2_xattr_set(const struct xattr_handler *handler, /* May be called from gfs_setattr with the glock locked. */ - if (!gfs2_glock_is_locked_by_me(ip->i_gl)) { - ret = gfs2_glock_nq_init(ip->i_gl, LM_ST_EXCLUSIVE, 0, &gh); + if (!gfs2_glock_is_locked_by_me(gl)) { + ret = gfs2_glock_nq_init(gl, LM_ST_EXCLUSIVE, 0, &gh); if (ret) goto out; } else { - if (WARN_ON_ONCE(ip->i_gl->gl_state != LM_ST_EXCLUSIVE)) { + if (WARN_ON_ONCE(gl->gl_state != LM_ST_EXCLUSIVE)) { ret = -EIO; goto out; } @@ -1265,6 +1276,7 @@ static int gfs2_xattr_set(const struct xattr_handler *handler, static int ea_dealloc_indirect(struct gfs2_inode *ip) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); struct gfs2_rgrp_list rlist; struct gfs2_rgrpd *rgd; @@ -1283,7 +1295,7 @@ static int ea_dealloc_indirect(struct gfs2_inode *ip) memset(&rlist, 0, sizeof(struct gfs2_rgrp_list)); - error = gfs2_meta_read(ip->i_gl, ip->i_eattr, DIO_WAIT, 0, &indbh); + error = gfs2_meta_read(gl, ip->i_eattr, DIO_WAIT, 0, &indbh); if (error) return error; @@ -1333,7 +1345,7 @@ static int ea_dealloc_indirect(struct gfs2_inode *ip) if (error) goto out_gunlock; - gfs2_trans_add_meta(ip->i_gl, indbh); + gfs2_trans_add_meta(gl, indbh); eablk = (__be64 *)(indbh->b_data + sizeof(struct gfs2_meta_header)); bstart = 0; @@ -1367,7 +1379,7 @@ static int ea_dealloc_indirect(struct gfs2_inode *ip) error = gfs2_meta_inode_buffer(ip, &dibh); if (!error) { - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); gfs2_dinode_out(ip, dibh->b_data); brelse(dibh); } @@ -1385,6 +1397,7 @@ static int ea_dealloc_indirect(struct gfs2_inode *ip) static int ea_dealloc_block(struct gfs2_inode *ip, bool initialized) { + struct gfs2_glock *gl = gfs2_inode_glock(&ip->i_inode); struct gfs2_sbd *sdp = GFS2_SB(&ip->i_inode); struct gfs2_rgrpd *rgd; struct buffer_head *dibh; @@ -1419,7 +1432,7 @@ static int ea_dealloc_block(struct gfs2_inode *ip, bool initialized) if (initialized) { error = gfs2_meta_inode_buffer(ip, &dibh); if (!error) { - gfs2_trans_add_meta(ip->i_gl, dibh); + gfs2_trans_add_meta(gl, dibh); gfs2_dinode_out(ip, dibh->b_data); brelse(dibh); } From 727b6fb855d5c456b5d51719f892ac7f8faffbff Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Thu, 20 Aug 2026 16:00:56 +0200 Subject: [PATCH 170/857] gfs2: annotate i_gl with __rcu ip->i_gl is always set over the lifetime of a gfs2 inode; it is initialized at inode create time and torn down during evict. However, gfs2_permission() can be called on an inode in non-blocking mode, when the inode may already be undergoing evict. So far, to make that work, we have been using rcu_dereference_check() in gfs2_permission() and rcu_assign_pointer() in gfs2_evict_inode(). However, ip->i_gl wasn't marked as __rcu so far. rcu_dereference_check() and rcu_assign_pointer() expect __rcu pointer arguments, and sparse complains when regular pointers are passed to those functions: fs/gfs2/super.c:1516:17: error: incompatible types in comparison expression (different address spaces): fs/gfs2/super.c:1516:17: struct gfs2_glock [noderef] __rcu * fs/gfs2/super.c:1516:17: struct gfs2_glock * fs/gfs2/inode.c:1988:14: error: incompatible types in comparison expression (different address spaces): fs/gfs2/inode.c:1988:14: struct gfs2_glock [noderef] __rcu * fs/gfs2/inode.c:1988:14: struct gfs2_glock * To fix those errors, turn ip->i_gl into a __rcu variable. Use rcu_assign_pointer() to assign to it, and rcu_dereference_protected() to access it when the pointer is known to be valid. Make sure not to use gfs2_inode_glock() in gfs2_permission(). Based on a patch from Adrian Garcia Casado . Signed-off-by: Andreas Gruenbacher --- fs/gfs2/incore.h | 4 ++-- fs/gfs2/inode.c | 8 ++++---- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/fs/gfs2/incore.h b/fs/gfs2/incore.h index 3ae8e2be486c90..0a48f8e8ed6c5b 100644 --- a/fs/gfs2/incore.h +++ b/fs/gfs2/incore.h @@ -391,7 +391,7 @@ struct gfs2_inode { u64 i_generation; u64 i_eattr; unsigned long i_flags; /* GIF_... */ - struct gfs2_glock *i_gl; + struct gfs2_glock __rcu *i_gl; struct gfs2_holder i_iopen_gh; struct gfs2_qadata *i_qadata; /* quota allocation data */ struct gfs2_holder i_rgd_gh; @@ -881,6 +881,6 @@ static inline unsigned gfs2_max_stuffed_size(const struct gfs2_inode *ip) static inline struct gfs2_glock *gfs2_inode_glock(struct inode *inode) { - return GFS2_I(inode)->i_gl; + return rcu_dereference_protected(GFS2_I(inode)->i_gl, 1); } #endif /* __INCORE_DOT_H__ */ diff --git a/fs/gfs2/inode.c b/fs/gfs2/inode.c index 5aaed0018fb338..d4e1e443391e0f 100644 --- a/fs/gfs2/inode.c +++ b/fs/gfs2/inode.c @@ -151,7 +151,7 @@ struct inode *gfs2_inode_lookup(struct super_block *sb, unsigned int type, &gl); if (unlikely(error)) goto fail; - ip->i_gl = gl; + rcu_assign_pointer(ip->i_gl, gl); error = gfs2_glock_get(sdp, no_addr, &gfs2_iopen_glops, CREATE, &io_gl); @@ -244,7 +244,7 @@ struct inode *gfs2_inode_lookup(struct super_block *sb, unsigned int type, gfs2_glock_dq_uninit(&i_gh); if (gl) { gfs2_glock_put(gl); - ip->i_gl = NULL; + rcu_assign_pointer(ip->i_gl, NULL); } iget_failed(inode); return ERR_PTR(error); @@ -840,7 +840,7 @@ static int gfs2_create_inode(struct inode *dir, struct dentry *dentry, error = gfs2_glock_get(sdp, ip->i_no_addr, &gfs2_inode_glops, CREATE, &gl); if (error) goto fail_dealloc_inode; - ip->i_gl = gl; + rcu_assign_pointer(ip->i_gl, gl); error = gfs2_glock_get(sdp, ip->i_no_addr, &gfs2_iopen_glops, CREATE, &io_gl); if (error) @@ -940,7 +940,7 @@ static int gfs2_create_inode(struct inode *dir, struct dentry *dentry, fail_free_inode: if (gl) { gfs2_glock_put(gl); - ip->i_gl = NULL; + rcu_assign_pointer(ip->i_gl, NULL); } gfs2_rs_deltree(&ip->i_res); gfs2_qa_put(ip); From 20623fbdd82062c827d62bd95cbc4c9200cbda89 Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Thu, 20 Aug 2026 15:15:35 +0200 Subject: [PATCH 171/857] gfs2: Silence sparse signedness warning Silence the following sparse warning: fs/gfs2/bmap.c:1718:42: warning: unsigned value that used to be signed checked against zero? fs/gfs2/bmap.c:2017:16: signed value source It's unclear to me what sparse is trying to tell me there, but making the height variables unsigned shuts it up. Signed-off-by: Andreas Gruenbacher --- fs/gfs2/bmap.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/gfs2/bmap.c b/fs/gfs2/bmap.c index b5ddd936f48925..3714364385c7cf 100644 --- a/fs/gfs2/bmap.c +++ b/fs/gfs2/bmap.c @@ -1699,7 +1699,7 @@ enum dealloc_states { }; static inline void -metapointer_range(struct metapath *mp, int height, +metapointer_range(struct metapath *mp, unsigned int height, __u16 *start_list, unsigned int start_aligned, __u16 *end_list, unsigned int end_aligned, __be64 **start, __be64 **end) @@ -1770,7 +1770,7 @@ static int punch_hole(struct gfs2_inode *ip, u64 offset, u64 length) unsigned int strip_h = ip->i_height - 1; u32 btotal = 0; int ret, state; - int mp_h; /* metapath buffers are read in to this height */ + unsigned int mp_h; /* metapath buffers are read in to this height */ u64 prev_bnr = 0; __be64 *start, *end; From ea31b5336369a76952767f1aa9f81a343de4d4ca Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Thu, 20 Aug 2026 23:39:40 +0200 Subject: [PATCH 172/857] gfs2: Improve resource group validation and error handling Validate the resource group boundaries, check against the device size, simplify the initialization logic, and check for resource group overlaps. Reported-by: syzbot+9d20c3ad7d29227de28d@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=9d20c3ad7d29227de28d Tested-by: syzbot+9d20c3ad7d29227de28d@syzkaller.appspotmail.com Signed-off-by: Andreas Gruenbacher --- fs/gfs2/rgrp.c | 123 ++++++++++++++++++++++++++----------------------- 1 file changed, 65 insertions(+), 58 deletions(-) diff --git a/fs/gfs2/rgrp.c b/fs/gfs2/rgrp.c index f6048a73e5c33a..3ac751eea05505 100644 --- a/fs/gfs2/rgrp.c +++ b/fs/gfs2/rgrp.c @@ -745,6 +745,7 @@ void gfs2_clear_rgrpd(struct gfs2_sbd *sdp) /** * compute_bitstructs - Compute the bitmap sizes + * @sb: The superblock * @rgd: The resource group descriptor * * Calculates bitmap descriptors, one for each block that contains bitmap data @@ -752,84 +753,74 @@ void gfs2_clear_rgrpd(struct gfs2_sbd *sdp) * Returns: errno */ -static int compute_bitstructs(struct gfs2_rgrpd *rgd) +static int compute_bitstructs(struct super_block *sb, struct gfs2_rgrpd *rgd) { struct gfs2_sbd *sdp = rgd->rd_sbd; struct gfs2_bitmap *bi; - u32 length = rgd->rd_length; /* # blocks in hdr & bitmap */ + u32 expected_length; u32 bytes_left, bytes; + u64 data_end; int x; - if (!length) - return -EINVAL; + /* + * The first resource group block has a gfs2_rgrp header; the remaining + * blocks have a gfs2_meta_header header. The rest of each block is + * filled with bitmap data. + */ + + if (rgd->rd_addr <= (GFS2_SB_ADDR >> sdp->sd_fsb2bb_shift)) { + gfs2_consist_rgrpd(rgd); + return -EIO; + } + if (check_add_overflow(rgd->rd_data0, rgd->rd_data, &data_end) || + rgd->rd_data == 0 || data_end > sb_bdev_nr_blocks(sb)) { + gfs2_consist_rgrpd(rgd); + return -EIO; + } + if (rgd->rd_bitbytes != DIV_ROUND_UP(rgd->rd_data, GFS2_NBBY)) { + gfs2_consist_rgrpd(rgd); + return -EIO; + } + expected_length = DIV_ROUND_UP(rgd->rd_bitbytes + + sizeof(struct gfs2_rgrp) - sizeof(struct gfs2_meta_header), + sdp->sd_sb.sb_bsize - sizeof(struct gfs2_meta_header)); + if (rgd->rd_length != expected_length) { + gfs2_consist_rgrpd(rgd); + return -EIO; + } + if (rgd->rd_data0 < rgd->rd_addr + rgd->rd_length) { + gfs2_consist_rgrpd(rgd); + return -EIO; + } - rgd->rd_bits = kzalloc_objs(struct gfs2_bitmap, length, GFP_NOFS); + rgd->rd_bits = kzalloc_objs(struct gfs2_bitmap, rgd->rd_length, GFP_NOFS); if (!rgd->rd_bits) return -ENOMEM; bytes_left = rgd->rd_bitbytes; - for (x = 0; x < length; x++) { + for (x = 0; x < rgd->rd_length; x++) { bi = rgd->rd_bits + x; bi->bi_flags = 0; - /* small rgrp; bitmap stored completely in header block */ - if (length == 1) { - bytes = bytes_left; - bi->bi_offset = sizeof(struct gfs2_rgrp); + if (x == 0) { + /* header block */ bi->bi_start = 0; - bi->bi_bytes = bytes; - bi->bi_blocks = bytes * GFS2_NBBY; - /* header block */ - } else if (x == 0) { - bytes = sdp->sd_sb.sb_bsize - sizeof(struct gfs2_rgrp); bi->bi_offset = sizeof(struct gfs2_rgrp); - bi->bi_start = 0; - bi->bi_bytes = bytes; - bi->bi_blocks = bytes * GFS2_NBBY; - /* last block */ - } else if (x + 1 == length) { - bytes = bytes_left; - bi->bi_offset = sizeof(struct gfs2_meta_header); - bi->bi_start = rgd->rd_bitbytes - bytes_left; - bi->bi_bytes = bytes; - bi->bi_blocks = bytes * GFS2_NBBY; - /* other blocks */ } else { - bytes = sdp->sd_sb.sb_bsize - - sizeof(struct gfs2_meta_header); + /* bitmap-only block */ + struct gfs2_bitmap *prev = bi - 1; + + bi->bi_start = prev->bi_start + prev->bi_bytes; bi->bi_offset = sizeof(struct gfs2_meta_header); - bi->bi_start = rgd->rd_bitbytes - bytes_left; - bi->bi_bytes = bytes; - bi->bi_blocks = bytes * GFS2_NBBY; } - + bytes = sdp->sd_sb.sb_bsize - bi->bi_offset; + if (bytes > bytes_left) + bytes = bytes_left; + bi->bi_bytes = bytes; + bi->bi_blocks = bytes * GFS2_NBBY; bytes_left -= bytes; } - - if (bytes_left) { - gfs2_consist_rgrpd(rgd); - return -EIO; - } - bi = rgd->rd_bits + (length - 1); - if ((bi->bi_start + bi->bi_bytes) * GFS2_NBBY != rgd->rd_data) { - gfs2_lm(sdp, - "ri_addr=%llu " - "ri_length=%u " - "ri_data0=%llu " - "ri_data=%u " - "ri_bitbytes=%u " - "start=%u len=%u offset=%u\n", - (unsigned long long)rgd->rd_addr, - rgd->rd_length, - (unsigned long long)rgd->rd_data0, - rgd->rd_data, - rgd->rd_bitbytes, - bi->bi_start, bi->bi_bytes, bi->bi_offset); - gfs2_consist_rgrpd(rgd); - return -EIO; - } - return 0; } @@ -864,6 +855,7 @@ static int rgd_insert(struct gfs2_rgrpd *rgd) { struct gfs2_sbd *sdp = rgd->rd_sbd; struct rb_node **newn = &sdp->sd_rindex_tree.rb_node, *parent = NULL; + struct rb_node *prevn; /* Figure out where to put new node */ while (*newn) { @@ -882,6 +874,19 @@ static int rgd_insert(struct gfs2_rgrpd *rgd) rb_link_node(&rgd->rd_node, parent, newn); rb_insert_color(&rgd->rd_node, &sdp->sd_rindex_tree); sdp->sd_rgrps++; + + prevn = rb_prev(&rgd->rd_node); + if (prevn) { + struct gfs2_rgrpd *prev = + rb_entry(prevn, struct gfs2_rgrpd, rd_node); + + if (prev->rd_data0 + prev->rd_data > rgd->rd_addr) { + fs_err(sdp, "overlapping resource groups.\n"); + rb_erase(&rgd->rd_node, &sdp->sd_rindex_tree); + return -ENOENT; + } + } + return 0; } @@ -928,7 +933,7 @@ static int read_rindex_entry(struct gfs2_inode *ip) if (error) goto fail; - error = compute_bitstructs(rgd); + error = compute_bitstructs(sdp->sd_vfs, rgd); if (error) goto fail_glock; @@ -944,7 +949,9 @@ static int read_rindex_entry(struct gfs2_inode *ip) return 0; } - error = 0; /* someone else read in the rgrp; free it and ignore it */ + /* If someone else read in the rgrp, free it and ignore it. */ + if (error == -EEXIST) + error = 0; fail_glock: gfs2_glock_put(rgd->rd_gl); From 1c08039e3887fafc17dad527333938a379d38f34 Mon Sep 17 00:00:00 2001 From: Hyunwoo Kim Date: Fri, 20 Mar 2026 00:14:58 +0900 Subject: [PATCH 173/857] Bluetooth: RFCOMM: Validate MTU in rfcomm_apply_pn() to prevent infinite loop rfcomm_apply_pn() accepts the MTU value from a remote PN (Parameter Negotiation) frame without checking for zero. When the remote peer sends an MTU of zero, d->mtu is set to 0. This causes the sendmsg path to enter an infinite loop when fragmenting data, as each fragment has size == min_t(size_t, len, 0) == 0, so the remaining length never decreases. The infinite allocation of zero-length skbs exhausts all system memory. Fix by clamping d->mtu to RFCOMM_DEFAULT_MTU when the negotiated value is zero, consistent with the initial value assigned in rfcomm_dlc_alloc(). Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Signed-off-by: Hyunwoo Kim Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/rfcomm/core.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/net/bluetooth/rfcomm/core.c b/net/bluetooth/rfcomm/core.c index 9cdfea666a2c6a..0e496b85e6cec8 100644 --- a/net/bluetooth/rfcomm/core.c +++ b/net/bluetooth/rfcomm/core.c @@ -1455,6 +1455,10 @@ static int rfcomm_apply_pn(struct rfcomm_dlc *d, int cr, struct rfcomm_pn *pn) d->mtu = __le16_to_cpu(pn->mtu); + /* MTU 0 causes an infinite loop when fragmenting in sendmsg */ + if (!d->mtu) + d->mtu = RFCOMM_DEFAULT_MTU; + if (cr && d->mtu > s->mtu) d->mtu = s->mtu; From b8d936018ce2b029cd307056078a7802e9420b93 Mon Sep 17 00:00:00 2001 From: Gongwei Li Date: Fri, 21 Aug 2026 10:45:55 +0800 Subject: [PATCH 174/857] Bluetooth: hci_uart: Fix false success return in hci_uart_setup() When reading the local version information for vendor detection fails, the error is only printed and 0 is returned, which masks the setup failure from the HCI core. Return PTR_ERR(skb) instead. Fixes: fb2ce8d11f039 ("Bluetooth: hci_uart: Add support for vendor detection flag") Fixes: 82f5169bf3d3b ("Bluetooth: hci_uart: add serdev driver support library") Cc: stable@vger.kernel.org Signed-off-by: Gongwei Li Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/hci_ldisc.c | 2 +- drivers/bluetooth/hci_serdev.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/bluetooth/hci_ldisc.c b/drivers/bluetooth/hci_ldisc.c index 58f5504a336e3e..697e32122cc89e 100644 --- a/drivers/bluetooth/hci_ldisc.c +++ b/drivers/bluetooth/hci_ldisc.c @@ -457,7 +457,7 @@ static int hci_uart_setup(struct hci_dev *hdev) if (IS_ERR(skb)) { BT_ERR("%s: Reading local version information failed (%ld)", hdev->name, PTR_ERR(skb)); - return 0; + return PTR_ERR(skb); } if (skb->len != sizeof(*ver)) { diff --git a/drivers/bluetooth/hci_serdev.c b/drivers/bluetooth/hci_serdev.c index 13346c20559105..48ba017e219bc3 100644 --- a/drivers/bluetooth/hci_serdev.c +++ b/drivers/bluetooth/hci_serdev.c @@ -221,7 +221,7 @@ static int hci_uart_setup(struct hci_dev *hdev) if (IS_ERR(skb)) { bt_dev_err(hdev, "Reading local version info failed (%ld)", PTR_ERR(skb)); - return 0; + return PTR_ERR(skb); } if (skb->len != sizeof(*ver)) From 07230d21630f48e74831efda665a3200c645e9c2 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 10 Aug 2026 04:29:24 -0700 Subject: [PATCH 175/857] locking/csd-lock: Pack csd_lock_wait_toolong() state into a struct csd_lock_wait_toolong() has some fields and they are being expanded now, separate them into a structure, that can be easily digestible. This simplify the function aslo, given the fields were passed by reference, and the ts0/ts1 names say nothing about what the two timestamps hold. Pack them into struct csd_wait_state and name the timestamps for what they store, ts_start and ts_report. The local ts2 becomes ts_now. Reporting a further timestamp, such as next patch, then costs a struct member rather than another argument. No functional change. Suggested-by: Dmitry Ilvokhin Signed-off-by: Breno Leitao Reviewed-by: Dmitry Ilvokhin Signed-off-by: Paul E. McKenney --- kernel/smp.c | 56 +++++++++++++++++++++++++++++----------------------- 1 file changed, 31 insertions(+), 25 deletions(-) diff --git a/kernel/smp.c b/kernel/smp.c index a0bb56bd8ddadb..1bd8a24349e096 100644 --- a/kernel/smp.c +++ b/kernel/smp.c @@ -223,50 +223,58 @@ bool csd_lock_is_stuck(void) return !!atomic_read(&n_csd_lock_stuck); } +/* State that csd_lock_wait_toolong() carries across the __csd_lock_wait() loop. */ +struct csd_wait_state { + u64 ts_start; /* When the wait began. */ + u64 ts_report; /* When the last complaint was printed. */ + int bug_id; + unsigned long nmessages; +}; + /* * Complain if too much time spent waiting. Note that only * the CSD_TYPE_SYNC/ASYNC types provide the destination CPU, * so waiting on other types gets much less information. */ -static bool csd_lock_wait_toolong(call_single_data_t *csd, u64 ts0, u64 *ts1, int *bug_id, unsigned long *nmessages) +static bool csd_lock_wait_toolong(call_single_data_t *csd, struct csd_wait_state *state) { int cpu = -1; int cpux; bool firsttime; - u64 ts2, ts_delta; + u64 ts_now, ts_delta; call_single_data_t *cpu_cur_csd; unsigned int flags = READ_ONCE(csd->node.u_flags); unsigned long long csd_lock_timeout_ns = csd_lock_timeout * NSEC_PER_MSEC; if (!(flags & CSD_FLAG_LOCK)) { - if (!unlikely(*bug_id)) + if (!unlikely(state->bug_id)) return true; cpu = csd_lock_wait_getcpu(csd); pr_alert("csd: CSD lock (#%d) got unstuck on CPU#%02d, CPU#%02d released the lock.\n", - *bug_id, raw_smp_processor_id(), cpu); + state->bug_id, raw_smp_processor_id(), cpu); atomic_dec(&n_csd_lock_stuck); return true; } - ts2 = ktime_get_mono_fast_ns(); + ts_now = ktime_get_mono_fast_ns(); /* How long since we last checked for a stuck CSD lock.*/ - ts_delta = ts2 - *ts1; - if (likely(ts_delta <= csd_lock_timeout_ns * (*nmessages + 1) * - (!*nmessages ? 1 : (ilog2(num_online_cpus()) / 2 + 1)) || + ts_delta = ts_now - state->ts_report; + if (likely(ts_delta <= csd_lock_timeout_ns * (state->nmessages + 1) * + (!state->nmessages ? 1 : (ilog2(num_online_cpus()) / 2 + 1)) || csd_lock_timeout_ns == 0)) return false; - if (ts0 > ts2) { + if (state->ts_start > ts_now) { /* Our own sched_clock went backward; don't blame another CPU. */ - ts_delta = ts0 - ts2; + ts_delta = state->ts_start - ts_now; pr_alert("sched_clock on CPU %d went backward by %llu ns\n", raw_smp_processor_id(), ts_delta); - *ts1 = ts2; + state->ts_report = ts_now; return false; } - firsttime = !*bug_id; + firsttime = !state->bug_id; if (firsttime) - *bug_id = atomic_inc_return(&csd_bug_count); + state->bug_id = atomic_inc_return(&csd_bug_count); cpu = csd_lock_wait_getcpu(csd); if (WARN_ONCE(cpu < 0 || cpu >= nr_cpu_ids, "%s: cpu = %d\n", __func__, cpu)) cpux = 0; @@ -274,11 +282,11 @@ static bool csd_lock_wait_toolong(call_single_data_t *csd, u64 ts0, u64 *ts1, in cpux = cpu; cpu_cur_csd = smp_load_acquire(&per_cpu(cur_csd, cpux)); /* Before func and info. */ /* How long since this CSD lock was stuck. */ - ts_delta = ts2 - ts0; + ts_delta = ts_now - state->ts_start; pr_alert("csd: %s non-responsive CSD lock (#%d) on CPU#%d, waiting %lld ns for CPU#%02d %pS(%ps).\n", - firsttime ? "Detected" : "Continued", *bug_id, raw_smp_processor_id(), (s64)ts_delta, + firsttime ? "Detected" : "Continued", state->bug_id, raw_smp_processor_id(), (s64)ts_delta, cpu, csd->func, csd->info); - (*nmessages)++; + state->nmessages++; if (firsttime) atomic_inc(&n_csd_lock_stuck); /* @@ -289,23 +297,23 @@ static bool csd_lock_wait_toolong(call_single_data_t *csd, u64 ts0, u64 *ts1, in BUG_ON(panic_on_ipistall > 0 && (s64)ts_delta > ((s64)panic_on_ipistall * NSEC_PER_MSEC)); if (cpu_cur_csd && csd != cpu_cur_csd) { pr_alert("\tcsd: CSD lock (#%d) handling prior %pS(%ps) request.\n", - *bug_id, READ_ONCE(per_cpu(cur_csd_func, cpux)), + state->bug_id, READ_ONCE(per_cpu(cur_csd_func, cpux)), READ_ONCE(per_cpu(cur_csd_info, cpux))); } else { pr_alert("\tcsd: CSD lock (#%d) %s.\n", - *bug_id, !cpu_cur_csd ? "unresponsive" : "handling this request"); + state->bug_id, !cpu_cur_csd ? "unresponsive" : "handling this request"); } if (cpu >= 0) { if (atomic_cmpxchg_acquire(&per_cpu(trigger_backtrace, cpu), 1, 0)) dump_cpu_task(cpu); if (!cpu_cur_csd) { - pr_alert("csd: Re-sending CSD lock (#%d) IPI from CPU#%02d to CPU#%02d\n", *bug_id, raw_smp_processor_id(), cpu); + pr_alert("csd: Re-sending CSD lock (#%d) IPI from CPU#%02d to CPU#%02d\n", state->bug_id, raw_smp_processor_id(), cpu); arch_send_call_function_single_ipi(cpu); } } if (firsttime) dump_stack(); - *ts1 = ts2; + state->ts_report = ts_now; return false; } @@ -319,13 +327,11 @@ static bool csd_lock_wait_toolong(call_single_data_t *csd, u64 ts0, u64 *ts1, in */ static void __csd_lock_wait(call_single_data_t *csd) { - unsigned long nmessages = 0; - int bug_id = 0; - u64 ts0, ts1; + struct csd_wait_state state = {}; - ts1 = ts0 = ktime_get_mono_fast_ns(); + state.ts_report = state.ts_start = ktime_get_mono_fast_ns(); for (;;) { - if (csd_lock_wait_toolong(csd, ts0, &ts1, &bug_id, &nmessages)) + if (csd_lock_wait_toolong(csd, &state)) break; cpu_relax(); } From 38308bbd9f164873eb9bbd431ebafbc7fd783e9b Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 10 Aug 2026 04:29:25 -0700 Subject: [PATCH 176/857] locking/csd-lock: Report how long a stuck CSD lock took to recover The CSD lock debug output is a useful way to catch IPI stalls, but when the lock finally recovers it only says that it did: smp: csd: CSD lock (#1) got unstuck on CPU#32, CPU#123 released the lock. How long the target took to answer is left out, even though csd_lock_wait_toolong() already has the timestamp the wait started from. At Meta's fleet, that line fired 211K times in the last 24 hours, so plenty of stalls get reported with no indication of how long they lasted. Print how long the lock was stuck, and, when an IPI was re-sent, how long after that re-send the target released, as: smp: csd: CSD lock (#1) got unstuck on CPU#00, CPU#01 released the lock after 8000854660 ns, 3000825778 ns after the last IPI re-send. Signed-off-by: Breno Leitao Reviewed-by: Dmitry Ilvokhin Signed-off-by: Paul E. McKenney --- kernel/smp.c | 23 +++++++++++++++++++++-- 1 file changed, 21 insertions(+), 2 deletions(-) diff --git a/kernel/smp.c b/kernel/smp.c index 1bd8a24349e096..9092315c144a98 100644 --- a/kernel/smp.c +++ b/kernel/smp.c @@ -227,10 +227,28 @@ bool csd_lock_is_stuck(void) struct csd_wait_state { u64 ts_start; /* When the wait began. */ u64 ts_report; /* When the last complaint was printed. */ + u64 ts_resend; /* When the last IPI was re-sent, 0 if never. */ int bug_id; unsigned long nmessages; }; +/* + * Report a CSD lock that came back, @ts_unstuck being when the release was + * noticed. Only mention the re-send delta if an IPI was actually re-sent. + */ +static void csd_lock_print_unstuck(struct csd_wait_state *state, int cpu, u64 ts_unstuck) +{ + if (state->ts_resend) + pr_alert("csd: CSD lock (#%d) got unstuck on CPU#%02d, CPU#%02d released the lock after %lld ns, %lld ns after the last IPI re-send.\n", + state->bug_id, raw_smp_processor_id(), cpu, + (s64)(ts_unstuck - state->ts_start), + (s64)(ts_unstuck - state->ts_resend)); + else + pr_alert("csd: CSD lock (#%d) got unstuck on CPU#%02d, CPU#%02d released the lock after %lld ns.\n", + state->bug_id, raw_smp_processor_id(), cpu, + (s64)(ts_unstuck - state->ts_start)); +} + /* * Complain if too much time spent waiting. Note that only * the CSD_TYPE_SYNC/ASYNC types provide the destination CPU, @@ -249,9 +267,9 @@ static bool csd_lock_wait_toolong(call_single_data_t *csd, struct csd_wait_state if (!(flags & CSD_FLAG_LOCK)) { if (!unlikely(state->bug_id)) return true; + ts_now = ktime_get_mono_fast_ns(); cpu = csd_lock_wait_getcpu(csd); - pr_alert("csd: CSD lock (#%d) got unstuck on CPU#%02d, CPU#%02d released the lock.\n", - state->bug_id, raw_smp_processor_id(), cpu); + csd_lock_print_unstuck(state, cpu, ts_now); atomic_dec(&n_csd_lock_stuck); return true; } @@ -309,6 +327,7 @@ static bool csd_lock_wait_toolong(call_single_data_t *csd, struct csd_wait_state if (!cpu_cur_csd) { pr_alert("csd: Re-sending CSD lock (#%d) IPI from CPU#%02d to CPU#%02d\n", state->bug_id, raw_smp_processor_id(), cpu); arch_send_call_function_single_ipi(cpu); + state->ts_resend = ktime_get_mono_fast_ns(); } } if (firsttime) From 915312ce14447560e21258be05d44fb7d6e84a4f Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 10 Aug 2026 04:29:26 -0700 Subject: [PATCH 177/857] lib/test_csd_lock: Add a module to stall a CPU on a CSD lock Add test_csd_lock, a module that keeps one CPU from answering an IPI for as long as its stall_ms parameter says, so that the CSD-lock debug code has a stall to report. With in_handler=1 the CPU stalls inside a CSD handler instead, which is the case where the IPI is not re-sent. The module needs CONFIG_CSD_LOCK_WAIT_DEBUG and csdlock_debug=1. Loading it runs one stall and then fails the load with -EAGAIN, the way test_lockup does, so that nothing is left loaded afterwards. Signed-off-by: Breno Leitao Signed-off-by: Paul E. McKenney --- lib/Kconfig.debug | 12 +++ lib/Makefile | 1 + lib/test_csd_lock.c | 173 ++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 186 insertions(+) create mode 100644 lib/test_csd_lock.c diff --git a/lib/Kconfig.debug b/lib/Kconfig.debug index 1244dcac2294ad..d693caaeae873a 100644 --- a/lib/Kconfig.debug +++ b/lib/Kconfig.debug @@ -1372,6 +1372,18 @@ config WQ_CPU_INTENSIVE_REPORT triggering likely indicates that the work item should be switched to use an unbound workqueue. +config TEST_CSD_LOCK + tristate "Test module to stall a CPU on a CSD lock" + depends on m + depends on CSD_LOCK_WAIT_DEBUG + help + This builds the "test_csd_lock" module, which keeps one CPU from + answering an IPI for as long as its stall_ms parameter says, so + that the CSD-lock debug code has a stall to report. It needs + csdlock_debug=1 to be of any use. + + If unsure, say N. + config TEST_LOCKUP tristate "Test module to generate lockups" depends on m diff --git a/lib/Makefile b/lib/Makefile index 7f75cc6edf94af..92f0ab7b740b48 100644 --- a/lib/Makefile +++ b/lib/Makefile @@ -99,6 +99,7 @@ obj-$(CONFIG_TEST_DEBUG_VIRTUAL) += test_debug_virtual.o obj-$(CONFIG_TEST_MEMCAT_P) += test_memcat_p.o obj-$(CONFIG_TEST_OBJAGG) += test_objagg.o obj-$(CONFIG_TEST_MEMINIT) += test_meminit.o +obj-$(CONFIG_TEST_CSD_LOCK) += test_csd_lock.o obj-$(CONFIG_TEST_LOCKUP) += test_lockup.o obj-$(CONFIG_TEST_HMM) += test_hmm.o obj-$(CONFIG_TEST_FREE_PAGES) += test_free_pages.o diff --git a/lib/test_csd_lock.c b/lib/test_csd_lock.c new file mode 100644 index 00000000000000..30c6c3332c3fc0 --- /dev/null +++ b/lib/test_csd_lock.c @@ -0,0 +1,173 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Keep one CPU from answering an IPI, so that the CSD-lock debug code in + * kernel/smp.c has a stall to report. + * + * Copyright (c) 2026 Meta Platforms, Inc. and affiliates + * Copyright (c) 2026 Breno Leitao + * + * The target either spins with interrupts disabled, which leaves it idle as + * far as the debug code can tell and gets the IPI re-sent, or spins inside a + * CSD handler, which does not. The recovery message differs between the two. + * + * Loading the module runs one stall, then fails the load with -EAGAIN so + * that nothing is left loaded afterwards: + * + * echo 500 > /sys/module/smp/parameters/csd_lock_timeout + * modprobe test_csd_lock stall_ms=1000 in_handler=0 + * + * csd_lock_timeout has to be below stall_ms for the stall to be reported at + * all, and the report has to come out before the CPU answers, so leave it + * some room. + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt + +#include +#include +#include +#include +#include +#include +#include + +#define STALL_MS_MAX 10000 + +static unsigned int stall_ms = 1000; +module_param(stall_ms, uint, 0444); +MODULE_PARM_DESC(stall_ms, "Time the target CPU ignores the IPI, in milliseconds."); + +static int stall_cpu = -1; +module_param(stall_cpu, int, 0444); +MODULE_PARM_DESC(stall_cpu, "CPU to stall, or -1 for the first online one."); + +static bool in_handler; +module_param(in_handler, bool, 0444); +MODULE_PARM_DESC(in_handler, "Stall inside a CSD handler instead of with interrupts disabled."); + +static int target_cpu; +static bool target_stalling; +static bool hog_launched; +static struct work_struct irqoff_work; +static struct work_struct sender_work; +static call_single_data_t hog_csd; +static DECLARE_COMPLETION(hog_done); + +static void csd_test_nop(void *unused) +{ +} + +static void csd_test_spin(void) +{ + u64 end = ktime_get_mono_fast_ns() + (u64)stall_ms * NSEC_PER_MSEC; + + while (ktime_get_mono_fast_ns() < end) + cpu_relax(); +} + +/* Nothing is running for the target while interrupts are off, so it gets a new IPI. */ +static void csd_test_irqoff_fn(struct work_struct *work) +{ + local_irq_disable(); + /* Pairs with the load in csd_test_sender_fn(), which waits for this. */ + smp_store_release(&target_stalling, true); + csd_test_spin(); + local_irq_enable(); +} + +/* Here cur_csd stays set on the target, which suppresses the re-send. */ +static void csd_test_hog_fn(void *unused) +{ + /* Pairs with the load in csd_test_sender_fn(), which waits for this. */ + smp_store_release(&target_stalling, true); + csd_test_spin(); + complete(&hog_done); +} + +/* + * Start the stall from here rather than from module init, so that however + * long this work item waits to be scheduled comes off before the target + * stops answering, not out of the middle of the stall. + */ +static void csd_test_sender_fn(struct work_struct *work) +{ + u64 deadline, ts; + int err; + + if (in_handler) { + hog_csd.func = csd_test_hog_fn; + err = smp_call_function_single_async(target_cpu, &hog_csd); + if (err) { + pr_err("cannot queue the CSD handler on CPU%d: %d\n", target_cpu, err); + return; + } + } else { + queue_work_on(target_cpu, system_highpri_wq, &irqoff_work); + } + WRITE_ONCE(hog_launched, true); + + deadline = ktime_get_mono_fast_ns() + (u64)STALL_MS_MAX * NSEC_PER_MSEC; + /* Pairs with the store in the stall functions: send once it is stuck. */ + while (!smp_load_acquire(&target_stalling)) { + if (ktime_get_mono_fast_ns() > deadline) { + pr_err("CPU%d never stopped answering\n", target_cpu); + return; + } + cpu_relax(); + } + + ts = ktime_get_mono_fast_ns(); + smp_call_function_single(target_cpu, csd_test_nop, NULL, 1); + pr_info("CPU%d answered after %llu ns\n", target_cpu, + ktime_get_mono_fast_ns() - ts); +} + +static int __init test_csd_lock_init(void) +{ + int sender_cpu; + int ret = 0; + + if (!stall_ms || stall_ms > STALL_MS_MAX) { + pr_err("stall_ms must be between 1 and %d\n", STALL_MS_MAX); + return -EINVAL; + } + + INIT_WORK(&irqoff_work, csd_test_irqoff_fn); + INIT_WORK(&sender_work, csd_test_sender_fn); + + cpus_read_lock(); + + target_cpu = stall_cpu < 0 ? cpumask_first(cpu_online_mask) : stall_cpu; + sender_cpu = nr_cpu_ids; + if (target_cpu < nr_cpu_ids && cpu_online(target_cpu)) + sender_cpu = cpumask_any_but(cpu_online_mask, target_cpu); + if (sender_cpu >= nr_cpu_ids) { + pr_err("need CPU%d and one other CPU online\n", target_cpu); + ret = -EINVAL; + goto unlock; + } + + pr_info("stalling CPU%d for %u ms %s, IPI from CPU%d\n", target_cpu, stall_ms, + in_handler ? "inside a CSD handler" : "with interrupts disabled", sender_cpu); + + queue_work_on(sender_cpu, system_highpri_wq, &sender_work); + flush_work(&sender_work); + flush_work(&irqoff_work); + + /* The CSD has to be idle again before this module goes away. */ + if (in_handler && READ_ONCE(hog_launched) && + !wait_for_completion_timeout(&hog_done, msecs_to_jiffies(2 * STALL_MS_MAX))) + pr_err("CSD handler on CPU%d never finished\n", target_cpu); + + /* The stall is over and there is nothing left to hold, so go away. */ + ret = -EAGAIN; +unlock: + cpus_read_unlock(); + + return ret; +} +module_init(test_csd_lock_init); + +MODULE_LICENSE("GPL"); +MODULE_AUTHOR("Breno Leitao "); +MODULE_DESCRIPTION("Test module to stall a CPU on a CSD lock"); From 0c9d2588a514f8719ee84640a8ec65bdfcb75384 Mon Sep 17 00:00:00 2001 From: Junjie Cao Date: Mon, 24 Aug 2026 13:32:27 +0800 Subject: [PATCH 178/857] Bluetooth: btusb: limit RTL8761B BROKEN_EXT_SCAN quirk to 0bda:a728 Commit 5ead2063611a ("Bluetooth: btrtl: fix RTL8761B/BU broken LE extended scan") set HCI_QUIRK_BROKEN_EXT_SCAN for every CHIP_ID_8761B device to cure repeated 0x2042 failures on an 0bda:a728 dongle. The brokenness is per-dongle, not per-chip: on a TP-Link UB500 (2357:0604, RTL8761BU, fw 0xdfc6d922) extended scan works, and the legacy scan path the quirk forces is what is broken -- LE Set Scan Enable (0x200c) times out with -110 about 30 s after firmware load, btusb resets the device, and the adapter re-enumerates in an endless loop (382 firmware reloads in one boot). 7.1.8, which predates the stable backport, runs clean on this unit; 7.1.9 loops. Move the quirk from btrtl's chip-wide switch to a btusb device-table flag on the USB id the original fix was verified against. Other 8761B dongles return to their earlier long-standing behaviour. Link: https://bugzilla.redhat.com/show_bug.cgi?id=2521504 Fixes: 5ead2063611a ("Bluetooth: btrtl: fix RTL8761B/BU broken LE extended scan") Cc: stable@vger.kernel.org Signed-off-by: Junjie Cao Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btrtl.c | 13 ------------- drivers/bluetooth/btusb.c | 8 ++++++++ 2 files changed, 8 insertions(+), 13 deletions(-) diff --git a/drivers/bluetooth/btrtl.c b/drivers/bluetooth/btrtl.c index 7f54d2d2d13a04..03fa9409e3ee49 100644 --- a/drivers/bluetooth/btrtl.c +++ b/drivers/bluetooth/btrtl.c @@ -1343,19 +1343,6 @@ void btrtl_set_quirks(struct hci_dev *hdev, struct btrtl_device_info *btrtl_dev) if (!btrtl_dev->ic_info) return; - switch (btrtl_dev->project_id) { - case CHIP_ID_8761B: - /* RTL8761B/BU reports HCI version 5.1 but does not support - * the LE Extended Scan commands (Opcode 0x2042), causing - * repeated -EBUSY failures when BlueZ attempts extended - * scanning while a connection is active. - */ - hci_set_quirk(hdev, HCI_QUIRK_BROKEN_EXT_SCAN); - break; - default: - break; - } - switch (btrtl_dev->ic_info->lmp_subver) { case RTL_ROM_LMP_8703B: /* 8723CS reports two pages for local ext features, diff --git a/drivers/bluetooth/btusb.c b/drivers/bluetooth/btusb.c index 2bae85b0016cee..84614e60d142f1 100644 --- a/drivers/bluetooth/btusb.c +++ b/drivers/bluetooth/btusb.c @@ -67,6 +67,7 @@ static struct usb_driver btusb_driver; #define BTUSB_INTEL_NO_WBS_SUPPORT BIT(26) #define BTUSB_ACTIONS_SEMI BIT(27) #define BTUSB_BARROT BIT(28) +#define BTUSB_BROKEN_EXT_SCAN BIT(29) static const struct usb_device_id btusb_table[] = { /* Generic Bluetooth USB device */ @@ -619,6 +620,10 @@ static const struct usb_device_id quirks_table[] = { { USB_DEVICE(0x0489, 0xe130), .driver_info = BTUSB_REALTEK | BTUSB_WIDEBAND_SPEECH }, + /* Realtek 8761BU Bluetooth devices */ + { USB_DEVICE(0x0bda, 0xa728), .driver_info = BTUSB_REALTEK | + BTUSB_BROKEN_EXT_SCAN }, + /* Realtek Bluetooth devices */ { USB_VENDOR_AND_INTERFACE_INFO(0x0bda, 0xe0, 0x01, 0x01), .driver_info = BTUSB_REALTEK }, @@ -4403,6 +4408,9 @@ static int btusb_probe(struct usb_interface *intf, if (id->driver_info & BTUSB_INVALID_LE_STATES) hci_set_quirk(hdev, HCI_QUIRK_BROKEN_LE_STATES); + if (id->driver_info & BTUSB_BROKEN_EXT_SCAN) + hci_set_quirk(hdev, HCI_QUIRK_BROKEN_EXT_SCAN); + if (id->driver_info & BTUSB_DIGIANSWER) { data->cmdreq_type = USB_TYPE_VENDOR; hci_set_quirk(hdev, HCI_QUIRK_RESET_ON_CLOSE); From 40c621391de9371bea89b2fe5c7a90129f5922cb Mon Sep 17 00:00:00 2001 From: Chengfeng Ye Date: Sun, 23 Aug 2026 00:43:41 +0800 Subject: [PATCH 179/857] Bluetooth: RFCOMM: serialize security confirmation handling rfcomm_security_cfm() looks up a session on session_list and then walks its DLC list without holding rfcomm_mutex. Since RFCOMM session teardown uses rfcomm_mutex, krfcommd can close and free the same session and DLCs concurrently: hci_rx_work krfcommd ----------- --------- rfcomm_session_get() rfcomm_lock() rfcomm_session_close() rfcomm_dlc_unlink() rfcomm_session_del() kfree(s) rfcomm_unlock() walk s->dlcs The callback can then read a freed session list head and touch freed DLCs while updating their flags or timers. Serialize the session lookup and DLC traversal in rfcomm_security_cfm() with rfcomm_mutex. This matches the existing RFCOMM session lifetime rules and prevents concurrent rfcomm_session_del() / rfcomm_dlc_unlink() from tearing the objects down while the callback is using them. KASAN reported: BUG: KASAN: slab-use-after-free in rfcomm_security_cfm+0x41c/0x440 Read of size 8 at addr ffff888111fb3960 by task kworker/u17:1/89 Workqueue: hci0 hci_rx_work Call Trace: rfcomm_security_cfm+0x41c/0x440 hci_encrypt_cfm+0x139/0x590 hci_encrypt_change_evt+0x37b/0xc40 hci_event_packet+0x71b/0xb20 hci_rx_work+0x293/0x730 Allocated by task 69: rfcomm_session_add+0x9e/0x2f0 rfcomm_run+0x44b/0x41e0 Freed by task 69: kfree+0x131/0x3c0 rfcomm_session_del+0x188/0x220 rfcomm_run+0x1985/0x41e0 Fixes: 08c30aca9e698faddebd34f81e1196295f9dc063 ("Bluetooth: Remove RFCOMM session refcnt") Cc: stable@vger.kernel.org Signed-off-by: Chengfeng Ye Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/rfcomm/core.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/net/bluetooth/rfcomm/core.c b/net/bluetooth/rfcomm/core.c index 0e496b85e6cec8..63fa0f542ccf1d 100644 --- a/net/bluetooth/rfcomm/core.c +++ b/net/bluetooth/rfcomm/core.c @@ -2217,9 +2217,13 @@ static void rfcomm_security_cfm(struct hci_conn *conn, u8 status, u8 encrypt) BT_DBG("conn %p status 0x%02x encrypt 0x%02x", conn, status, encrypt); + rfcomm_lock(); + s = rfcomm_session_get(&conn->hdev->bdaddr, &conn->dst); - if (!s) + if (!s) { + rfcomm_unlock(); return; + } list_for_each_entry_safe(d, n, &s->dlcs, list) { if (test_and_clear_bit(RFCOMM_SEC_PENDING, &d->flags)) { @@ -2251,6 +2255,8 @@ static void rfcomm_security_cfm(struct hci_conn *conn, u8 status, u8 encrypt) set_bit(RFCOMM_AUTH_REJECT, &d->flags); } + rfcomm_unlock(); + rfcomm_schedule(); } From a6491451f8b8481d8ae0d4057fc046eaf7cb5ff1 Mon Sep 17 00:00:00 2001 From: George Maraveyas Date: Sat, 22 Aug 2026 01:37:33 +0200 Subject: [PATCH 180/857] Bluetooth: mt7925: trigger reset on WMT timeout The MT7925 Bluetooth USB function can enumerate successfully after a warm reboot while the WMT function-control command remains unresponsive. When that command times out, btmtk_usb_setup() currently returns -ETIMEDOUT without entering the existing MediaTek reset path. The existing USB reset and recovery machinery is therefore never reached. For MT7925, call btmtk_reset_sync() when the WMT function-control command times out. This enters the existing reset path in btusb_mtk_reset(), which performs the MediaTek subsystem reset and queues a USB device reset. Runtime tracing on the affected hardware showed the resulting path through usb_queue_reset_device(), usb_reset_device() and usb_reset_and_verify_device(). When reset and verification could not restore the device, the USB core escalated to a logical disconnect and re-enumeration. Recovery succeeded in three controlled Windows-to-Linux tests. Runtime tracing showed the existing USB reset path escalating to logical disconnect and re-enumeration. In two of those tests, tracing continued through the subsequent enumeration failures and directly captured usb_acpi_port_prr_reset(), after which the MT7925 re-enumerated and Bluetooth recovered. These tests were performed on top of Chia-Lin Kao's ACPI _PRR hub patch, which remains a prerequisite for this patch. A fourth Windows-to-Linux test was then performed with the diagnostic btusb blacklist removed and btusb binding normally during boot. The WMT timeout reproduced and Bluetooth recovered automatically without manual intervention. Signed-off-by: George Maraveyas Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmtk.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/drivers/bluetooth/btmtk.c b/drivers/bluetooth/btmtk.c index c0ed51567ed4dd..73ba4a029e9ccf 100644 --- a/drivers/bluetooth/btmtk.c +++ b/drivers/bluetooth/btmtk.c @@ -1418,6 +1418,10 @@ int btmtk_usb_setup(struct hci_dev *hdev) err = btmtk_usb_hci_wmt_sync(hdev, &wmt_params); if (err < 0) { bt_dev_err(hdev, "Failed to send wmt func ctrl (%d)", err); + + if (dev_id == 0x7925 && err == -ETIMEDOUT) + btmtk_reset_sync(hdev); + return err; } From aadb3cd4bbb4744ecb274c232d33947f77e45815 Mon Sep 17 00:00:00 2001 From: Radek Podgorny Date: Mon, 24 Aug 2026 13:00:20 +0200 Subject: [PATCH 181/857] Bluetooth: do not leak an hci_conn when a second LE connect is rejected create_le_conn_complete() decides whether the failed connection is still pending by comparing it against hci_lookup_le_connect(), which returns the first LE connection in BT_CONNECT. That is the same connection only while at most one is pending. Two can be pending. Connections created on the passive scan path sit in BT_CONNECT with HCI_CONN_SCANNING set and are invisible to hci_lookup_le_connect() until hci_le_create_conn_sync() clears the flag when their command is issued, so the -EBUSY guard in hci_connect_le() does not prevent a second connection from being queued while the first is still on the scan path. Whenever two connections are in BT_CONNECT at once, the lookup may return one connection while create_le_conn_complete() is reporting the failure of the other; the early exit then drops the error and hci_conn_failed() never runs on the connection that failed. The controller also rejects a second HCI_OP_LE_CREATE_CONN issued while another connection creation is still outstanding, per Core Spec Vol 4, Part E. The spec calls for Command Disallowed there; the bcm43438 observed here answers with an LMP/LL error code instead, which bt_to_errno() maps to the -EPROTO (-71) in the log below. The leaked connection stays in BT_CONNECT forever, and because hci_connect_le() refuses to dial while hci_lookup_le_connect() finds anything, every subsequent attempt to reach any peer fails with -EBUSY and no command reaches the controller at all. Seen on a bcm43438 with two BLE peers polled on the same interval (state 5 is BT_CONNECT; both handles are UNSET ones, allocated from the ida above HCI_CONN_HANDLE_MAX): Bluetooth: hci1: Opcode 0x2013 failed: -71 # hcitool con < LE 14:9C:EF:03:68:81 handle 3840 state 5 lm CENTRAL < LE C4:D3:6A:8C:B5:38 handle 3841 state 5 lm CENTRAL A btmon capture across the next ten minutes of connect attempts contains no HCI_OP_LE_CREATE_CONN at all; outgoing LE connections do not recover until the adapter is reset. With this change the same scenario fails the rejected connection cleanly and further connects to both peers go through. Ask about the connection itself instead of about the device. Fixes: c9f73a2178c1 ("Bluetooth: hci_conn: Fix hci_connect_le_sync") Signed-off-by: Radek Podgorny Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/hci_sync.c | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/net/bluetooth/hci_sync.c b/net/bluetooth/hci_sync.c index 7150037a864b44..24eeb76f72074d 100644 --- a/net/bluetooth/hci_sync.c +++ b/net/bluetooth/hci_sync.c @@ -7279,8 +7279,13 @@ static void create_le_conn_complete(struct hci_dev *hdev, void *data, int err) goto unlock; } - /* Check if connection is still pending */ - if (conn != hci_lookup_le_connect(hdev)) + /* Check if this connection is still pending. + * + * hci_lookup_le_connect() returns only the first LE connection + * in BT_CONNECT, which is not necessarily this one when two are + * pending at once, so ask the connection itself. + */ + if (conn->state != BT_CONNECT) goto unlock; /* Flush to make sure we send create conn cancel command if needed */ From fe3897b4ab57b7c934532d86df79c129d6ac17a9 Mon Sep 17 00:00:00 2001 From: Chengfeng Ye Date: Sat, 22 Aug 2026 23:06:19 +0800 Subject: [PATCH 182/857] Bluetooth: RFCOMM: serialize session teardown rfcomm_kill_listener() walks session_list and deletes every session without holding rfcomm_mutex, unlike the normal session processing and connect error paths. Under normal operation, an open RFCOMM socket pins rfcomm.ko, so rfcomm_kill_listener() does not run concurrently with rfcomm_dlc_open(). However, forced module unload via delete_module(O_TRUNC) can stop krfcommd while a failed connect is still unwinding. connect task forced unload / krfcommd ------------ ------------------------ rfcomm_lock() rfcomm_session_add() delete_module("rfcomm", O_TRUNC) rfcomm_kill_listener() fetch session from session_list kernel_connect() fails rfcomm_session_del() remove and free session rfcomm_session_del(session) The final call then reads the freed session and may corrupt the list. KASAN reported with mdelay() to enlarge critical window: BUG: KASAN: slab-use-after-free in rfcomm_run+0x3802/0x3f00 [rfcomm] Read of size 8 at addr ffff888111058d40 by task krfcommd/79 Tainted: [R]=FORCED_RMMOD Allocated by task 86: rfcomm_session_add+0xa1/0x300 [rfcomm] rfcomm_dlc_open+0x8b2/0xf30 [rfcomm] rfcomm_sock_connect+0x34c/0x530 [rfcomm] Freed by task 86: kfree+0x121/0x3c0 rfcomm_dlc_open+0xab7/0xf30 [rfcomm] rfcomm_sock_connect+0x34c/0x530 [rfcomm] Hold rfcomm_mutex across the teardown traversal so every reachable session_list walk uses the same serialization. Reviewed-by: Ali Ahmet Memis Tested-by: Ali Ahmet Memis Reviewed-by: Pauli Virtanen Signed-off-by: Chengfeng Ye Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/rfcomm/core.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/net/bluetooth/rfcomm/core.c b/net/bluetooth/rfcomm/core.c index 63fa0f542ccf1d..f7463f09228329 100644 --- a/net/bluetooth/rfcomm/core.c +++ b/net/bluetooth/rfcomm/core.c @@ -2182,8 +2182,10 @@ static void rfcomm_kill_listener(void) BT_DBG(""); + rfcomm_lock(); list_for_each_entry_safe(s, n, &session_list, list) rfcomm_session_del(s); + rfcomm_unlock(); } static int rfcomm_run(void *unused) From 26b2c52685c146b65f354b0623b0044626c35494 Mon Sep 17 00:00:00 2001 From: Frank Li Date: Thu, 23 Jul 2026 15:14:15 -0400 Subject: [PATCH 183/857] dt-bindings: display: bridge: ldb: allow a single reg for fsl,imx6sx-ldb The i.MX6SX LDB only provides a single register region for the LDB block, while other supported LDB variants require two register regions. Update the binding schema to allow a single reg entry for fsl,imx6sx-ldb while keeping the existing constraints unchanged for the other compatible strings. Fix below DTB_CHECK warings: arch/arm/boot/dts/nxp/imx/imx6sx-nitrogen6sx.dtb: bridge@18 (fsl,imx6sx-ldb): reg: [[24, 4]] is too short Acked-by: Rob Herring (Arm) Signed-off-by: Frank Li --- .../bindings/display/bridge/fsl,ldb.yaml | 23 +++++++++++-------- 1 file changed, 14 insertions(+), 9 deletions(-) diff --git a/Documentation/devicetree/bindings/display/bridge/fsl,ldb.yaml b/Documentation/devicetree/bindings/display/bridge/fsl,ldb.yaml index 7f380879fffdfe..e5a8870bb76a62 100644 --- a/Documentation/devicetree/bindings/display/bridge/fsl,ldb.yaml +++ b/Documentation/devicetree/bindings/display/bridge/fsl,ldb.yaml @@ -28,9 +28,11 @@ properties: const: ldb reg: + minItems: 1 maxItems: 2 reg-names: + minItems: 1 items: - const: ldb - const: lvds @@ -83,15 +85,6 @@ allOf: ports: properties: port@2: false - - if: - not: - properties: - compatible: - contains: - const: fsl,imx6sx-ldb - then: - required: - - reg-names - if: properties: @@ -100,7 +93,19 @@ allOf: const: fsl,imx6sx-ldb then: properties: + reg: + maxItems: 1 + reg-names: + maxItems: 1 nxp,enable-termination-resistor: false + else: + required: + - reg-names + properties: + reg: + minItems: 2 + reg-names: + minItems: 2 additionalProperties: false From 58627dbbd8e8ea7f4429b69b15b1c443681bf05a Mon Sep 17 00:00:00 2001 From: Frank Li Date: Thu, 23 Jul 2026 15:14:16 -0400 Subject: [PATCH 184/857] dt-bindings: soc: imx-iomuxc-gpr: allow bridge@18 as child node The legacy i.MX6SX (>15 year) SoC imx-iomuxc-gpr contains one LDB_CTRL register. Allow the LVDS Display Bridge(LDB) child node under imx-iomuxc-gpr. Fix below CHECK_DTBS warnings: arch/arm/boot/dts/nxp/imx/imx6sx-nitrogen6sx.dtb: syscon@20e4000 (fsl,imx6sx-iomuxc-gpr): '#address-cells', '#size-cells', 'bridge@18' do not match any of the regexes: '^ipu[12]_csi[01]_mux$', '^pinctrl-[0-9]+$ Reviewed-by: Rob Herring (Arm) Signed-off-by: Frank Li --- .../bindings/soc/imx/fsl,imx-iomuxc-gpr.yaml | 62 +++++++++++++++++++ 1 file changed, 62 insertions(+) diff --git a/Documentation/devicetree/bindings/soc/imx/fsl,imx-iomuxc-gpr.yaml b/Documentation/devicetree/bindings/soc/imx/fsl,imx-iomuxc-gpr.yaml index 721a67e84c137a..49813adce71c70 100644 --- a/Documentation/devicetree/bindings/soc/imx/fsl,imx-iomuxc-gpr.yaml +++ b/Documentation/devicetree/bindings/soc/imx/fsl,imx-iomuxc-gpr.yaml @@ -47,10 +47,23 @@ properties: reg: maxItems: 1 + '#address-cells': + const: 1 + + '#size-cells': + const: 1 + + ranges: true + mux-controller: type: object $ref: /schemas/mux/reg-mux.yaml + bridge@18: + type: object + $ref: /schemas/display/bridge/fsl,ldb.yaml# + unevaluatedProperties: false + patternProperties: "^ipu[12]_csi[01]_mux$": type: object @@ -67,6 +80,19 @@ allOf: patternProperties: '^ipu[12]_csi[01]_mux$': false + - if: + properties: + compatible: + not: + contains: + const: fsl,imx6sx-iomuxc-gpr + then: + properties: + bridge@18: false + '#address-cells': false + '#size-cells': false + ranges: false + additionalProperties: false required: @@ -87,4 +113,40 @@ examples: }; }; + - | + #include + + syscon@20e4000 { + compatible = "fsl,imx6sx-iomuxc-gpr", "fsl,imx6q-iomuxc-gpr", "syscon", "simple-mfd"; + reg = <0x020e4000 0x4000>; + #address-cells = <1>; + #size-cells = <1>; + ranges; + + bridge@18 { + compatible = "fsl,imx6sx-ldb"; + reg = <0x18 0x4>; + clocks = <&clks IMX6SX_CLK_LDB_DI0>; + clock-names = "ldb"; + + ports { + #address-cells = <1>; + #size-cells = <0>; + + port@0 { + reg = <0>; + + endpoint { + }; + }; + + port@1 { + reg = <1>; + + endpoint { + }; + }; + }; + }; + }; ... From 17c5e5980b6fe3be98b21c046adf2aaa62006f0d Mon Sep 17 00:00:00 2001 From: Frank Li Date: Thu, 23 Jul 2026 15:14:17 -0400 Subject: [PATCH 185/857] dt-bindings: display: lcdif: Allow display0 child node for i.MX6UL The legacy i.MX6UL LCDIF binding uses a display0 child node to describe the attached display. Update the binding schema to allow this child node for fsl,imx6ul-lcdif. Fixes the following CHECK_DTBS warning: arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul-0010.dtb: lcdif@21c8000 (fsl,imx6ul-lcdif): 'disp0' does not match any of the regexes: '^pinctrl-[0-9]+$' A follow-up patch renames the child node from disp0 to display0 to match the updated binding. Acked-by: Rob Herring (Arm) Signed-off-by: Frank Li --- Documentation/devicetree/bindings/display/fsl,lcdif.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/Documentation/devicetree/bindings/display/fsl,lcdif.yaml b/Documentation/devicetree/bindings/display/fsl,lcdif.yaml index 2dd0411ec65161..2b123ddf068415 100644 --- a/Documentation/devicetree/bindings/display/fsl,lcdif.yaml +++ b/Documentation/devicetree/bindings/display/fsl,lcdif.yaml @@ -182,6 +182,7 @@ allOf: contains: enum: - fsl,imx28-lcdif + - fsl,imx6ul-lcdif then: properties: dmas: false From 955326b06028ae72e1a46b23c2d3db749e9be47c Mon Sep 17 00:00:00 2001 From: Frank Li Date: Thu, 23 Jul 2026 15:14:18 -0400 Subject: [PATCH 186/857] ARM: dts: imx6ul-tx6ul: rename disp0 to display0 Change node name disp0 to display0 to match binding define. Fix the following CHECK_DTBS warning: arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul-0010.dtb: lcdif@21c8000 (fsl,imx6ul-lcdif): 'disp0' does not match any of the regexes: '^pinctrl-[0-9]+$' Signed-off-by: Frank Li --- arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul.dtsi | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul.dtsi b/arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul.dtsi index 192c6a95ae589f..2f6916c0359c5b 100644 --- a/arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul.dtsi +++ b/arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul.dtsi @@ -369,7 +369,7 @@ display = <&display>; status = "okay"; - display: disp0 { + display: display0 { bits-per-pixel = <32>; bus-width = <24>; status = "okay"; From aff53f8f5831c55cba94e0588dcf7d1dcde8b75d Mon Sep 17 00:00:00 2001 From: Mehmet Fide Date: Fri, 14 Aug 2026 09:46:57 +0200 Subject: [PATCH 187/857] ARM: dts: vf-colibri: apply the EXT_IO pin group pinctrl_gpio_ext describes the three Colibri EXT_IO pins, but no node references it, so the pin controller never applies it and the pads keep whatever mux and bias they came up with. It is the only pin group in the Vybrid Colibri device trees that is defined and never used. The three pins are GPIOs on gpio2 and belong to no device, so hog the group on that controller. Referencing it from &iomuxc would work too, but the pin controller would then depend on one of its own children, which fw_devlink reports as a dependency cycle. Signed-off-by: Mehmet Fide Signed-off-by: Frank Li --- arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi b/arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi index 98f9ee1b00306a..74afc7b0d74b94 100644 --- a/arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi +++ b/arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi @@ -170,6 +170,15 @@ status = "okay"; }; +&gpio2 { + /* + * EXT_IO_0..2 belong to no device, so hog the pin group here rather + * than on the pin controller, which would depend on its own child. + */ + pinctrl-names = "default"; + pinctrl-0 = <&pinctrl_gpio_ext>; +}; + &iomuxc { pinctrl_flexcan0: can0grp { fsl,pins = < From 04f9032ad6b65fe527cf4e3e825be3556ba92ad0 Mon Sep 17 00:00:00 2001 From: Mehmet Fide Date: Fri, 14 Aug 2026 09:46:58 +0200 Subject: [PATCH 188/857] ARM: dts: vf-colibri: name the SODIMM gpio lines The Colibri modules expose most of their GPIOs on numbered SODIMM pins, and the i.MX ones already carry those numbers as gpio-line-names, so the gpiod tools print SODIMM_43 rather than a bank and an offset. Do the same for the Vybrid modules. The 99 names come from the pinout table of the Colibri VFxx datasheet (Toradex 101355). Verified on a Colibri VF50, where gpioinfo now names every pin the datasheet lists. Signed-off-by: Mehmet Fide Signed-off-by: Frank Li --- arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi | 47 ++++++++++++++++++++++++ 1 file changed, 47 insertions(+) diff --git a/arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi b/arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi index 74afc7b0d74b94..a9269857df2c12 100644 --- a/arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi +++ b/arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi @@ -170,6 +170,44 @@ status = "okay"; }; +&gpio0 { + gpio-line-names = "", "", "SODIMM_89", "", + "", "SODIMM_102", "SODIMM_18", "SODIMM_134", + "SODIMM_16", "SODIMM_14", "SODIMM_23", "", + "", "", "SODIMM_47", "SODIMM_190", + "SODIMM_192", "SODIMM_49", "SODIMM_51", "SODIMM_53", + "SODIMM_37", "SODIMM_29", "", "SODIMM_30", + "SODIMM_20", "SODIMM_24", "SODIMM_21", "SODIMM_19", + "SODIMM_94", "SODIMM_81", "SODIMM_28"; +}; + +&gpio1 { + gpio-line-names = "SODIMM_35", "SODIMM_33", "SODIMM_27", "SODIMM_25", + "SODIMM_196", "SODIMM_194", "SODIMM_63", "SODIMM_55", + "SODIMM_65", "SODIMM_45", "SODIMM_43", "SODIMM_73", + "SODIMM_77", "SODIMM_71", "SODIMM_98", "SODIMM_101", + "SODIMM_103", "SODIMM_79", "SODIMM_97", "", + "", "SODIMM_85", "", "", + "", "", "", "", + "", "", "", "SODIMM_106"; +}; + +&gpio3 { + gpio-line-names = "SODIMM_105", "", "SODIMM_93", "", + "", "", "", "SODIMM_95", + "SODIMM_22", "SODIMM_68", "SODIMM_82", "SODIMM_56", + "SODIMM_131", "SODIMM_44", "SODIMM_144", "SODIMM_146", + "SODIMM_52", "SODIMM_54", "SODIMM_66", "SODIMM_64", + "SODIMM_57", "SODIMM_61", "SODIMM_140", "SODIMM_142", + "SODIMM_80", "SODIMM_46", "SODIMM_62", "SODIMM_48", + "SODIMM_74", "SODIMM_50", "SODIMM_136", "SODIMM_138"; +}; + +&gpio4 { + gpio-line-names = "SODIMM_76", "SODIMM_70", "SODIMM_60", "SODIMM_58", + "SODIMM_78", "SODIMM_72", "SODIMM_96"; +}; + &gpio2 { /* * EXT_IO_0..2 belong to no device, so hog the pin group here rather @@ -177,6 +215,15 @@ */ pinctrl-names = "default"; pinctrl-0 = <&pinctrl_gpio_ext>; + + gpio-line-names = "SODIMM_69", "SODIMM_99", "SODIMM_104", "SODIMM_107", + "SODIMM_127", "SODIMM_184", "SODIMM_186", "", + "", "", "", "", + "", "", "", "SODIMM_38", + "SODIMM_36", "SODIMM_34", "SODIMM_32", "SODIMM_129", + "SODIMM_86", "SODIMM_90", "SODIMM_92", "SODIMM_88", + "SODIMM_133", "SODIMM_135", "SODIMM_188", "SODIMM_75", + "SODIMM_100"; }; &iomuxc { From 3d8a67a1fbc8a18b99acfdb7af06d79e4f4ebcb8 Mon Sep 17 00:00:00 2001 From: Mehmet Fide Date: Fri, 14 Aug 2026 09:46:59 +0200 Subject: [PATCH 189/857] dt-bindings: arm: fsl: add the Colibri VF50 and VF61 on Iris The Iris is a small off-the-shelf carrier board for the Colibri family. Extend the evaluation board entries with the compatibles for a Colibri VF50 or VF61 sitting on it. Reviewed-by: Krzysztof Kozlowski Signed-off-by: Mehmet Fide Signed-off-by: Frank Li --- Documentation/devicetree/bindings/arm/fsl.yaml | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/Documentation/devicetree/bindings/arm/fsl.yaml b/Documentation/devicetree/bindings/arm/fsl.yaml index 86876311ec59a3..6f95c17b363c09 100644 --- a/Documentation/devicetree/bindings/arm/fsl.yaml +++ b/Documentation/devicetree/bindings/arm/fsl.yaml @@ -1705,9 +1705,11 @@ properties: - fsl,vf610 - fsl,vf610m4 - - description: Toradex Colibri VF50 Module on Colibri Evaluation Board + - description: Toradex Colibri VF50 Module on a carrier board items: - - const: toradex,vf500-colibri_vf50-on-eval + - enum: + - toradex,vf500-colibri_vf50-on-eval + - toradex,vf500-colibri-vf50-on-iris - const: toradex,vf500-colibri_vf50 - const: fsl,vf500 @@ -1719,9 +1721,11 @@ properties: - phytec,vf610-cosmic # PHYTEC Cosmic/Cosmic+ Board - const: fsl,vf610 - - description: Toradex Colibri VF61 Module on Colibri Evaluation Board + - description: Toradex Colibri VF61 Module on a carrier board items: - - const: toradex,vf610-colibri_vf61-on-eval + - enum: + - toradex,vf610-colibri_vf61-on-eval + - toradex,vf610-colibri-vf61-on-iris - const: toradex,vf610-colibri_vf61 - const: fsl,vf610 From 43d4b5153ae513d03b45a5767aa084b676302c72 Mon Sep 17 00:00:00 2001 From: Mehmet Fide Date: Fri, 14 Aug 2026 09:47:00 +0200 Subject: [PATCH 190/857] ARM: dts: vf: add Iris carrier board support for Colibri VF50 and VF61 The Iris is a small off-the-shelf carrier board for the Colibri family. Describe what the board itself provides: Ethernet, one USB host and one OTG port, the pin header UARTs, i2c with the carrier RTC on it, the PWM pins on the X16 header, SPI and the microSD slot. The host port gets its VBUS through a regulator on SODIMM 129, and the two GPIOs that turn the RS232 transceivers on, SODIMM 102 and 104, are hogged. The touchscreen is an add-on rather than part of the board, so it is left out. Describe an Iris V1.1. A V2.0 carries the same signals on the same pins; what it adds, and what this does not describe, is the regulator on SODIMM 100 that can cut power to the uSD slot. Reviewed-by: Krzysztof Kozlowski Signed-off-by: Mehmet Fide Signed-off-by: Frank Li --- arch/arm/boot/dts/nxp/vf/Makefile | 2 + arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi | 126 ++++++++++++++++++ .../boot/dts/nxp/vf/vf500-colibri-iris.dts | 14 ++ .../boot/dts/nxp/vf/vf610-colibri-iris.dts | 14 ++ 4 files changed, 156 insertions(+) create mode 100644 arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi create mode 100644 arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts create mode 100644 arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts diff --git a/arch/arm/boot/dts/nxp/vf/Makefile b/arch/arm/boot/dts/nxp/vf/Makefile index 0a4a7f9dd43e4e..47be8bc318e285 100644 --- a/arch/arm/boot/dts/nxp/vf/Makefile +++ b/arch/arm/boot/dts/nxp/vf/Makefile @@ -1,8 +1,10 @@ # SPDX-License-Identifier: GPL-2.0 dtb-$(CONFIG_SOC_VF610) += \ vf500-colibri-eval-v3.dtb \ + vf500-colibri-iris.dtb \ vf610-bk4.dtb \ vf610-colibri-eval-v3.dtb \ + vf610-colibri-iris.dtb \ vf610m4-colibri.dtb \ vf610-cosmic.dtb \ vf610m4-cosmic.dtb \ diff --git a/arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi b/arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi new file mode 100644 index 00000000000000..36e34dd766aee0 --- /dev/null +++ b/arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: GPL-2.0-or-later OR MIT +/* + * Copyright 2026 Mehmet Fide + * + * Toradex Iris carrier board V1.1, common part for Colibri VF50 and VF61. + * + * The Colibri modules are pin compatible, so the carrier signals sit on the + * same SODIMM pins as they do for the i.MX modules, whose Iris device trees + * are already upstream. The two SODIMM pins below are taken from the Colibri + * VFxx datasheet (102 -> PTA12 -> PORT0[5], 104 -> PTD28 -> PORT2[2]). + * + * An Iris V2.0 carries the same signals on the same pins, so everything here + * works on it as well. What it adds is a regulator on SODIMM 100 that can cut + * power to the uSD slot, which this file does not describe. + */ + +/ { + chosen { + stdout-path = "serial0:115200n8"; + }; + + reg_usbh_vbus: regulator-usbh-vbus { + compatible = "regulator-fixed"; + pinctrl-names = "default"; + pinctrl-0 = <&pinctrl_usbh1_reg>; + regulator-name = "VCC_USB[1-4]"; + regulator-min-microvolt = <5000000>; + regulator-max-microvolt = <5000000>; + gpio = <&gpio2 19 GPIO_ACTIVE_LOW>; /* SODIMM 129, USBH_PEN */ + }; +}; + +&gpio0 { + /* + * The RS232 transceivers on the carrier are switched on by these two + * lines. Delete the hog to turn a transceiver off from userspace. + * The pin group belongs to the gpio controller rather than to iomuxc, + * which would depend on its own child and log a dependency cycle. + */ + pinctrl-names = "default"; + pinctrl-0 = <&pinctrl_iris_uart1_tx_on>; + + uart1-tx-on-hog { + gpio-hog; + gpios = <5 GPIO_ACTIVE_HIGH>; /* SODIMM 102 */ + output-high; + }; +}; + +&gpio2 { + pinctrl-names = "default"; + /* keep the EXT_IO group from vf-colibri.dtsi alongside ours */ + pinctrl-0 = <&pinctrl_gpio_ext>, <&pinctrl_iris_uart25_tx_on>; + + uart25-tx-on-hog { + gpio-hog; + gpios = <2 GPIO_ACTIVE_HIGH>; /* SODIMM 104 */ + output-high; + }; +}; + +&iomuxc { + pinctrl_iris_uart1_tx_on: irisuart1txongrp { + fsl,pins = < + VF610_PAD_PTA12__GPIO_5 0x22ef + >; + }; + + pinctrl_iris_uart25_tx_on: irisuart25txongrp { + fsl,pins = < + VF610_PAD_PTD28__GPIO_66 0x22ef + >; + }; +}; + +/* Colibri SSP, on the extension connector */ +&dspi1 { + status = "okay"; +}; + +&esdhc1 { + status = "okay"; +}; + +&fec1 { + status = "okay"; +}; + +&pwm0 { + status = "okay"; +}; + +&pwm1 { + status = "okay"; +}; + +&i2c0 { + status = "okay"; + + /* M41T0M6 real time clock on the carrier */ + rtc@68 { + compatible = "st,m41t0"; + reg = <0x68>; + }; +}; + +&uart0 { + status = "okay"; +}; + +&uart1 { + status = "okay"; +}; + +&uart2 { + status = "okay"; +}; + +&usbdev0 { + status = "okay"; +}; + +&usbh1 { + vbus-supply = <®_usbh_vbus>; + status = "okay"; +}; diff --git a/arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts b/arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts new file mode 100644 index 00000000000000..aa91537cab3777 --- /dev/null +++ b/arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts @@ -0,0 +1,14 @@ +// SPDX-License-Identifier: GPL-2.0-or-later OR MIT +/* + * Copyright 2026 Mehmet Fide + */ + +/dts-v1/; +#include "vf500-colibri.dtsi" +#include "vf-colibri-iris.dtsi" + +/ { + model = "Toradex Colibri VF50 on Iris Carrier Board"; + compatible = "toradex,vf500-colibri-vf50-on-iris", "toradex,vf500-colibri_vf50", + "fsl,vf500"; +}; diff --git a/arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts b/arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts new file mode 100644 index 00000000000000..7db5aa490966ba --- /dev/null +++ b/arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts @@ -0,0 +1,14 @@ +// SPDX-License-Identifier: GPL-2.0-or-later OR MIT +/* + * Copyright 2026 Mehmet Fide + */ + +/dts-v1/; +#include "vf610-colibri.dtsi" +#include "vf-colibri-iris.dtsi" + +/ { + model = "Toradex Colibri VF61 on Iris Carrier Board"; + compatible = "toradex,vf610-colibri-vf61-on-iris", "toradex,vf610-colibri_vf61", + "fsl,vf610"; +}; From a8d1c746bf1a67e8339e234d10b43961b3721d0c Mon Sep 17 00:00:00 2001 From: YuXin Xiao Date: Tue, 25 Aug 2026 02:15:59 +0800 Subject: [PATCH 191/857] i2c: acpi: Add DELL0A86 to i2c_acpi_force_100khz_device_ids The ELAN 04F3:3185 touchpad exposed as DELL0A86 on the Dell Inspiron 13 5310 intermittently exhibits excessive smoothing when the I2C bus runs at 400 kHz. During an affected period pointer movement becomes severely sluggish and sticky for tens of seconds. The ACPI firmware configures the touchpad bus for 400 kHz. Add DELL0A86 to i2c_acpi_force_100khz_device_ids so that the bus runs at 100 kHz, as is already done for other touchpads exhibiting the same excessive smoothing problem. With the quirk applied, the kernel reports that the firmware requested 400 kHz and that the bus is forced to 100 kHz. The problem did not recur during extended heavy use, including an S4 hibernate/resume cycle. An independent 2021 report from another Dell Inspiron 13 5310 user describes the same intermittent sticky behavior and identifies the same DELL0A86 / 04F3:3185 touchpad. Link: https://www.reddit.com/r/linuxquestions/comments/nsso5b/help_needed_with_sticky_touchpad_spoiling_brand/ Signed-off-by: YuXin Xiao Assisted-by: Codex:ChatGPT-5.6-Sol Assisted-by: OpenCode:Ox Alpha (x-preview-f-free) Acked-by: Mika Westerberg Signed-off-by: Andi Shyti Link: https://patch.msgid.link/20260824181559.146133-1-xiaoyueyoqwq@gmail.com --- drivers/i2c/i2c-core-acpi.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/i2c/i2c-core-acpi.c b/drivers/i2c/i2c-core-acpi.c index 8f3bdd50186e39..3af0e05511851e 100644 --- a/drivers/i2c/i2c-core-acpi.c +++ b/drivers/i2c/i2c-core-acpi.c @@ -371,6 +371,7 @@ static const struct acpi_device_id i2c_acpi_force_100khz_device_ids[] = { * the device works without issues on Windows at what is expected to be * a 400KHz frequency. The root cause of the issue is not known. */ + { "DELL0A86", 0 }, { "DLL0945", 0 }, { "ELAN0678", 0 }, { "ELAN06FA", 0 }, From bcb621272aa04a116a74ab5fcc9324231ffa3acb Mon Sep 17 00:00:00 2001 From: Kiran K Date: Tue, 25 Aug 2026 22:53:00 +0530 Subject: [PATCH 192/857] Bluetooth: btintel_pcie: Clear automask on spurious interrupts On spurious interrupt where the TX and RX causes are not set, driver was not clearing the auto mask which can block all the interrupts. Driver needs to clear the automask even if no causes are set. Fixes: c2b636b3f788 ("Bluetooth: btintel_pcie: Add support for PCIe transport") Signed-off-by: Kiran K Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btintel_pcie.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/drivers/bluetooth/btintel_pcie.c b/drivers/bluetooth/btintel_pcie.c index baa621b3fef931..30923eaabed717 100644 --- a/drivers/bluetooth/btintel_pcie.c +++ b/drivers/bluetooth/btintel_pcie.c @@ -2051,6 +2051,9 @@ static irqreturn_t btintel_pcie_irq_msix_handler(int irq, void *dev_id) if (unlikely(!(intr_fh | intr_hw))) { /* Ignore interrupt, inta == 0 */ + bt_warn_ratelimited("Bluetooth: btintel_pcie: Received spurious interrupt\n"); + btintel_pcie_wr_reg32(data, BTINTEL_PCIE_CSR_MSIX_AUTOMASK_ST, + BIT(entry->entry)); return IRQ_NONE; } From 4ed7013b77f56a3bab06b77aa25754d832b22870 Mon Sep 17 00:00:00 2001 From: Alex Tran Date: Wed, 26 Aug 2026 10:19:26 -0700 Subject: [PATCH 193/857] i2c: busses: Use pm_runtime_resume_and_get() Utilize the provided pm_runtime_resume_and_get api to increase the usage count and call the rpm resume callback. Upon failure, the function takes care of calling pm_runtime_put_noidle. Remove the explicit call to put no idle. No functional change added. Signed-off-by: Alex Tran Reviewed-by: Mukesh Kumar Savaliya Signed-off-by: Andi Shyti Link: https://patch.msgid.link/20260826-i2c-qcom-geni-pm-runtime-resume-get-v1-1-25ee55d1f0c8@oss.qualcomm.com --- drivers/i2c/busses/i2c-qcom-geni.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/drivers/i2c/busses/i2c-qcom-geni.c b/drivers/i2c/busses/i2c-qcom-geni.c index 658636c1ee0e26..d5f4de3e0a9de2 100644 --- a/drivers/i2c/busses/i2c-qcom-geni.c +++ b/drivers/i2c/busses/i2c-qcom-geni.c @@ -963,10 +963,9 @@ static int geni_i2c_xfer(struct i2c_adapter *adap, struct geni_i2c_dev *gi2c = i2c_get_adapdata(adap); int ret; - ret = pm_runtime_get_sync(gi2c->se.dev); + ret = pm_runtime_resume_and_get(gi2c->se.dev); if (ret < 0) { dev_err(gi2c->se.dev, "error turning SE resources:%d\n", ret); - pm_runtime_put_noidle(gi2c->se.dev); /* Set device in suspended since resume failed */ pm_runtime_set_suspended(gi2c->se.dev); return ret; From 3cae804dd2a459b18a130f4f97d9d8598eb862b4 Mon Sep 17 00:00:00 2001 From: Krzysztof Kozlowski Date: Wed, 26 Aug 2026 12:30:43 +0200 Subject: [PATCH 194/857] dt-bindings: i2c: xlnx,xps-iic-2.00.a: Drop bouncing mocean-labs.com Address info@mocean-labs.com bounces permanently (reason: 550 Host unknown), so switch the maintainer to Michal Simek. Signed-off-by: Krzysztof Kozlowski Acked-by: Conor Dooley Acked-by: Michal Simek Signed-off-by: Andi Shyti Link: https://patch.msgid.link/20260826103042.147735-2-krzysztof.kozlowski@oss.qualcomm.com --- Documentation/devicetree/bindings/i2c/xlnx,xps-iic-2.00.a.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/devicetree/bindings/i2c/xlnx,xps-iic-2.00.a.yaml b/Documentation/devicetree/bindings/i2c/xlnx,xps-iic-2.00.a.yaml index 658ae92fa86df4..d1f26d5015f790 100644 --- a/Documentation/devicetree/bindings/i2c/xlnx,xps-iic-2.00.a.yaml +++ b/Documentation/devicetree/bindings/i2c/xlnx,xps-iic-2.00.a.yaml @@ -7,7 +7,7 @@ $schema: http://devicetree.org/meta-schemas/core.yaml# title: Xilinx IIC controller maintainers: - - info@mocean-labs.com + - Michal Simek allOf: - $ref: /schemas/i2c/i2c-controller.yaml# From 797dcf32256f2705cfcc8cceb08cb95c15e227f3 Mon Sep 17 00:00:00 2001 From: Geert Uytterhoeven Date: Fri, 21 Aug 2026 12:03:30 +0200 Subject: [PATCH 195/857] i2c: bcm2835: Make sure clk_init_data is fully initialized The clk_init_data structure contains several mutually-exclusive members for different methods to specify the possible parents of a clock, prompting drivers to initialize only the members they need. However, not initializing all members may cause subtle issues, which are only exposed when CONFIG_INIT_STACK_ALL_PATTERN or CONFIG_INIT_STACK_NONE is enabled. Make sure all members are fully initialized, to avoid such bugs, and to prevent future breakage when converting drivers to a different method for specifying the parents. Signed-off-by: Geert Uytterhoeven Reviewed-by: Florian Fainelli Reviewed-by: Brian Masney Signed-off-by: Andi Shyti Link: https://patch.msgid.link/9f82f37e6d6c069cd44326bcd5e5a2a8069a13a9.1787239980.git.geert+renesas@glider.be --- drivers/i2c/busses/i2c-bcm2835.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/i2c/busses/i2c-bcm2835.c b/drivers/i2c/busses/i2c-bcm2835.c index 0d7e2654a534e9..b30f7e0e45f160 100644 --- a/drivers/i2c/busses/i2c-bcm2835.c +++ b/drivers/i2c/busses/i2c-bcm2835.c @@ -166,7 +166,7 @@ static struct clk *bcm2835_i2c_register_div(struct device *dev, struct clk *mclk, struct bcm2835_i2c_dev *i2c_dev) { - struct clk_init_data init; + struct clk_init_data init = {}; struct clk_bcm2835_i2c *priv; char name[32]; const char *mclk_name; From d1b432080f57a2799a1b2d03531f7990be044414 Mon Sep 17 00:00:00 2001 From: Namjae Jeon Date: Wed, 26 Aug 2026 23:19:58 +0900 Subject: [PATCH 196/857] exfat: map allocated extents for swap activation exFAT reports allocated ranges beyond valid_size as IOMAP_HOLE when IOMAP_REPORT is set. iomap_swapfile_activate() also uses IOMAP_REPORT while collecting physical extents for a swap file, so it treats the preallocated tail as unallocated and rejects the file with -EINVAL. Use swap-specific iomap operations that suppress IOMAP_REPORT before mapping the file. This reports physically allocated ranges as IOMAP_UNWRITTEN during swap activation without changing the byte-accurate SEEK_HOLE and SEEK_DATA behavior of the regular iomap operations. Fixes: 03a43677ca91 ("exfat: add swap_activate support") Tested-by: Petr Vorel Signed-off-by: Namjae Jeon --- fs/exfat/iomap.c | 28 +++++++++++++++++++++++++++- 1 file changed, 27 insertions(+), 1 deletion(-) diff --git a/fs/exfat/iomap.c b/fs/exfat/iomap.c index 8911aa84a730ea..533cdb4f0929b8 100644 --- a/fs/exfat/iomap.c +++ b/fs/exfat/iomap.c @@ -157,6 +157,27 @@ const struct iomap_ops exfat_iomap_ops = { .iomap_next = exfat_iomap_next, }; +#ifdef CONFIG_SWAP +static int exfat_swap_iomap_begin(struct inode *inode, loff_t offset, + loff_t length, unsigned int flags, struct iomap *iomap, + struct iomap *srcmap) +{ + /* + * Swap activation needs the physical mappings of preallocated + * ranges. Do not report the VDL tail as a hole. + */ + return __exfat_iomap_begin(inode, offset, length, + flags & ~IOMAP_REPORT, iomap, false); +} + +static DEFINE_IOMAP_ITER_NEXT(exfat_swap_iomap_next, + exfat_swap_iomap_begin); + +static const struct iomap_ops exfat_swap_iomap_ops = { + .iomap_next = exfat_swap_iomap_next, +}; +#endif + /* * exfat_write_iomap_end - Update the state after write * @@ -275,5 +296,10 @@ const struct iomap_read_ops exfat_iomap_bio_read_ops = { int exfat_iomap_swap_activate(struct swap_info_struct *sis, struct file *file, sector_t *span) { - return iomap_swapfile_activate(sis, file, span, &exfat_iomap_ops); +#ifdef CONFIG_SWAP + return iomap_swapfile_activate(sis, file, span, + &exfat_swap_iomap_ops); +#else + return -EIO; +#endif } From 3df1021c7eca82bd95b83b8ba3dc3f596af51ec6 Mon Sep 17 00:00:00 2001 From: Yang Wen Date: Fri, 28 Aug 2026 20:51:54 +0800 Subject: [PATCH 197/857] exfat: validate vendor allocation directory entries The exfat entry validator checks the type and ordering of benign secondary entries, but does not validate Vendor Allocation entry fields. In addition, exfat_find() reads only the first two entries, allowing malformed trailing Vendor Allocation entries to bypass validation. Validate the allocation flags, VendorGuid, FirstCluster, DataLength, and NoFatChain extent. Read the complete entry set during lookup so malformed Vendor Allocation entries are rejected. Signed-off-by: Yang Wen Signed-off-by: Namjae Jeon --- fs/exfat/dir.c | 49 +++++++++++++++++++++++++++++++++++++++++++++--- fs/exfat/namei.c | 3 ++- 2 files changed, 48 insertions(+), 4 deletions(-) diff --git a/fs/exfat/dir.c b/fs/exfat/dir.c index fe73b1380c5d87..fa9abedf4d8494 100644 --- a/fs/exfat/dir.c +++ b/fs/exfat/dir.c @@ -678,11 +678,54 @@ enum exfat_validate_dentry_mode { ES_MODE_GET_BENIGN_SEC_ENTRY, }; -static bool exfat_validate_entry(unsigned int type, - enum exfat_validate_dentry_mode *mode) +static bool exfat_validate_vendor_alloc(struct super_block *sb, + struct exfat_dentry *ep) +{ + struct exfat_sb_info *sbi = EXFAT_SB(sb); + u8 flags = ep->dentry.vendor_alloc.flags; + u32 start_clu = le32_to_cpu(ep->dentry.vendor_alloc.start_clu); + u64 size = le64_to_cpu(ep->dentry.vendor_alloc.size); + u64 max_size = exfat_cluster_to_bytes(sbi, + (u64)EXFAT_DATA_CLUSTER_COUNT(sbi)); + u64 num_clusters; + + /* AllocationPossible is required for Vendor Allocation entries. */ + if (!(flags & ALLOC_POSSIBLE)) + return false; + + /* The null GUID does not identify a valid vendor allocation. */ + if (!memchr_inv(ep->dentry.vendor_alloc.vendor_guid, 0, + sizeof(ep->dentry.vendor_alloc.vendor_guid))) + return false; + + if (!start_clu) + return !size && !(flags & (ALLOC_NO_FAT_CHAIN ^ ALLOC_FAT_CHAIN)); + + if (!is_valid_cluster(sbi, start_clu) || size > max_size) + return false; + + if ((flags & ALLOC_NO_FAT_CHAIN) == ALLOC_NO_FAT_CHAIN) { + if (!size) + return false; + + num_clusters = DIV_ROUND_UP_ULL(size, sbi->cluster_size); + if (num_clusters > sbi->num_clusters - start_clu) + return false; + } + + return true; +} + +static bool exfat_validate_entry(struct super_block *sb, + struct exfat_dentry *ep, enum exfat_validate_dentry_mode *mode) { + unsigned int type = exfat_get_entry_type(ep); + if (type == TYPE_UNUSED || type == TYPE_DELETED) return false; + if (type == TYPE_VENDOR_ALLOC && + !exfat_validate_vendor_alloc(sb, ep)) + return false; switch (*mode) { case ES_MODE_GET_FILE_ENTRY: @@ -836,7 +879,7 @@ int exfat_get_dentry_set(struct exfat_entry_set_cache *es, /* validate cached dentries */ for (i = ES_IDX_STREAM; i < es->num_entries; i++) { ep = exfat_get_dentry_cached(es, i); - if (!exfat_validate_entry(exfat_get_entry_type(ep), &mode)) + if (!exfat_validate_entry(sb, ep, &mode)) goto put_es; } return 0; diff --git a/fs/exfat/namei.c b/fs/exfat/namei.c index a4dc83b5949c46..80f60e80786ea6 100644 --- a/fs/exfat/namei.c +++ b/fs/exfat/namei.c @@ -645,7 +645,8 @@ static int exfat_find(struct inode *dir, const struct qstr *qname, info->entry = dentry; info->num_subdirs = 0; - if (exfat_get_dentry_set(&es, sb, &cdir, dentry, ES_2_ENTRIES)) + /* Validate the complete set, including recognized benign entries. */ + if (exfat_get_dentry_set(&es, sb, &cdir, dentry, ES_ALL_ENTRIES)) return -EIO; ep = exfat_get_dentry_cached(&es, ES_IDX_FILE); ep2 = exfat_get_dentry_cached(&es, ES_IDX_STREAM); From 7e30d1a901afdd86bbd2f8e4282b482a5dcc46f5 Mon Sep 17 00:00:00 2001 From: Geert Uytterhoeven Date: Wed, 19 Aug 2026 21:05:17 +0200 Subject: [PATCH 198/857] hwmon: (ltc4282) Make sure clk_init_data is fully initialized The clk_init_data structure contains several mutually-exclusive members for different methods to specify the possible parents of a clock, prompting drivers to initialize only the members they need. However, not initializing all members may cause subtle issues, which are only exposed when CONFIG_INIT_STACK_ALL_PATTERN or CONFIG_INIT_STACK_NONE is enabled. ltc428_clk_provider_setup() does not fill in any parent clocks, and assumes that init.num_parents is NULL. However, the latter in uninitialized, and thus may cause a crash. Make sure all members are fully initialized, to fix such bugs, and to avoid future breakage when converting drivers to a different method for specifying the parents. Fixes: cbc29538dbf7d740 ("hwmon: Add driver for LTC4282") Signed-off-by: Geert Uytterhoeven Link: https://patch.msgid.link/8ec3c5cbd2df675a938f090470f5da5f22008517.1787165329.git.geert+renesas@glider.be Reviewed-by: Brian Masney Signed-off-by: Guenter Roeck --- drivers/hwmon/ltc4282.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/hwmon/ltc4282.c b/drivers/hwmon/ltc4282.c index b1675dc5b3c7fd..54ba4b8542e94b 100644 --- a/drivers/hwmon/ltc4282.c +++ b/drivers/hwmon/ltc4282.c @@ -1106,7 +1106,7 @@ static const struct clk_ops ltc4282_ops = { static int ltc428_clk_provider_setup(struct ltc4282_state *st, struct device *dev) { - struct clk_init_data init; + struct clk_init_data init = {}; int ret; if (!IS_ENABLED(CONFIG_COMMON_CLK)) From e8837d36df01be2595d73a8c815592055f6c97c0 Mon Sep 17 00:00:00 2001 From: Nikhil Gurudasani Date: Wed, 19 Aug 2026 23:37:01 +0530 Subject: [PATCH 199/857] hwmon: (mcp9982) Propagate one-shot polling errors When a device is in standby, the driver starts a one-shot conversion and polls the BUSY flag before reading temperature, alarm, or fault data. The poll result is currently ignored. Therefore, a timeout or a status-register read failure can be hidden by a later successful read, causing stale data to be returned as valid. Return the polling error before reading the requested attribute. Fixes: e2fe950f34e5 ("hwmon: add support for MCP998X") Cc: stable@vger.kernel.org Signed-off-by: Nikhil Gurudasani Link: https://patch.msgid.link/20260819180701.34797-1-nikhilgurudasani314@gmail.com Signed-off-by: Guenter Roeck --- drivers/hwmon/mcp9982.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/drivers/hwmon/mcp9982.c b/drivers/hwmon/mcp9982.c index 9e19e2697e25ff..3918dc36c9463c 100644 --- a/drivers/hwmon/mcp9982.c +++ b/drivers/hwmon/mcp9982.c @@ -395,6 +395,8 @@ static int mcp9982_read(struct device *dev, enum hwmon_sensor_types type, u32 at reg_status, !(reg_status & MCP9982_STATUS_BUSY), MCP9982_WAKE_UP_TIME_US, MCP9982_WAKE_UP_TIME_US * 10); + if (ret) + return ret; break; } break; From ecc97fcee702084df9d7a291a013f72a6193bc55 Mon Sep 17 00:00:00 2001 From: Fan Wu Date: Wed, 19 Aug 2026 03:33:17 +0000 Subject: [PATCH 200/857] hwmon: (gpio-fan) Fix use-after-free in alarm work fan_alarm_irq_handler() queues fan_data->alarm_work, but nothing cancels it. fan_alarm_notify() dereferences fan_data and its hwmon device. On unbind, devres frees the interrupt, which only waits for the handler itself, and then releases the hwmon device and fan_data, so a pending fan_alarm_notify() can run after those frees. Replace INIT_WORK() with devm_work_autocancel(), registered before devm_request_irq(). The devres cleanup then frees the interrupt first, so no new work can be queued, and cancels the work while fan_data and the hwmon device are still alive. This issue was found by an in-house static analysis tool. Fixes: d6fe1360f42e ("hwmon: add generic GPIO fan driver") Cc: stable@vger.kernel.org Assisted-by: Codex:gpt-5.6 Signed-off-by: Fan Wu Link: https://patch.msgid.link/20260819033317.446191-1-fanwu01@zju.edu.cn Signed-off-by: Guenter Roeck --- drivers/hwmon/gpio-fan.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/drivers/hwmon/gpio-fan.c b/drivers/hwmon/gpio-fan.c index 084828e1e2817b..7f36e5f6f22308 100644 --- a/drivers/hwmon/gpio-fan.c +++ b/drivers/hwmon/gpio-fan.c @@ -12,6 +12,7 @@ #include #include #include +#include #include #include #include @@ -84,6 +85,7 @@ static DEVICE_ATTR_RO(fan1_alarm); static int fan_alarm_init(struct gpio_fan_data *fan_data) { int alarm_irq; + int err; struct device *dev = fan_data->dev; /* @@ -94,7 +96,11 @@ static int fan_alarm_init(struct gpio_fan_data *fan_data) if (alarm_irq <= 0) return 0; - INIT_WORK(&fan_data->alarm_work, fan_alarm_notify); + err = devm_work_autocancel(dev, &fan_data->alarm_work, + fan_alarm_notify); + if (err) + return err; + irq_set_irq_type(alarm_irq, IRQ_TYPE_EDGE_BOTH); return devm_request_irq(dev, alarm_irq, fan_alarm_irq_handler, IRQF_SHARED, "GPIO fan alarm", fan_data); From 40e979885aa2ef4dd5fd7022350343215e1e9533 Mon Sep 17 00:00:00 2001 From: Guenter Roeck Date: Thu, 20 Aug 2026 21:40:48 -0700 Subject: [PATCH 201/857] Documentation/hwmon: Document hwmon_notify_event() The hwmon core provides hwmon_notify_event() for drivers to report events such as alarm or fault conditions to userspace via sysfs notifications and uevents, as well as to the thermal subsystem for temperature sensors. However, this function is not documented in the hwmon kernel API guide. Add the function prototype and description of hwmon_notify_event() to Documentation/hwmon/hwmon-kernel-api.rst. Cc: Kalesh AP Reviewed-by: Kalesh AP Fixes: 1597b374af222 ("hwmon: Add notification support") Signed-off-by: Guenter Roeck --- Documentation/hwmon/hwmon-kernel-api.rst | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/Documentation/hwmon/hwmon-kernel-api.rst b/Documentation/hwmon/hwmon-kernel-api.rst index 9fcde32a140df6..c3eb433a78f61e 100644 --- a/Documentation/hwmon/hwmon-kernel-api.rst +++ b/Documentation/hwmon/hwmon-kernel-api.rst @@ -42,6 +42,9 @@ register/unregister functions:: char *devm_hwmon_sanitize_name(struct device *dev, const char *name); + int hwmon_notify_event(struct device *dev, enum hwmon_sensor_types type, + u32 attr, int channel); + void hwmon_lock(struct device *dev); void hwmon_unlock(struct device *dev); @@ -90,6 +93,18 @@ implemented in the driver, or debugfs functions, hwmon_lock() and hwmon_unlock() can be used to ensure that calls to those functions are serialized. Those functions also support guard() and scoped_guard() variants. +Drivers can call hwmon_notify_event() to notify userspace and the thermal +subsystem when a hardware monitoring event (such as an alarm or a fault +condition) occurs or clears. The parameters are the hwmon device, the sensor +type, the attribute identifier associated with the event (such as +hwmon_temp_max_alarm or hwmon_fan_fault), and the sensor channel number. +hwmon_notify_event() generates a sysfs event (calling sysfs_notify()) and a +udev event with the attribute name passed in the NAME environment property +(e.g., "NAME=temp1_max_alarm"). If the event is for a temperature sensor and +the sensor is attached to a thermal zone, it also notifies the thermal +subsystem to update the thermal zone. hwmon_notify_event() returns 0 on +success or a negative error code on failure. + Using devm_hwmon_device_register_with_info() -------------------------------------------- From 783007669c34c8df91994d90745d1bde9e7ff21b Mon Sep 17 00:00:00 2001 From: Guenter Roeck Date: Thu, 20 Aug 2026 10:51:50 -0700 Subject: [PATCH 202/857] hwmon: Fix potential UAF in pec_store Sashiko reports: In pec_store(), a guard(mutex)(&hwdev->lock) is taken. If the chip write operation returns an error other than -EOPNOTSUPP, the code jumps to the put label, which calls put_device(hdev). If this drops the final reference, the device is freed. When the function then returns, the guard cleanup function runs and attempts to unlock the freed mutex. Use scoped_guard() instead of guard() to avoid the problem. Fixes: 3ad2a7b9b15d5 ("hwmon: Serialize accesses in hwmon core") Signed-off-by: Guenter Roeck --- drivers/hwmon/hwmon.c | 21 ++++++++++----------- 1 file changed, 10 insertions(+), 11 deletions(-) diff --git a/drivers/hwmon/hwmon.c b/drivers/hwmon/hwmon.c index 41755910a25a03..3e65fc6d25ebcd 100644 --- a/drivers/hwmon/hwmon.c +++ b/drivers/hwmon/hwmon.c @@ -371,18 +371,17 @@ static ssize_t pec_store(struct device *dev, const struct device_attribute *deva * handling is not required. */ hwdev = to_hwmon_device(hdev); - guard(mutex)(&hwdev->lock); - if (hwdev->chip->ops->write) { - err = hwdev->chip->ops->write(hdev, hwmon_chip, hwmon_chip_pec, 0, val); - if (err && err != -EOPNOTSUPP) - goto put; + scoped_guard(mutex, &hwdev->lock) { + if (hwdev->chip->ops->write) { + err = hwdev->chip->ops->write(hdev, hwmon_chip, hwmon_chip_pec, 0, val); + if (err && err != -EOPNOTSUPP) + goto put; + } + if (!val) + client->flags &= ~I2C_CLIENT_PEC; + else + client->flags |= I2C_CLIENT_PEC; } - - if (!val) - client->flags &= ~I2C_CLIENT_PEC; - else - client->flags |= I2C_CLIENT_PEC; - err = count; put: put_device(hdev); From 5c427104264b4a9dfc036e2f00f2b202f89b67e6 Mon Sep 17 00:00:00 2001 From: Jared Kangas Date: Thu, 20 Aug 2026 06:09:21 -0700 Subject: [PATCH 203/857] hwmon: (ina2xx) Acquire hwmon_lock in shunt_resistor_show() shunt_resistor_store() currently acquires hwmon_lock to set data->rshunt, but the corresponding access in shunt_resistor_show() is unprotected. Acquire the lock in shunt_resistor_show() as well to ensure proper synchronization. Fixes: 3ad867001c91 ("hwmon: (ina2xx) fix sysfs shunt resistor read access") Reported-by: Sashiko Closes: https://lore.kernel.org/all/20260729162836.89BDF1F00A3A@smtp.kernel.org/ Signed-off-by: Jared Kangas Link: https://patch.msgid.link/20260820-upstream-ina2xx-in0-curr1-alarms-v2-1-fdce35abc41e@redhat.com Signed-off-by: Guenter Roeck --- drivers/hwmon/ina2xx.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/drivers/hwmon/ina2xx.c b/drivers/hwmon/ina2xx.c index 5c6dc2c370d8ca..f1f988d099a541 100644 --- a/drivers/hwmon/ina2xx.c +++ b/drivers/hwmon/ina2xx.c @@ -883,8 +883,12 @@ static ssize_t shunt_resistor_show(struct device *dev, struct device_attribute *da, char *buf) { struct ina2xx_data *data = dev_get_drvdata(dev); + long rshunt; - return sysfs_emit(buf, "%li\n", data->rshunt); + scoped_guard(hwmon_lock, dev) { + rshunt = data->rshunt; + } + return sysfs_emit(buf, "%li\n", rshunt); } static ssize_t shunt_resistor_store(struct device *dev, From a40a97b27cd6bd8465937439418964468f784bc4 Mon Sep 17 00:00:00 2001 From: Guenter Roeck Date: Thu, 20 Aug 2026 22:19:13 -0700 Subject: [PATCH 204/857] hwmon: Ensure that 'dev' passed to hwmon_notify_event() is a hwmon device The device parameter of hwmon_notify_event() must be a hardware monitoring device. Since this is easy to get wrong, and since passing a non-hwmon device may result in a crash, generate a warning traceback and abort if a wrong device class is passed as parameter. Signed-off-by: Guenter Roeck --- drivers/hwmon/hwmon.c | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/drivers/hwmon/hwmon.c b/drivers/hwmon/hwmon.c index 3e65fc6d25ebcd..10d2df3efdfaee 100644 --- a/drivers/hwmon/hwmon.c +++ b/drivers/hwmon/hwmon.c @@ -318,6 +318,11 @@ static int hwmon_attr_base(enum hwmon_sensor_types type) return 1; } +static bool is_hwmon_device(struct device *dev) +{ + return dev->class == &hwmon_class; +} + #if IS_REACHABLE(CONFIG_I2C) /* @@ -338,7 +343,7 @@ static int hwmon_attr_base(enum hwmon_sensor_types type) static int hwmon_match_device(struct device *dev, const void *data) { - return dev->class == &hwmon_class; + return is_hwmon_device(dev); } static ssize_t pec_show(struct device *dev, const struct device_attribute *dummy, @@ -781,6 +786,9 @@ int hwmon_notify_event(struct device *dev, enum hwmon_sensor_types type, const char *template; int base; + if (WARN(!is_hwmon_device(dev), "%s is not a hardware monitoring device\n", + dev_name(dev))) + return -EINVAL; if (type >= ARRAY_SIZE(__templates)) return -EINVAL; if (attr >= __templates_size[type]) From a2791f66b0c3bb751624dafb2029da1af4fb2ea7 Mon Sep 17 00:00:00 2001 From: Jared Kangas Date: Thu, 20 Aug 2026 06:09:22 -0700 Subject: [PATCH 205/857] hwmon: (ina2xx) Parameterize ina2xx_data in ina226_alert_read() Mirror ina226_alert_limit_read/write and use struct ina2xx_data instead of struct regmap in ina226_alert_read's parameters. Signed-off-by: Jared Kangas Link: https://patch.msgid.link/20260820-upstream-ina2xx-in0-curr1-alarms-v2-2-fdce35abc41e@redhat.com Signed-off-by: Guenter Roeck --- drivers/hwmon/ina2xx.c | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/drivers/hwmon/ina2xx.c b/drivers/hwmon/ina2xx.c index f1f988d099a541..0fd17a6d4f90d5 100644 --- a/drivers/hwmon/ina2xx.c +++ b/drivers/hwmon/ina2xx.c @@ -498,12 +498,12 @@ static int ina2xx_chip_read(struct device *dev, u32 attr, long *val) return 0; } -static int ina226_alert_read(struct regmap *regmap, u32 mask, long *val) +static int ina226_alert_read(struct ina2xx_data *data, u32 mask, long *val) { unsigned int regval; int ret; - ret = regmap_read_bypassed(regmap, INA226_MASK_ENABLE, ®val); + ret = regmap_read_bypassed(data->regmap, INA226_MASK_ENABLE, ®val); if (ret) return ret; @@ -538,9 +538,9 @@ static int ina2xx_in_read(struct device *dev, u32 attr, int channel, long *val) return ina226_alert_limit_read(data, over_voltage_mask, voltage_reg, val); case hwmon_in_lcrit_alarm: - return ina226_alert_read(regmap, under_voltage_mask, val); + return ina226_alert_read(data, under_voltage_mask, val); case hwmon_in_crit_alarm: - return ina226_alert_read(regmap, over_voltage_mask, val); + return ina226_alert_read(data, over_voltage_mask, val); default: return -EOPNOTSUPP; } @@ -597,7 +597,7 @@ static int ina2xx_power_read(struct device *dev, u32 attr, long *val) return ina226_alert_limit_read(data, INA226_POWER_OVER_LIMIT_MASK, INA2XX_POWER, val); case hwmon_power_crit_alarm: - return ina226_alert_read(data->regmap, INA226_POWER_OVER_LIMIT_MASK, val); + return ina226_alert_read(data, INA226_POWER_OVER_LIMIT_MASK, val); default: return -EOPNOTSUPP; } @@ -639,9 +639,9 @@ static int ina2xx_curr_read(struct device *dev, u32 attr, long *val) return ina226_alert_limit_read(data, INA226_SHUNT_OVER_VOLTAGE_MASK, INA2XX_CURRENT, val); case hwmon_curr_lcrit_alarm: - return ina226_alert_read(regmap, INA226_SHUNT_UNDER_VOLTAGE_MASK, val); + return ina226_alert_read(data, INA226_SHUNT_UNDER_VOLTAGE_MASK, val); case hwmon_curr_crit_alarm: - return ina226_alert_read(regmap, INA226_SHUNT_OVER_VOLTAGE_MASK, val); + return ina226_alert_read(data, INA226_SHUNT_OVER_VOLTAGE_MASK, val); default: return -EOPNOTSUPP; } From 38f381092cd634828e2f6f892eba8e9ccd99e38c Mon Sep 17 00:00:00 2001 From: Jared Kangas Date: Thu, 20 Aug 2026 06:09:23 -0700 Subject: [PATCH 206/857] hwmon: (ina2xx) Replace masks with enum in alert functions Instead of passing an explicit mask to alert/limit functions like ina226_alert_read(), introduce an enum ina2xx_alert_type that can be converted to a mask internally. This semantically separates current from shunt voltage in helpers that use function masks, which previously saw the same mask for the two functions. Signed-off-by: Jared Kangas Link: https://patch.msgid.link/20260820-upstream-ina2xx-in0-curr1-alarms-v2-3-fdce35abc41e@redhat.com Signed-off-by: Guenter Roeck --- drivers/hwmon/ina2xx.c | 90 +++++++++++++++++++++++++++++++----------- 1 file changed, 67 insertions(+), 23 deletions(-) diff --git a/drivers/hwmon/ina2xx.c b/drivers/hwmon/ina2xx.c index 0fd17a6d4f90d5..75e97e30bdcd4d 100644 --- a/drivers/hwmon/ina2xx.c +++ b/drivers/hwmon/ina2xx.c @@ -129,6 +129,17 @@ enum ina2xx_ids { sy24655 }; +enum ina2xx_alert_type { + INA2XX_ALERT_NONE, + INA2XX_ALERT_CURRENT_LOW, + INA2XX_ALERT_CURRENT_HIGH, + INA2XX_ALERT_POWER_HIGH, + INA2XX_ALERT_BUS_VOLTAGE_LOW, + INA2XX_ALERT_BUS_VOLTAGE_HIGH, + INA2XX_ALERT_SHUNT_VOLTAGE_LOW, + INA2XX_ALERT_SHUNT_VOLTAGE_HIGH, +}; + struct ina2xx_config { u16 config_default; bool has_alerts; /* chip supports alerts and limits */ @@ -428,16 +439,43 @@ static u16 ina226_alert_to_reg(struct ina2xx_data *data, int reg, long val) } } -static int ina226_alert_limit_read(struct ina2xx_data *data, u32 mask, int reg, long *val) +static u32 ina2xx_alert_type_to_mask(enum ina2xx_alert_type alert) +{ + switch (alert) { + case INA2XX_ALERT_CURRENT_LOW: + case INA2XX_ALERT_SHUNT_VOLTAGE_LOW: + return INA226_SHUNT_UNDER_VOLTAGE_MASK; + case INA2XX_ALERT_CURRENT_HIGH: + case INA2XX_ALERT_SHUNT_VOLTAGE_HIGH: + return INA226_SHUNT_OVER_VOLTAGE_MASK; + case INA2XX_ALERT_BUS_VOLTAGE_LOW: + return INA226_BUS_UNDER_VOLTAGE_MASK; + case INA2XX_ALERT_BUS_VOLTAGE_HIGH: + return INA226_BUS_OVER_VOLTAGE_MASK; + case INA2XX_ALERT_POWER_HIGH: + return INA226_POWER_OVER_LIMIT_MASK; + case INA2XX_ALERT_NONE: + return 0; + default: + /* programmer error */ + WARN_ON_ONCE(1); + return 0; + } +} + +static int ina226_alert_limit_read(struct ina2xx_data *data, enum ina2xx_alert_type alert, + int reg, long *val) { struct regmap *regmap = data->regmap; int regval; + u32 mask; int ret; ret = regmap_read(regmap, INA226_MASK_ENABLE, ®val); if (ret) return ret; + mask = ina2xx_alert_type_to_mask(alert); if (regval & mask) { ret = regmap_read(regmap, INA226_ALERT_LIMIT, ®val); if (ret) @@ -449,9 +487,11 @@ static int ina226_alert_limit_read(struct ina2xx_data *data, u32 mask, int reg, return 0; } -static int ina226_alert_limit_write(struct ina2xx_data *data, u32 mask, int reg, long val) +static int ina226_alert_limit_write(struct ina2xx_data *data, enum ina2xx_alert_type alert, + int reg, long val) { struct regmap *regmap = data->regmap; + u32 mask; int ret; if (val < 0) @@ -472,9 +512,11 @@ static int ina226_alert_limit_write(struct ina2xx_data *data, u32 mask, int reg, if (ret < 0) return ret; - if (val) + if (val) { + mask = ina2xx_alert_type_to_mask(alert); return regmap_update_bits(regmap, INA226_MASK_ENABLE, INA226_ALERT_CONFIG_MASK, mask); + } return 0; } @@ -498,15 +540,17 @@ static int ina2xx_chip_read(struct device *dev, u32 attr, long *val) return 0; } -static int ina226_alert_read(struct ina2xx_data *data, u32 mask, long *val) +static int ina226_alert_read(struct ina2xx_data *data, enum ina2xx_alert_type alert, long *val) { unsigned int regval; + u32 mask; int ret; ret = regmap_read_bypassed(data->regmap, INA226_MASK_ENABLE, ®val); if (ret) return ret; + mask = ina2xx_alert_type_to_mask(alert); *val = (regval & mask) && (regval & INA226_ALERT_FUNCTION_FLAG); return 0; @@ -515,10 +559,10 @@ static int ina226_alert_read(struct ina2xx_data *data, u32 mask, long *val) static int ina2xx_in_read(struct device *dev, u32 attr, int channel, long *val) { int voltage_reg = channel ? INA2XX_BUS_VOLTAGE : INA2XX_SHUNT_VOLTAGE; - u32 under_voltage_mask = channel ? INA226_BUS_UNDER_VOLTAGE_MASK - : INA226_SHUNT_UNDER_VOLTAGE_MASK; - u32 over_voltage_mask = channel ? INA226_BUS_OVER_VOLTAGE_MASK - : INA226_SHUNT_OVER_VOLTAGE_MASK; + enum ina2xx_alert_type under_voltage_alert = channel ? INA2XX_ALERT_BUS_VOLTAGE_LOW + : INA2XX_ALERT_SHUNT_VOLTAGE_LOW; + enum ina2xx_alert_type over_voltage_alert = channel ? INA2XX_ALERT_BUS_VOLTAGE_HIGH + : INA2XX_ALERT_SHUNT_VOLTAGE_HIGH; struct ina2xx_data *data = dev_get_drvdata(dev); struct regmap *regmap = data->regmap; unsigned int regval; @@ -532,15 +576,15 @@ static int ina2xx_in_read(struct device *dev, u32 attr, int channel, long *val) *val = ina2xx_get_value(data, voltage_reg, regval); break; case hwmon_in_lcrit: - return ina226_alert_limit_read(data, under_voltage_mask, + return ina226_alert_limit_read(data, under_voltage_alert, voltage_reg, val); case hwmon_in_crit: - return ina226_alert_limit_read(data, over_voltage_mask, + return ina226_alert_limit_read(data, over_voltage_alert, voltage_reg, val); case hwmon_in_lcrit_alarm: - return ina226_alert_read(data, under_voltage_mask, val); + return ina226_alert_read(data, under_voltage_alert, val); case hwmon_in_crit_alarm: - return ina226_alert_read(data, over_voltage_mask, val); + return ina226_alert_read(data, over_voltage_alert, val); default: return -EOPNOTSUPP; } @@ -594,10 +638,10 @@ static int ina2xx_power_read(struct device *dev, u32 attr, long *val) case hwmon_power_average: return sy24655_average_power_read(data, SY24655_EIN, val); case hwmon_power_crit: - return ina226_alert_limit_read(data, INA226_POWER_OVER_LIMIT_MASK, + return ina226_alert_limit_read(data, INA2XX_ALERT_POWER_HIGH, INA2XX_POWER, val); case hwmon_power_crit_alarm: - return ina226_alert_read(data, INA226_POWER_OVER_LIMIT_MASK, val); + return ina226_alert_read(data, INA2XX_ALERT_POWER_HIGH, val); default: return -EOPNOTSUPP; } @@ -633,15 +677,15 @@ static int ina2xx_curr_read(struct device *dev, u32 attr, long *val) *val = ina2xx_get_value(data, INA2XX_CURRENT, regval); return 0; case hwmon_curr_lcrit: - return ina226_alert_limit_read(data, INA226_SHUNT_UNDER_VOLTAGE_MASK, + return ina226_alert_limit_read(data, INA2XX_ALERT_CURRENT_LOW, INA2XX_CURRENT, val); case hwmon_curr_crit: - return ina226_alert_limit_read(data, INA226_SHUNT_OVER_VOLTAGE_MASK, + return ina226_alert_limit_read(data, INA2XX_ALERT_CURRENT_HIGH, INA2XX_CURRENT, val); case hwmon_curr_lcrit_alarm: - return ina226_alert_read(data, INA226_SHUNT_UNDER_VOLTAGE_MASK, val); + return ina226_alert_read(data, INA2XX_ALERT_CURRENT_LOW, val); case hwmon_curr_crit_alarm: - return ina226_alert_read(data, INA226_SHUNT_OVER_VOLTAGE_MASK, val); + return ina226_alert_read(data, INA2XX_ALERT_CURRENT_HIGH, val); default: return -EOPNOTSUPP; } @@ -685,12 +729,12 @@ static int ina2xx_in_write(struct device *dev, u32 attr, int channel, long val) switch (attr) { case hwmon_in_lcrit: return ina226_alert_limit_write(data, - channel ? INA226_BUS_UNDER_VOLTAGE_MASK : INA226_SHUNT_UNDER_VOLTAGE_MASK, + channel ? INA2XX_ALERT_BUS_VOLTAGE_LOW : INA2XX_ALERT_SHUNT_VOLTAGE_LOW, channel ? INA2XX_BUS_VOLTAGE : INA2XX_SHUNT_VOLTAGE, val); case hwmon_in_crit: return ina226_alert_limit_write(data, - channel ? INA226_BUS_OVER_VOLTAGE_MASK : INA226_SHUNT_OVER_VOLTAGE_MASK, + channel ? INA2XX_ALERT_BUS_VOLTAGE_HIGH : INA2XX_ALERT_SHUNT_VOLTAGE_HIGH, channel ? INA2XX_BUS_VOLTAGE : INA2XX_SHUNT_VOLTAGE, val); default: @@ -705,7 +749,7 @@ static int ina2xx_power_write(struct device *dev, u32 attr, long val) switch (attr) { case hwmon_power_crit: - return ina226_alert_limit_write(data, INA226_POWER_OVER_LIMIT_MASK, + return ina226_alert_limit_write(data, INA2XX_ALERT_POWER_HIGH, INA2XX_POWER, val); default: return -EOPNOTSUPP; @@ -719,10 +763,10 @@ static int ina2xx_curr_write(struct device *dev, u32 attr, long val) switch (attr) { case hwmon_curr_lcrit: - return ina226_alert_limit_write(data, INA226_SHUNT_UNDER_VOLTAGE_MASK, + return ina226_alert_limit_write(data, INA2XX_ALERT_CURRENT_LOW, INA2XX_CURRENT, val); case hwmon_curr_crit: - return ina226_alert_limit_write(data, INA226_SHUNT_OVER_VOLTAGE_MASK, + return ina226_alert_limit_write(data, INA2XX_ALERT_CURRENT_HIGH, INA2XX_CURRENT, val); default: return -EOPNOTSUPP; From 89dbb2405dcf6e99bc058902c2476b49b4982857 Mon Sep 17 00:00:00 2001 From: Jared Kangas Date: Thu, 20 Aug 2026 06:09:24 -0700 Subject: [PATCH 207/857] hwmon: (ina2xx) Decouple in0 and curr1 alarms INA2XX current limits are converted into shunt voltage limits internally using the shunt resistor value. Once a current limit's corresponding voltage limit is written to the hardware, shunt voltage and current alarms are indistinguishable from each other. This causes two issues: 1. in0/curr1 alarms may be unintentionally cleared by reading from the opposite input's alarm. 2. When a limit for either in0 (shunt voltage) or curr1 (current) is set, both of their alarms are triggered, and both of their limits read nonzero. An example of this behavior on an INA231: # cd /sys/class/hwmon/hwmon0 # head {curr1,in0}_input ==> curr1_input <== 1713 ==> in0_input <== 2 # echo 1800 >curr1_lcrit # head {curr1,in0}_lcrit_alarm ==> curr1_lcrit_alarm <== 1 ==> in0_lcrit_alarm <== 0 # head {in0,curr1}_lcrit_alarm ==> in0_lcrit_alarm <== 1 ==> curr1_lcrit_alarm <== 0 # head {in0,curr1}_lcrit_alarm ==> in0_lcrit_alarm <== 1 ==> curr1_lcrit_alarm <== 1 This is because curr1 uses the same underlying masks (INA226_SHUNT_*_VOLTAGE_MASK) as in0 on the hardware. As a result, ina2xx_{curr,in}_read() both read the shunt voltage alarms/limits without considering whether the voltage or current is currently set. To fix this, track the active alarm type in ina2xx_data and guard alarm/limit reads with a check that returns zero if the active alarm is for a different type. The new field is initialized based on the MASK_ENABLE register's set function, assuming voltage instead of current when the shunt voltage mask is set. After this fix, the alarms only read back 1 if their corresponding limit is set: # echo 0 >curr1_lcrit # head {curr1,in0}_lcrit_alarm ==> curr1_lcrit_alarm <== 0 ==> in0_lcrit_alarm <== 0 # echo 9999 >curr1_lcrit # head {curr1,in0}_lcrit_alarm ==> curr1_lcrit_alarm <== 1 ==> in0_lcrit_alarm <== 0 # echo 9999 >in0_lcrit # head {curr1,in0}_lcrit_alarm ==> curr1_lcrit_alarm <== 0 ==> in0_lcrit_alarm <== 1 Fixes: 4d5c2d986757 ("hwmon: (ina2xx) Add support for current limits") Signed-off-by: Jared Kangas Link: https://patch.msgid.link/20260820-upstream-ina2xx-in0-curr1-alarms-v2-4-fdce35abc41e@redhat.com Signed-off-by: Guenter Roeck --- drivers/hwmon/ina2xx.c | 65 ++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 63 insertions(+), 2 deletions(-) diff --git a/drivers/hwmon/ina2xx.c b/drivers/hwmon/ina2xx.c index 75e97e30bdcd4d..19b35f3bf3a3c1 100644 --- a/drivers/hwmon/ina2xx.c +++ b/drivers/hwmon/ina2xx.c @@ -8,6 +8,7 @@ */ #include +#include #include #include #include @@ -159,6 +160,7 @@ struct ina2xx_data { const struct ina2xx_config *config; enum ina2xx_ids chip; + enum ina2xx_alert_type active_alert; long rshunt; long current_lsb_uA; long power_lsb_uW; @@ -463,6 +465,35 @@ static u32 ina2xx_alert_type_to_mask(enum ina2xx_alert_type alert) } } +static enum ina2xx_alert_type ina2xx_mask_to_alert_type(u32 mask) +{ + int top_bit = fls(mask & INA226_ALERT_CONFIG_MASK); + + if (!top_bit) + return INA2XX_ALERT_NONE; + + /* + * Multiple bits may be set, with the highest-set function taking + * precedence according to the datasheet. Shunt voltage masks are + * assumed to map to voltage monitoring rather than current monitoring, + * since the latter isn't directly implemented in the hardware. + */ + switch (BIT(top_bit - 1)) { + case INA226_SHUNT_OVER_VOLTAGE_MASK: + return INA2XX_ALERT_SHUNT_VOLTAGE_HIGH; + case INA226_SHUNT_UNDER_VOLTAGE_MASK: + return INA2XX_ALERT_SHUNT_VOLTAGE_LOW; + case INA226_BUS_OVER_VOLTAGE_MASK: + return INA2XX_ALERT_BUS_VOLTAGE_HIGH; + case INA226_BUS_UNDER_VOLTAGE_MASK: + return INA2XX_ALERT_BUS_VOLTAGE_LOW; + case INA226_POWER_OVER_LIMIT_MASK: + return INA2XX_ALERT_POWER_HIGH; + default: + return INA2XX_ALERT_NONE; + } +} + static int ina226_alert_limit_read(struct ina2xx_data *data, enum ina2xx_alert_type alert, int reg, long *val) { @@ -471,6 +502,12 @@ static int ina226_alert_limit_read(struct ina2xx_data *data, enum ina2xx_alert_t u32 mask; int ret; + /* Avoid nonzero reads from inactive alerts caused by shared limit register */ + if (data->active_alert != alert) { + *val = 0; + return 0; + } + ret = regmap_read(regmap, INA226_MASK_ENABLE, ®val); if (ret) return ret; @@ -506,6 +543,7 @@ static int ina226_alert_limit_write(struct ina2xx_data *data, enum ina2xx_alert_ INA226_ALERT_CONFIG_MASK, 0); if (ret < 0) return ret; + data->active_alert = INA2XX_ALERT_NONE; ret = regmap_write(regmap, INA226_ALERT_LIMIT, ina226_alert_to_reg(data, reg, val)); @@ -514,9 +552,13 @@ static int ina226_alert_limit_write(struct ina2xx_data *data, enum ina2xx_alert_ if (val) { mask = ina2xx_alert_type_to_mask(alert); - return regmap_update_bits(regmap, INA226_MASK_ENABLE, - INA226_ALERT_CONFIG_MASK, mask); + ret = regmap_update_bits(regmap, INA226_MASK_ENABLE, + INA226_ALERT_CONFIG_MASK, mask); + if (ret < 0) + return ret; + data->active_alert = alert; } + return 0; } @@ -546,6 +588,15 @@ static int ina226_alert_read(struct ina2xx_data *data, enum ina2xx_alert_type al u32 mask; int ret; + /* + * With alert latching, reading alerts from hardware also clears the + * alert, so return early if the alert is inactive. + */ + if (data->active_alert != alert) { + *val = 0; + return 0; + } + ret = regmap_read_bypassed(data->regmap, INA226_MASK_ENABLE, ®val); if (ret) return ret; @@ -988,6 +1039,16 @@ static int ina2xx_init(struct device *dev, struct ina2xx_data *data) if (data->config->has_alerts) { bool active_high = device_property_read_bool(dev, "ti,alert-polarity-active-high"); + unsigned int mask_enable; + + /* + * Infer active alert from MASK_ENABLE in case it's already + * configured (e.g., by a past probe or firmware) + */ + ret = regmap_read(regmap, INA226_MASK_ENABLE, &mask_enable); + if (ret < 0) + return ret; + data->active_alert = ina2xx_mask_to_alert_type(mask_enable); regmap_update_bits(regmap, INA226_MASK_ENABLE, INA226_ALERT_LATCH_ENABLE | INA226_ALERT_POLARITY, From ba21d4f1384d8216534fe744f58c2c12d85cbd3d Mon Sep 17 00:00:00 2001 From: Antonin Godard Date: Tue, 18 Aug 2026 10:08:40 +0200 Subject: [PATCH 208/857] Documentation: hwmon: replace full-width colon by a standard ASCII colon MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit It prevented the pdfdocs target to complete, prompting the following error: Latexmk: ====Problematic refs and citations with line #s in .tex file: Missing character: There is no : (U+FF1A) in font DejaVu Serif/OT:script=latn;l Fixes: 69001f21ded78 ("hwmon: document: add gpd-fan") Signed-off-by: Antonin Godard Link: https://patch.msgid.link/20260818-doc-hwmon-remove-confusable-v2-1-c1dff1ec01cd@bootlin.com Signed-off-by: Guenter Roeck --- Documentation/hwmon/gpd-fan.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/hwmon/gpd-fan.rst b/Documentation/hwmon/gpd-fan.rst index 29527a77fe882f..b27657d3305633 100644 --- a/Documentation/hwmon/gpd-fan.rst +++ b/Documentation/hwmon/gpd-fan.rst @@ -67,7 +67,7 @@ pwm1_enable at full speed. Write "1" to set to manual, write "2" to let the EC control decide fan speed. Read this attribute to see current status. - NB:In consideration of the safety of the device, when setting to manual mode, + NB: In consideration of the safety of the device, when setting to manual mode, the pwm speed will be set to the maximum value (255) by default. You can set a different value by writing pwm1 later. From 59c212ba59b2a3e85bd3327e1db6b9fbf114353d Mon Sep 17 00:00:00 2001 From: hanzhijian Date: Fri, 21 Aug 2026 19:57:20 +0800 Subject: [PATCH 209/857] hwmon: (yogafan) fix non-kernel-doc comment The file description comment starts with "/**" which is reserved for kernel-doc comments, triggering a kernel-doc checker warning. Change it to a plain "/*" comment since it does not document any function or struct. Fixes: c67c248ca406a ("hwmon: (yogafan) Add support for Lenovo Yoga/Legion fan monitoring") Signed-off-by: hanzhijian Link: https://patch.msgid.link/20260821115720.2017516-1-hanzhijian1991@gmail.com Signed-off-by: Guenter Roeck --- drivers/hwmon/yogafan.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/hwmon/yogafan.c b/drivers/hwmon/yogafan.c index 48fa5148d9e2c0..278cb089b0fd17 100644 --- a/drivers/hwmon/yogafan.c +++ b/drivers/hwmon/yogafan.c @@ -1,5 +1,5 @@ // SPDX-License-Identifier: GPL-2.0-only -/** +/* * yoga_fan.c - Lenovo Yoga/Legion Fan Hardware Monitoring Driver * * Provides fan speed monitoring for Lenovo Yoga, Legion, and IdeaPad From cf7865ca31bf75c35f53f3a03b2fb4e89c3d78a2 Mon Sep 17 00:00:00 2001 From: Guenter Roeck Date: Fri, 21 Aug 2026 07:49:15 -0700 Subject: [PATCH 210/857] hwmon: (sht4x) Add missing locks Sashiko reports: Heater sysfs callbacks (heater_enable_store, heater_power_store, and heater_time_store) are exposed to data races without the hwmon lock. If a user-space process reads hwmon data while another process enables the heater, heater_enable_store() executes without holding hwmon_lock(dev). This can interleave I2C commands and mutate shared state (data->heating_complete and data->data_pending) concurrently with sht4x_read_values(), leading to corrupted I2C sequences. Fixes: 53dfa12299c1 ("hwmon: (sht4x) Rely on subsystem locking") Cc: Alessandro Zini Signed-off-by: Guenter Roeck Link: https://patch.msgid.link/20260821144916.2889031-1-linux@roeck-us.net --- drivers/hwmon/sht4x.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/drivers/hwmon/sht4x.c b/drivers/hwmon/sht4x.c index 9cace0e8acdabc..7a0dc2ed723d8d 100644 --- a/drivers/hwmon/sht4x.c +++ b/drivers/hwmon/sht4x.c @@ -277,6 +277,8 @@ static ssize_t heater_enable_store(struct device *dev, heating_time_bound = 1100; } + guard(hwmon_lock)(dev); + if (time_before(jiffies, data->heating_complete)) return -EBUSY; @@ -314,6 +316,8 @@ static ssize_t heater_power_store(struct device *dev, if (power != 20 && power != 110 && power != 200) return -EINVAL; + guard(hwmon_lock)(dev); + data->heater_power = power; return count; @@ -344,6 +348,8 @@ static ssize_t heater_time_store(struct device *dev, if (time != 100 && time != 1000) return -EINVAL; + guard(hwmon_lock)(dev); + data->heater_time = time; return count; From c38894b51167524b8f2bc7a08b69d2d8a4dd3b23 Mon Sep 17 00:00:00 2001 From: Guenter Roeck Date: Fri, 21 Aug 2026 07:49:16 -0700 Subject: [PATCH 211/857] hwmon: (sht4x) Fix return value from heater_enable_store() Sashiko reports: The return value in heater_enable_store() causes an unexpected write failure in user-space. When the heater is successfully enabled, the function returns 0 instead of count: drivers/hwmon/sht4x.c:heater_enable_store() { ... data->heating_complete = jiffies + msecs_to_jiffies(heating_time_bound); data->data_pending = true; return 0; } Returning 0 signals to VFS that no bytes were processed. Standard user-space tools will retry the write with the remaining bytes. On the retry, time_before(jiffies, data->heating_complete) evaluates to true, and the function immediately fails with -EBUSY. Return count as expected to fix the problem. Fixes: 0eed6fc3d2b9e ("hwmon: (sht4x): add heater support") Cc: Antoni Pokusinski Cc: Alessandro Zini Signed-off-by: Guenter Roeck Link: https://patch.msgid.link/20260821144916.2889031-2-linux@roeck-us.net --- drivers/hwmon/sht4x.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/hwmon/sht4x.c b/drivers/hwmon/sht4x.c index 7a0dc2ed723d8d..a97dda9e92dc5d 100644 --- a/drivers/hwmon/sht4x.c +++ b/drivers/hwmon/sht4x.c @@ -288,7 +288,7 @@ static ssize_t heater_enable_store(struct device *dev, data->heating_complete = jiffies + msecs_to_jiffies(heating_time_bound); data->data_pending = true; - return 0; + return count; } static ssize_t heater_power_show(struct device *dev, From 7c1f21a1d7ec7b74d57a424d77d0606a5cba466e Mon Sep 17 00:00:00 2001 From: Cong Nguyen Date: Fri, 28 Aug 2026 17:54:13 +0700 Subject: [PATCH 212/857] hwmon: (applesmc) fix key backlight workqueue leak on register failure applesmc_create_key_backlight() allocates applesmc_led_wq before calling led_classdev_register(). When register fails, the error is returned to applesmc_init(), which jumps to out_light_sysfs and skips applesmc_release_key_backlight(), leaking the workqueue. Destroy the workqueue on the register failure path. The bug was introduced when the inline init block was refactored into a helper that returns errors directly, dropping the old out_light_wq unwind label. Fixes: 0b0b5dff8967 ("hwmon: (applesmc) Simplify feature sysfs handling") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-4 Signed-off-by: Cong Nguyen Link: https://patch.msgid.link/20260828105413.2401385-1-congnt264@gmail.com Signed-off-by: Guenter Roeck --- drivers/hwmon/applesmc.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/drivers/hwmon/applesmc.c b/drivers/hwmon/applesmc.c index ca56bd8b170e6e..ec392886c21bc0 100644 --- a/drivers/hwmon/applesmc.c +++ b/drivers/hwmon/applesmc.c @@ -1128,12 +1128,17 @@ static void applesmc_release_light_sensor(void) static int applesmc_create_key_backlight(void) { + int ret; + if (!smcreg.has_key_backlight) return 0; applesmc_led_wq = create_singlethread_workqueue("applesmc-led"); if (!applesmc_led_wq) return -ENOMEM; - return led_classdev_register(&pdev->dev, &applesmc_backlight); + ret = led_classdev_register(&pdev->dev, &applesmc_backlight); + if (ret) + destroy_workqueue(applesmc_led_wq); + return ret; } static void applesmc_release_key_backlight(void) From 2b78e2651f99a5cae7af2aa0a9ff4049b8c5de6f Mon Sep 17 00:00:00 2001 From: Javier Carrasco Date: Sun, 23 Aug 2026 19:59:01 +0200 Subject: [PATCH 213/857] hwmon: chipcap2: fix channels in humidity alarm notifications hwmon_notify_event() expects the channel number as its last argument, taken into account with the type parameter that it is a humidity sensor type. Given that this device only provides one humidity channel, 0 must be passed. The custom construct to enumerate the channels makes wrong assumptions by listing all types together (temperature and humidity). Remove the custom channel enumeration and pass the right channel to hwmon_notify_event() for hwmon_humidity_min_alarm and hwmon_humidity_max_alarm. Fixes: 3af350929e75 ("hwmon: Add support for Amphenol ChipCap 2") Cc: stable@vger.kernel.org Signed-off-by: Javier Carrasco Link: https://patch.msgid.link/20260823-chipcap2_locks-v2-1-6a26c8e9e2fc@gmail.com Signed-off-by: Guenter Roeck --- drivers/hwmon/chipcap2.c | 9 ++------- 1 file changed, 2 insertions(+), 7 deletions(-) diff --git a/drivers/hwmon/chipcap2.c b/drivers/hwmon/chipcap2.c index 086571d556b7eb..9bef767b589ee2 100644 --- a/drivers/hwmon/chipcap2.c +++ b/drivers/hwmon/chipcap2.c @@ -92,11 +92,6 @@ struct cc2_data { bool process_irqs; }; -enum cc2_chan_addr { - CC2_CHAN_TEMP = 0, - CC2_CHAN_HUMIDITY, -}; - /* %RH as a per cent mille from a register value */ static long cc2_rh_convert(u16 data) { @@ -499,7 +494,7 @@ static irqreturn_t cc2_low_interrupt(int irq, void *data) if (cc2->process_irqs) { hwmon_notify_event(cc2->hwmon, hwmon_humidity, - hwmon_humidity_min_alarm, CC2_CHAN_HUMIDITY); + hwmon_humidity_min_alarm, 0); cc2->rh_alarm.low_alarm = true; } @@ -512,7 +507,7 @@ static irqreturn_t cc2_high_interrupt(int irq, void *data) if (cc2->process_irqs) { hwmon_notify_event(cc2->hwmon, hwmon_humidity, - hwmon_humidity_max_alarm, CC2_CHAN_HUMIDITY); + hwmon_humidity_max_alarm, 0); cc2->rh_alarm.high_alarm = true; } From 11140fd6873d382c01d8bb93c237d495d15cf6db Mon Sep 17 00:00:00 2001 From: Vishnu Razdan Date: Mon, 24 Aug 2026 23:58:00 -0700 Subject: [PATCH 214/857] hwmon: (pmbus) Clear generic status alarms with CLEAR_FAULTS Some hwmon alarms fall back to STATUS_WORD summary bits when no individual limit alarm is available. On PMBus 1.2 and newer devices, pmbus_get_boolean() acknowledges these alarms with the same byte-data write used for detailed status registers. For example, PB_STATUS_INPUT is 0x2000, so it is truncated to zero when passed to _pmbus_write_byte_data(). The resulting write cannot acknowledge the input alarm. PMBus 1.3 Part II, sections 10.2.4 and 10.2.5, excludes ordinary STATUS_BYTE and STATUS_WORD summary bits from individual clearing. Their summary bits clear when the underlying status bits clear, so changing this to a word-data write would not fix the generic input alarm either. Use the existing page CLEAR_FAULTS path for generic STATUS_WORD alarms, including devices whose status accessor uses STATUS_BYTE. Keep individual byte writes for detailed status registers on PMBus 1.2 and newer devices. As with the existing older-device fallback, CLEAR_FAULTS can clear other latched status; an active condition can reassert its status. Fixes: 35f165f08950 ("hwmon: (pmbus) Clear pmbus fault/warning bits after read") Cc: stable@vger.kernel.org Assisted-by: LLM Signed-off-by: Vishnu Razdan Link: https://patch.msgid.link/20260824-vrazdan-pmbus-status-word-b4-v1-1-2606ecd0c029@openai.com Signed-off-by: Guenter Roeck --- drivers/hwmon/pmbus/pmbus_core.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/drivers/hwmon/pmbus/pmbus_core.c b/drivers/hwmon/pmbus/pmbus_core.c index 806c9a4913bb08..5f69c1420b4e15 100644 --- a/drivers/hwmon/pmbus/pmbus_core.c +++ b/drivers/hwmon/pmbus/pmbus_core.c @@ -1275,7 +1275,9 @@ static int pmbus_get_boolean(struct i2c_client *client, struct pmbus_boolean *b, regval = status & mask; if (regval) { - if (data->revision >= PMBUS_REV_12) { + /* Generic STATUS_WORD alarms are not individually clearable. */ + if (data->revision >= PMBUS_REV_12 && + reg != PMBUS_STATUS_WORD) { ret = _pmbus_write_byte_data(client, page, reg, regval); if (ret) return ret; From f6a57feaa747dd0dd32950b1f0f117bdd0e43c97 Mon Sep 17 00:00:00 2001 From: Christoph Berliner Date: Fri, 21 Aug 2026 16:25:11 +0200 Subject: [PATCH 215/857] watchdog: sp5100_tco: allow unreserved MMIO on GA-78LMT-USB3 The Gigabyte GA-78LMT-USB3 firmware programs the legacy SP5100 watchdog MMIO window at 0xfec000f0. This address falls inside the IOAPIC resource, so sp5100_tco fails to reserve it and aborts probing. Do not relocate or reprogram the watchdog. Instead, add a narrowly scoped DMI quirk for this board which permits use of the firmware-provided MMIO address without reserving it. The exception is limited to the legacy SP5100 register layout, the GA-78LMT-USB3 DMI identity, and the firmware address 0xfec000f0. All other systems retain the existing resource reservation behavior. On the affected system the watchdog initializes successfully and /dev/watchdog0 is registered while the firmware-programmed watchdog base remains unchanged at 0xfec000f0 during load and unload. Tested on a Gigabyte GA-78LMT-USB3 with AMD SBx00 SMBus controller (PCI 1002:4385, revision 0x3c). Signed-off-by: Christoph Berliner Link: https://patch.msgid.link/20260821142511.49934-1-caberliner@gmail.com Signed-off-by: Guenter Roeck --- drivers/watchdog/sp5100_tco.c | 61 ++++++++++++++++++++++++++++++++--- 1 file changed, 56 insertions(+), 5 deletions(-) diff --git a/drivers/watchdog/sp5100_tco.c b/drivers/watchdog/sp5100_tco.c index 7e99c3b1f3676b..6c85f11cff2416 100644 --- a/drivers/watchdog/sp5100_tco.c +++ b/drivers/watchdog/sp5100_tco.c @@ -33,6 +33,7 @@ #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt #include +#include #include #include #include @@ -240,6 +241,35 @@ static u32 sp5100_tco_read_pm_reg32(u8 index) return val; } +/* + * The Gigabyte GA-78LMT-USB3 firmware programs the legacy SP5100 watchdog + * MMIO window at 0xfec000f0. This address lies inside the IOAPIC resource, + * so the generic resource reservation fails even though firmware explicitly + * assigns the watchdog to this address. + * + * Keep this exception narrowly scoped to the affected system and firmware + * address. Do not relocate or otherwise reprogram the watchdog. + */ +#define SP5100_WDT_GA78LMT_MMIO 0xfec000f0 + +static const struct dmi_system_id sp5100_tco_unreserved_mmio_dmi[] = { + { + .matches = { + DMI_MATCH(DMI_SYS_VENDOR, "Gigabyte Technology Co., Ltd."), + DMI_MATCH(DMI_PRODUCT_NAME, "GA-78LMT-USB3"), + }, + }, + {} +}; + +static bool sp5100_tco_allow_unreserved_mmio(struct sp5100_tco *tco, + u32 mmio_addr) +{ + return tco->tco_reg_layout == sp5100 && + mmio_addr == SP5100_WDT_GA78LMT_MMIO && + dmi_check_system(sp5100_tco_unreserved_mmio_dmi); +} + static u32 sp5100_tco_request_region(struct device *dev, u32 mmio_addr, const char *dev_name) @@ -259,6 +289,7 @@ static u32 sp5100_tco_prepare_base(struct sp5100_tco *tco, const char *dev_name) { struct device *dev = tco->wdd.parent; + bool reserved = false; dev_dbg(dev, "Got 0x%08x from SBResource_MMIO register\n", mmio_addr); @@ -266,11 +297,29 @@ static u32 sp5100_tco_prepare_base(struct sp5100_tco *tco, return -ENODEV; /* Check for MMIO address and alternate MMIO address conflicts */ - if (mmio_addr) - mmio_addr = sp5100_tco_request_region(dev, mmio_addr, dev_name); + if (mmio_addr) { + u32 requested_addr; + + requested_addr = sp5100_tco_request_region(dev, mmio_addr, + dev_name); + if (requested_addr) { + mmio_addr = requested_addr; + reserved = true; + } else if (sp5100_tco_allow_unreserved_mmio(tco, mmio_addr)) { + dev_info(dev, + "Using firmware watchdog MMIO 0x%08x without reserving it\n", + mmio_addr); + } else { + mmio_addr = 0; + } + } - if (!mmio_addr && alt_mmio_addr) - mmio_addr = sp5100_tco_request_region(dev, alt_mmio_addr, dev_name); + if (!mmio_addr && alt_mmio_addr) { + mmio_addr = sp5100_tco_request_region(dev, alt_mmio_addr, + dev_name); + if (mmio_addr) + reserved = true; + } if (!mmio_addr) { dev_err(dev, "Failed to reserve MMIO or alternate MMIO region\n"); @@ -280,7 +329,9 @@ static u32 sp5100_tco_prepare_base(struct sp5100_tco *tco, tco->tcobase = devm_ioremap(dev, mmio_addr, SP5100_WDT_MEM_MAP_SIZE); if (!tco->tcobase) { dev_err(dev, "MMIO address 0x%08x failed mapping\n", mmio_addr); - devm_release_mem_region(dev, mmio_addr, SP5100_WDT_MEM_MAP_SIZE); + if (reserved) + devm_release_mem_region(dev, mmio_addr, + SP5100_WDT_MEM_MAP_SIZE); return -ENOMEM; } From 11f93e639d513cbfaa78237cd163039d27fea33c Mon Sep 17 00:00:00 2001 From: Zexin Wang Date: Mon, 17 Aug 2026 10:37:49 +0800 Subject: [PATCH 216/857] watchdog: sbsa_gwdt: add early_enable module parameter On SBSA platforms using standard UEFI firmware (such as EDK II), the watchdog timer is often enabled during early boot stages but explicitly disabled by the firmware before handing over control to the OS (e.g., during ExitBootServices). This is done to prevent unintended resets while the OS is loading, assuming the OS watchdog driver will take over. However, this leaves a protection gap. If the system hangs between the firmware handover and the userspace watchdog daemon startup, the hardware watchdog will not fire to recover the system. For safety-critical systems that require continuous hardware watchdog protection from the earliest possible moment, this gap is problematic. Add an 'early_enable' module parameter to allow the kernel driver to re-enable the watchdog immediately during probe if it was left disabled by the firmware. By setting the WDOG_HW_RUNNING status bit, the watchdog core is instructed that the hardware is active. As a result, the core's pre-userspace handler (controlled by 'handle_boot_enabled') will automatically issue periodic keepalives until userspace opens the device. This bridges the protection gap seamlessly without requiring firmware modifications and without risking unintended resets during kernel boot. The parameter defaults to false to preserve the traditional behavior. Signed-off-by: Zexin Wang Link: https://patch.msgid.link/20260817023838.6459-1-ot_zexin.wang@mediatek.com Signed-off-by: Guenter Roeck --- .../watchdog/watchdog-parameters.rst | 2 ++ drivers/watchdog/sbsa_gwdt.c | 19 +++++++++++++++++-- 2 files changed, 19 insertions(+), 2 deletions(-) diff --git a/Documentation/watchdog/watchdog-parameters.rst b/Documentation/watchdog/watchdog-parameters.rst index 502cb6adbeda74..348229ffcb43f0 100644 --- a/Documentation/watchdog/watchdog-parameters.rst +++ b/Documentation/watchdog/watchdog-parameters.rst @@ -515,6 +515,8 @@ sbsa_gwdt: nowayout: Watchdog cannot be stopped once started (default=kernel config parameter) + early_enable: + Watchdog is started on module insertion (default=0) ------------------------------------------------- diff --git a/drivers/watchdog/sbsa_gwdt.c b/drivers/watchdog/sbsa_gwdt.c index e04d42cc7774da..3c5bfd8641c62a 100644 --- a/drivers/watchdog/sbsa_gwdt.c +++ b/drivers/watchdog/sbsa_gwdt.c @@ -122,6 +122,11 @@ MODULE_PARM_DESC(nowayout, "Watchdog cannot be stopped once started (default=" __MODULE_STRING(WATCHDOG_NOWAYOUT) ")"); +static bool early_enable; +module_param(early_enable, bool, 0); +MODULE_PARM_DESC(early_enable, + "Watchdog is started on module insertion (default=0)"); + /* * Arm Base System Architecture 1.0 introduces watchdog v1 which * increases the length watchdog offset register to 48 bits. @@ -296,6 +301,7 @@ static int sbsa_gwdt_probe(struct platform_device *pdev) struct sbsa_gwdt *gwdt; int ret, irq; u32 status; + bool early_action; gwdt = devm_kzalloc(dev, sizeof(*gwdt), GFP_KERNEL); if (!gwdt) @@ -386,14 +392,23 @@ static int sbsa_gwdt_probe(struct platform_device *pdev) */ sbsa_gwdt_set_timeout(wdd, wdd->timeout); + early_action = early_enable && !(status & SBSA_GWDT_WCS_EN); + if (early_action) { + sbsa_gwdt_start(wdd); + set_bit(WDOG_HW_RUNNING, &wdd->status); + } + watchdog_stop_on_reboot(wdd); ret = devm_watchdog_register_device(dev, wdd); - if (ret) + if (ret) { + if (early_action) + sbsa_gwdt_stop(wdd); return ret; + } dev_info(dev, "Initialized with %ds timeout @ %u Hz, action=%d.%s\n", wdd->timeout, gwdt->clk, action, - status & SBSA_GWDT_WCS_EN ? " [enabled]" : ""); + watchdog_hw_running(wdd) ? " [enabled]" : ""); return 0; } From a2fa8442ed8cfa921cd64d223bdea7284b38b266 Mon Sep 17 00:00:00 2001 From: Aiden Isik Date: Fri, 21 Aug 2026 16:14:49 +0100 Subject: [PATCH 217/857] dt-bindings: watchdog: samsung-wdt: Add exynos5515-wdt compatible Add a dt-binding compatible for the Exynos5515 watchdog timer. This watchdog requires a syscon phandle, and the cluster index should *not* be specified, as that does not make sense on the Exynos5515 SoC (due to it only having a single core cluster). Signed-off-by: Aiden Isik Reviewed-by: Krzysztof Kozlowski Link: https://patch.msgid.link/20260821-for-next-lucky7-watchdog-v4-1-d070cee5009f@member.fsf.org Signed-off-by: Guenter Roeck --- .../bindings/watchdog/samsung-wdt.yaml | 23 ++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/Documentation/devicetree/bindings/watchdog/samsung-wdt.yaml b/Documentation/devicetree/bindings/watchdog/samsung-wdt.yaml index 41aee1655b0c22..a32c478315779b 100644 --- a/Documentation/devicetree/bindings/watchdog/samsung-wdt.yaml +++ b/Documentation/devicetree/bindings/watchdog/samsung-wdt.yaml @@ -22,6 +22,7 @@ properties: - samsung,s3c6410-wdt # for S3C6410, S5PV210 and Exynos4 - samsung,exynos5250-wdt # for Exynos5250 - samsung,exynos5420-wdt # for Exynos5420 + - samsung,exynos5515-wdt - samsung,exynos7-wdt # for Exynos7 - samsung,exynos850-wdt # for Exynos850 - samsung,exynos990-wdt # for Exynos990 @@ -57,7 +58,7 @@ properties: $ref: /schemas/types.yaml#/definitions/phandle description: Phandle to the PMU system controller node (in case of Exynos5250, - Exynos5420, Exynos7, Exynos850, Exynos990 and gs101). + Exynos5420, Exynos5515, Exynos7, Exynos850, Exynos990 and gs101). required: - compatible @@ -93,6 +94,26 @@ allOf: - samsung,cluster-index - samsung,syscon-phandle + - if: + properties: + compatible: + contains: + enum: + - samsung,exynos5515-wdt + then: + properties: + clocks: + items: + - description: Bus clock, used for register interface + - description: Source clock (driving watchdog counter) + clock-names: + items: + - const: watchdog + - const: watchdog_src + samsung,cluster-index: false + required: + - samsung,syscon-phandle + - if: properties: compatible: From 4e2583a67c0d3978a160e830923938527caf6600 Mon Sep 17 00:00:00 2001 From: Aiden Isik Date: Fri, 21 Aug 2026 16:14:50 +0100 Subject: [PATCH 218/857] watchdog: s3c2410_wdt: Add exynos5515-wdt compatible data Add driver data for the Exynos5515 SoC's watchdog timer. Unlike similar SoCs such as GS101 or Exynos990, Exynos5515's PMU does not require explicit counter enablement for the watchdog to tick. Signed-off-by: Aiden Isik Reviewed-by: Krzysztof Kozlowski Link: https://patch.msgid.link/20260821-for-next-lucky7-watchdog-v4-2-d070cee5009f@member.fsf.org Signed-off-by: Guenter Roeck --- drivers/watchdog/s3c2410_wdt.c | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/drivers/watchdog/s3c2410_wdt.c b/drivers/watchdog/s3c2410_wdt.c index e31f93db050966..e61439a3f4b283 100644 --- a/drivers/watchdog/s3c2410_wdt.c +++ b/drivers/watchdog/s3c2410_wdt.c @@ -232,6 +232,20 @@ static const struct s3c2410_wdt_variant drv_data_exynos5420 = { QUIRK_HAS_PMU_RST_STAT | QUIRK_HAS_PMU_AUTO_DISABLE, }; +/* + * Unlike similar SoCs like GS101, Exynos5515's PMU does not require + * explicit counter enablement. Hence, QUIRK_HAS_PMU_CNT_EN is not set. + */ +static const struct s3c2410_wdt_variant drv_data_exynos5515 = { + .mask_reset_reg = EXYNOSAUTOV920_CLUSTER0_NONCPU_INT_EN, + .mask_bit = 2, + .mask_reset_inv = true, + .rst_stat_reg = EXYNOS5_RST_STAT_REG_OFFSET, + .rst_stat_bit = 24, + .quirks = QUIRK_HAS_WTCLRINT_REG | QUIRK_HAS_PMU_MASK_RESET | \ + QUIRK_HAS_PMU_RST_STAT | QUIRK_HAS_DBGACK_BIT, +}; + static const struct s3c2410_wdt_variant drv_data_exynos7 = { .disable_reg = EXYNOS5_WDT_DISABLE_REG_OFFSET, .mask_reset_reg = EXYNOS5_WDT_MASK_RESET_REG_OFFSET, @@ -379,6 +393,8 @@ static const struct of_device_id s3c2410_wdt_match[] = { .data = &drv_data_exynos5250 }, { .compatible = "samsung,exynos5420-wdt", .data = &drv_data_exynos5420 }, + { .compatible = "samsung,exynos5515-wdt", + .data = &drv_data_exynos5515 }, { .compatible = "samsung,exynos7-wdt", .data = &drv_data_exynos7 }, { .compatible = "samsung,exynos850-wdt", From 42b33d917a759bfc8c1bd73e4f4939eca7f1fe12 Mon Sep 17 00:00:00 2001 From: Wanming Gao Date: Thu, 27 Aug 2026 17:26:12 +0800 Subject: [PATCH 219/857] watchdog: mediatek: acknowledge pretimeout interrupt The MediaTek watchdog pretimeout interrupt is level-triggered and does not have a separate acknowledge register. The interrupt is cleared by changing WDT_MODE_IRQ_LEVEL_EN and then restoring it to its original state. Without this transition, the interrupt may remain asserted and cause an interrupt storm. After changing WDT_MODE_IRQ_LEVEL_EN, wait 70 us before restoring it. This is longer than two 32 kHz watchdog clock cycles, allowing the level change to propagate across the clock domain. WDT_MODE is also updated by the watchdog start, stop, and pretimeout operations. Protect its read-modify-write sequences and the complete IRQ acknowledge sequence with the watchdog spinlock so that concurrent updates cannot overwrite the temporary IRQ level state. Signed-off-by: Wanming Gao Tested-by: Tzung-Bi Shih Reviewed-by: Tzung-Bi Shih Link: https://patch.msgid.link/20260827092616.2724197-1-wanming.gao@mediatek.com Signed-off-by: Guenter Roeck --- drivers/watchdog/mtk_wdt.c | 62 ++++++++++++++++++++++++++++++++++---- 1 file changed, 56 insertions(+), 6 deletions(-) diff --git a/drivers/watchdog/mtk_wdt.c b/drivers/watchdog/mtk_wdt.c index d9c30e4c80e3ff..ddc4d2ff9d9f96 100644 --- a/drivers/watchdog/mtk_wdt.c +++ b/drivers/watchdog/mtk_wdt.c @@ -35,6 +35,7 @@ #define WDT_MAX_TIMEOUT 31 #define WDT_MIN_TIMEOUT 2 #define WDT_LENGTH_TIMEOUT(n) ((n) << 5) +#define WDT_IRQ_LEVEL_SYNC_US 70 #define WDT_LENGTH 0x04 #define WDT_LENGTH_KEY 0x8 @@ -49,6 +50,7 @@ #define WDT_MODE_EXRST_EN (1 << 2) #define WDT_MODE_IRQ_EN (1 << 3) #define WDT_MODE_AUTO_START (1 << 4) +#define WDT_MODE_IRQ_LEVEL_EN (1 << 5) #define WDT_MODE_DUAL_EN (1 << 6) #define WDT_MODE_CNT_SEL (1 << 8) #define WDT_MODE_KEY 0x22000000 @@ -72,7 +74,7 @@ static unsigned int timeout; struct mtk_wdt_dev { struct watchdog_device wdt_dev; void __iomem *wdt_base; - spinlock_t lock; /* protects WDT_SWSYSRST reg */ + spinlock_t lock; /* protects WDT_MODE and WDT_SWSYSRST reg */ struct reset_controller_dev rcdev; bool disable_wdt_extrst; bool reset_by_toprgu; @@ -212,8 +214,6 @@ static int toprgu_register_reset_controller(struct platform_device *pdev, int ret; struct mtk_wdt_dev *mtk_wdt = platform_get_drvdata(pdev); - spin_lock_init(&mtk_wdt->lock); - mtk_wdt->rcdev.owner = THIS_MODULE; mtk_wdt->rcdev.nr_resets = rst_num; mtk_wdt->rcdev.ops = &toprgu_reset_ops; @@ -302,12 +302,15 @@ static int mtk_wdt_stop(struct watchdog_device *wdt_dev) { struct mtk_wdt_dev *mtk_wdt = watchdog_get_drvdata(wdt_dev); void __iomem *wdt_base = mtk_wdt->wdt_base; + unsigned long flags; u32 reg; + spin_lock_irqsave(&mtk_wdt->lock, flags); reg = readl(wdt_base + WDT_MODE); reg &= ~WDT_MODE_EN; reg |= WDT_MODE_KEY; iowrite32(reg, wdt_base + WDT_MODE); + spin_unlock_irqrestore(&mtk_wdt->lock, flags); return 0; } @@ -317,12 +320,14 @@ static int mtk_wdt_start(struct watchdog_device *wdt_dev) u32 reg; struct mtk_wdt_dev *mtk_wdt = watchdog_get_drvdata(wdt_dev); void __iomem *wdt_base = mtk_wdt->wdt_base; + unsigned long flags; int ret; ret = mtk_wdt_set_timeout(wdt_dev, wdt_dev->timeout); if (ret < 0) return ret; + spin_lock_irqsave(&mtk_wdt->lock, flags); reg = ioread32(wdt_base + WDT_MODE); if (wdt_dev->pretimeout) reg |= (WDT_MODE_IRQ_EN | WDT_MODE_DUAL_EN); @@ -334,6 +339,7 @@ static int mtk_wdt_start(struct watchdog_device *wdt_dev) reg |= WDT_MODE_CNT_SEL; reg |= (WDT_MODE_EN | WDT_MODE_KEY); iowrite32(reg, wdt_base + WDT_MODE); + spin_unlock_irqrestore(&mtk_wdt->lock, flags); return 0; } @@ -343,7 +349,11 @@ static int mtk_wdt_set_pretimeout(struct watchdog_device *wdd, { struct mtk_wdt_dev *mtk_wdt = watchdog_get_drvdata(wdd); void __iomem *wdt_base = mtk_wdt->wdt_base; - u32 reg = ioread32(wdt_base + WDT_MODE); + unsigned long flags; + u32 reg; + + spin_lock_irqsave(&mtk_wdt->lock, flags); + reg = ioread32(wdt_base + WDT_MODE); if (timeout && !wdd->pretimeout) { wdd->pretimeout = wdd->timeout / 2; @@ -352,19 +362,54 @@ static int mtk_wdt_set_pretimeout(struct watchdog_device *wdd, wdd->pretimeout = 0; reg &= ~(WDT_MODE_IRQ_EN | WDT_MODE_DUAL_EN); } else { + spin_unlock_irqrestore(&mtk_wdt->lock, flags); return 0; } reg |= WDT_MODE_KEY; iowrite32(reg, wdt_base + WDT_MODE); + spin_unlock_irqrestore(&mtk_wdt->lock, flags); return mtk_wdt_set_timeout(wdd, wdd->timeout); } +static void mtk_wdt_deassert_irq(struct watchdog_device *wdd) +{ + struct mtk_wdt_dev *mtk_wdt = watchdog_get_drvdata(wdd); + void __iomem *wdt_base = mtk_wdt->wdt_base; + unsigned long flags; + u32 reg; + + spin_lock_irqsave(&mtk_wdt->lock, flags); + + reg = ioread32(wdt_base + WDT_MODE); + reg ^= WDT_MODE_IRQ_LEVEL_EN; + iowrite32(reg | WDT_MODE_KEY, wdt_base + WDT_MODE); + + /* + * Wait for two 32 kHz watchdog clock cycles so the IRQ level + * change can propagate across the clock domain. + */ + udelay(WDT_IRQ_LEVEL_SYNC_US); + + reg = ioread32(wdt_base + WDT_MODE); + reg ^= WDT_MODE_IRQ_LEVEL_EN; + iowrite32(reg | WDT_MODE_KEY, wdt_base + WDT_MODE); + + /* + * Flush the posted write before releasing the lock and notifying + * the watchdog core. + */ + ioread32(wdt_base + WDT_MODE); + + spin_unlock_irqrestore(&mtk_wdt->lock, flags); +} + static irqreturn_t mtk_wdt_isr(int irq, void *arg) { struct watchdog_device *wdd = arg; + mtk_wdt_deassert_irq(wdd); watchdog_notify_pretimeout(wdd); return IRQ_HANDLED; @@ -406,6 +451,8 @@ static int mtk_wdt_probe(struct platform_device *pdev) if (!mtk_wdt) return -ENOMEM; + spin_lock_init(&mtk_wdt->lock); + platform_set_drvdata(pdev, mtk_wdt); mtk_wdt->wdt_base = devm_platform_ioremap_resource(pdev, 0); @@ -414,8 +461,8 @@ static int mtk_wdt_probe(struct platform_device *pdev) irq = platform_get_irq_optional(pdev, 0); if (irq > 0) { - err = devm_request_irq(&pdev->dev, irq, mtk_wdt_isr, 0, "wdt_bark", - &mtk_wdt->wdt_dev); + err = devm_request_irq(&pdev->dev, irq, mtk_wdt_isr, IRQF_NO_AUTOEN, + "wdt_bark", &mtk_wdt->wdt_dev); if (err) return err; @@ -447,6 +494,9 @@ static int mtk_wdt_probe(struct platform_device *pdev) if (unlikely(err)) return err; + if (irq > 0) + enable_irq(irq); + dev_info(dev, "Watchdog enabled (timeout=%d sec, nowayout=%d)\n", mtk_wdt->wdt_dev.timeout, nowayout); From 65a821b50468f4c819041419b0b60cb11443a5ea Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Mon, 17 Aug 2026 20:25:37 +0100 Subject: [PATCH 220/857] dt-bindings: watchdog: renesas,r9a09g057-wdt: Add CPG/MSSR syscon support On the Renesas RZ/T2H SoC, the Watchdog Timer Control Register (WDTDCR) resides within the CPG/MSSR block rather than the WDT address space itself. Previously, this was handled by including a second register range in the "reg" property. However, this is architecturally incorrect as the CPG/MSSR block consists of two distinct regions (0x80280000 and 0x81280000) that contain registers for multiple peripheral blocks. Since the CPG/MSSR block is a multi-function block, introduce the "renesas,sysc" phandle-array property to allow the WDT driver to access its control register via this shared regmap. Mark the use of a second "reg" entry as deprecated in favor of the new phandle-array approach for SoCs that require WDTDCR access. Signed-off-by: Lad Prabhakar Reviewed-by: Rob Herring (Arm) Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260817192540.423994-5-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Guenter Roeck --- .../watchdog/renesas,r9a09g057-wdt.yaml | 29 +++++++++++++++++-- 1 file changed, 27 insertions(+), 2 deletions(-) diff --git a/Documentation/devicetree/bindings/watchdog/renesas,r9a09g057-wdt.yaml b/Documentation/devicetree/bindings/watchdog/renesas,r9a09g057-wdt.yaml index 975c5aa4d747f5..322b6921c147f2 100644 --- a/Documentation/devicetree/bindings/watchdog/renesas,r9a09g057-wdt.yaml +++ b/Documentation/devicetree/bindings/watchdog/renesas,r9a09g057-wdt.yaml @@ -48,6 +48,17 @@ properties: resets: maxItems: 1 + renesas,sysc: + description: + System controller registers control the start/stop of the WDT, and halt debug. + $ref: /schemas/types.yaml#/definitions/phandle-array + items: + - items: + - description: phandle to system controller + - description: watchdog IP instance index + minimum: 0 + maximum: 5 + timeout-sec: true required: @@ -73,15 +84,29 @@ allOf: minItems: 2 clock-names: minItems: 2 + renesas,sysc: false else: properties: clocks: maxItems: 1 clock-names: maxItems: 1 - reg: - minItems: 2 resets: false + allOf: + - if: + required: + - renesas,sysc + then: + properties: + reg: + maxItems: 1 + else: + properties: + reg: + description: Deprecated. Use the renesas,sysc property along with + the watchdog IP instance index instead. + minItems: 2 + deprecated: true additionalProperties: false From 26f5abb98a9cde6825d3cdb12a690d0ea7ede5b4 Mon Sep 17 00:00:00 2001 From: "Jason A. Donenfeld" Date: Sun, 30 Aug 2026 21:46:44 -0600 Subject: [PATCH 221/857] virt: vmgenid: move to using dev_set/get_drvdata The prior commit moved the order of initializing driver_data around. In looking through the tree at what is normally done, it appears that actually few drives set or get driver_data directly, but instead go through the dev_set/get_drvdata helpers, which are simple inline helpers that amount to the same exact code. So, for the sake of consistency, use the helpers. Signed-off-by: Jason A. Donenfeld --- drivers/virt/vmgenid.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/drivers/virt/vmgenid.c b/drivers/virt/vmgenid.c index 0d269edf283d89..ce957f5b1c317d 100644 --- a/drivers/virt/vmgenid.c +++ b/drivers/virt/vmgenid.c @@ -25,7 +25,7 @@ struct vmgenid_state { static void vmgenid_notify(struct device *device) { - struct vmgenid_state *state = device->driver_data; + struct vmgenid_state *state = dev_get_drvdata(device); u8 old_id[VMGENID_SIZE]; memcpy(old_id, state->this_id, sizeof(old_id)); @@ -83,7 +83,7 @@ static int vmgenid_add_acpi(struct device *dev, struct vmgenid_state *state) } setup_vmgenid_state(state, virt_addr); - dev->driver_data = state; + dev_set_drvdata(dev, state); status = acpi_install_notify_handler(device->handle, ACPI_DEVICE_NOTIFY, vmgenid_acpi_handler, dev); @@ -125,7 +125,7 @@ static int vmgenid_add_of(struct platform_device *pdev, if (ret < 0) return ret; - pdev->dev.driver_data = state; + dev_set_drvdata(&pdev->dev, state); ret = devm_request_irq(&pdev->dev, ret, vmgenid_of_irq_handler, IRQF_SHARED, "vmgenid", &pdev->dev); From 719ecb98ffe7ce040abc263e00befbb201d5be97 Mon Sep 17 00:00:00 2001 From: "Matthew Wilcox (Oracle)" Date: Thu, 6 Aug 2026 12:58:24 -0400 Subject: [PATCH 222/857] buffer_head: Remove b_page All users except bh_offset() have been converted to use b_folio instead. Convert bh_offset() and remove b_page. Signed-off-by: Matthew Wilcox (Oracle) Signed-off-by: Chao Shi Acked-by: Weidong Zhu Reviewed-by: Jan Kara Link: https://patch.msgid.link/e9d168901578902bffe78ac8b5dfaf1210ee7fb3.1785951556.git.coshi036@gmail.com Signed-off-by: Christian Brauner (Amutable) --- include/linux/buffer_head.h | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index fd2c7115c05427..699970b4bbf28d 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -59,10 +59,7 @@ struct address_space; struct buffer_head { unsigned long b_state; /* buffer state bitmap (see above) */ struct buffer_head *b_this_page;/* circular list of page's buffers */ - union { - struct page *b_page; /* the page this bh is mapped to */ - struct folio *b_folio; /* the folio this bh is mapped to */ - }; + struct folio *b_folio; /* the folio this bh is mapped to */ sector_t b_blocknr; /* start block number */ size_t b_size; /* size of mapping */ @@ -172,7 +169,7 @@ static __always_inline int buffer_uptodate(const struct buffer_head *bh) static inline unsigned long bh_offset(const struct buffer_head *bh) { - return (unsigned long)(bh)->b_data & (page_size(bh->b_page) - 1); + return (unsigned long)(bh)->b_data & (folio_size(bh->b_folio) - 1); } /* If we *know* page->private refers to buffer_heads */ From 8deae22849765920653b7b69d2bdaca413703bee Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:25 -0400 Subject: [PATCH 223/857] buffer: allow a buffer_head to point at memory outside the page cache jbd2 builds a temporary buffer_head to write out the frozen copy of a metadata block, and that copy lives in slab memory. Today jbd2 points the temporary buffer at the slab folio backing it. A slab folio's ->mapping is not an address_space, so anything that follows bh->b_folio->mapping there gets garbage rather than NULL; mark_buffer_write_io_error() does exactly that, and we are about to start calling it on this buffer. Rather than teach every such helper about slab folios, allow bh->b_folio to be NULL and let b_data point straight at the memory. Code that needs the folio has to check. There are two places in this file: - __bh_submit() adds the data by virtual address using bio_add_virt_nofail(), and skips the cgroup accounting: a buffer that is not in the page cache has no owning folio to attribute writeback to. - buffer_set_crypto_ctx() returns early. fscrypt has no interest in a buffer that is not part of a file mapping, which is why it already returns when the folio has no mapping. Nothing sets b_folio to NULL yet, so this patch is a no-op on its own. A buffer_head without a folio is a narrow thing, not a new general capability. Most of the buffer_head API assumes a folio and will fault or corrupt state without one - touch_buffer(), bh_offset(), the async read completion path, and plenty more - so it is up to whoever builds such a buffer to keep it away from all of that. What NULL buys us is that getting it wrong fails loudly instead of quietly following a slab folio's overloaded ->mapping. It is also only valid over memory that is always mapped: buffers over highmem have no permanent kernel virtual address, which is why folio_set_bh() records a folio and an offset instead. Suggested-by: Matthew Wilcox (Oracle) Acked-by: Weidong Zhu Signed-off-by: Chao Shi Reviewed-by: Jan Kara Link: https://patch.msgid.link/bb6fab2111a48d7ba61887fc1372362576fce6e6.1785951556.git.coshi036@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/buffer.c | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/fs/buffer.c b/fs/buffer.c index ed966fa73b1ba2..4f3b33ae598f00 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -1070,12 +1070,16 @@ EXPORT_SYMBOL(__bforget); static void buffer_set_crypto_ctx(struct bio *bio, const struct buffer_head *bh, gfp_t gfp_mask) { - const struct address_space *mapping = folio_mapping(bh->b_folio); + const struct address_space *mapping; /* * The ext4 journal (jbd2) can submit a buffer_head it directly created - * for a non-pagecache page. fscrypt doesn't care about these. + * for memory that is not in the page cache at all. fscrypt doesn't + * care about these. */ + if (!bh->b_folio) + return; + mapping = folio_mapping(bh->b_folio); if (!mapping) return; fscrypt_set_bio_crypt_ctx(bio, mapping->host, @@ -1116,7 +1120,11 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, bio->bi_iter.bi_sector = bh->b_blocknr * (bh->b_size >> 9); bio->bi_write_hint = write_hint; - bio_add_folio_nofail(bio, bh->b_folio, bh->b_size, bh_offset(bh)); + if (bh->b_folio) + bio_add_folio_nofail(bio, bh->b_folio, bh->b_size, + bh_offset(bh)); + else + bio_add_virt_nofail(bio, bh->b_data, bh->b_size); bio->bi_end_io = end_bio; bio->bi_private = bh; @@ -1126,7 +1134,8 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, if (wbc) { wbc_init_bio(wbc, bio); - wbc_account_cgroup_owner(wbc, bh->b_folio, bh->b_size); + if (bh->b_folio) + wbc_account_cgroup_owner(wbc, bh->b_folio, bh->b_size); } blk_crypto_submit_bio(bio); From 5febcba29792135ec3a9260449bc72fa2b99003a Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:26 -0400 Subject: [PATCH 224/857] jbd2: point the shadow buffer at the frozen data directly When a metadata buffer has to be copied out before it can be journalled, jbd2_journal_write_metadata_buffer() writes jh->b_frozen_data rather than the page cache copy. b_frozen_data is kmalloc()ed, so folio_set_bh() makes the shadow buffer point at a slab folio. That is not something the buffer_head layer can reason about. A slab folio overloads ->mapping, so a shadow buffer looks like it belongs to an address_space when it does not. buffer_set_crypto_ctx() already has to work around this, and it is the reason mark_buffer_write_io_error() cannot be called on a shadow buffer today. Point the shadow buffer at the frozen data itself instead: leave b_folio NULL, which it already is out of alloc_buffer_head(), and set b_data. The previous patch taught fs/buffer.c to submit such a buffer. folio_set_bh() is now needed on only one path - the one that journals the page cache copy directly - so it moves there, and new_folio, new_offset and the flag that used to pick between them all go away. The two commit-path checksum helpers reach the shadow buffer's contents through a new kmap_local_bh()/kunmap_local_bh() pair, which handle a buffer with or without a folio. Memory outside the page cache is always mapped, so for those there is nothing to map or unmap. Mapping it anyway would be worse than pointless: with CONFIG_DEBUG_KMAP_LOCAL_FORCE_MAP, kmap_local_page() hands back a one page mapping even for such memory, which is not enough for a buffer bigger than a page. Tested with ext4 mounted data=journal,journal_checksum on a metadata_csum filesystem, writing files whose every block begins with the JBD2 magic so that escaping forces the copy-out, then crashing with sysrq-b without unmounting and replaying the journal on the next mount. Recovery completed, the file contents matched, e2fsck -fn was clean, and an instrumented build confirmed the b_folio == NULL path was taken. Suggested-by: Matthew Wilcox (Oracle) Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/6140cd23beb88e99f40eaeff4044a16213f6caab.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/jbd2/commit.c | 8 ++++---- fs/jbd2/journal.c | 29 +++++++++++++++++------------ include/linux/buffer_head.h | 29 +++++++++++++++++++++++++++++ 3 files changed, 50 insertions(+), 16 deletions(-) diff --git a/fs/jbd2/commit.c b/fs/jbd2/commit.c index 3029cb6f6d640e..0c85af91f9b21f 100644 --- a/fs/jbd2/commit.c +++ b/fs/jbd2/commit.c @@ -330,9 +330,9 @@ static __u32 jbd2_checksum_data(__u32 crc32_sum, struct buffer_head *bh) char *addr; __u32 checksum; - addr = kmap_local_folio(bh->b_folio, bh_offset(bh)); + addr = kmap_local_bh(bh); checksum = crc32_be(crc32_sum, addr, bh->b_size); - kunmap_local(addr); + kunmap_local_bh(bh, addr); return checksum; } @@ -357,10 +357,10 @@ static void jbd2_block_tag_csum_set(journal_t *j, journal_block_tag_t *tag, return; seq = cpu_to_be32(sequence); - addr = kmap_local_folio(bh->b_folio, bh_offset(bh)); + addr = kmap_local_bh(bh); csum32 = jbd2_chksum(j->j_csum_seed, (__u8 *)&seq, sizeof(seq)); csum32 = jbd2_chksum(csum32, addr, bh->b_size); - kunmap_local(addr); + kunmap_local_bh(bh, addr); if (jbd2_has_feature_csum3(j)) tag3->t_checksum = cpu_to_be32(csum32); diff --git a/fs/jbd2/journal.c b/fs/jbd2/journal.c index 00f5a98f3d4fe6..a4f63d73833723 100644 --- a/fs/jbd2/journal.c +++ b/fs/jbd2/journal.c @@ -328,8 +328,6 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, { int do_escape = 0; struct buffer_head *new_bh; - struct folio *new_folio; - unsigned int new_offset; struct buffer_head *bh_in = jh2bh(jh_in); journal_t *journal = transaction->t_journal; @@ -349,24 +347,31 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, /* keep subsequent assertions sane */ atomic_set(&new_bh->b_count, 1); + /* + * b_frozen_data is slab memory, not page cache, so when we use it the + * shadow buffer gets no folio at all: b_folio stays NULL from the + * allocation and b_data points straight at the copy. Pointing it at + * the slab folio instead would hand its overloaded ->mapping to + * anything that goes looking for an address_space. + */ + spin_lock(&jh_in->b_state_lock); /* * If a new transaction has already done a buffer copy-out, then * we use that version of the data for the commit. */ if (jh_in->b_frozen_data) { - new_folio = virt_to_folio(jh_in->b_frozen_data); - new_offset = offset_in_folio(new_folio, jh_in->b_frozen_data); do_escape = jbd2_data_needs_escaping(jh_in->b_frozen_data); if (do_escape) jbd2_data_do_escape(jh_in->b_frozen_data); + new_bh->b_data = jh_in->b_frozen_data; } else { + struct folio *folio = bh_in->b_folio; + unsigned int offset = offset_in_folio(folio, bh_in->b_data); char *tmp; char *mapped_data; - new_folio = bh_in->b_folio; - new_offset = offset_in_folio(new_folio, bh_in->b_data); - mapped_data = kmap_local_folio(new_folio, new_offset); + mapped_data = kmap_local_folio(folio, offset); /* * Fire data frozen trigger if data already wasn't frozen. Do * this before checking for escaping, as the trigger may modify @@ -380,8 +385,10 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, /* * Do we need to do a data copy? */ - if (!do_escape) + if (!do_escape) { + folio_set_bh(new_bh, folio, offset); goto escape_done; + } spin_unlock(&jh_in->b_state_lock); tmp = kmalloc(bh_in->b_size, GFP_NOFS | __GFP_NOFAIL); @@ -392,7 +399,7 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, } jh_in->b_frozen_data = tmp; - memcpy_from_folio(tmp, new_folio, new_offset, bh_in->b_size); + memcpy_from_folio(tmp, folio, offset, bh_in->b_size); /* * This isn't strictly necessary, as we're using frozen * data for the escaping, but it keeps consistency with @@ -401,13 +408,11 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, jh_in->b_frozen_triggers = jh_in->b_triggers; copy_done: - new_folio = virt_to_folio(jh_in->b_frozen_data); - new_offset = offset_in_folio(new_folio, jh_in->b_frozen_data); jbd2_data_do_escape(jh_in->b_frozen_data); + new_bh->b_data = jh_in->b_frozen_data; } escape_done: - folio_set_bh(new_bh, new_folio, new_offset); new_bh->b_size = bh_in->b_size; new_bh->b_bdev = journal->j_dev; new_bh->b_blocknr = blocknr; diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index 699970b4bbf28d..20b8fca1abfaab 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -172,6 +172,35 @@ static inline unsigned long bh_offset(const struct buffer_head *bh) return (unsigned long)(bh)->b_data & (folio_size(bh->b_folio) - 1); } +/** + * kmap_local_bh - Map the data of a buffer. + * @bh: The buffer. + * + * Buffers usually live in the page cache, but a few are built over memory + * which is not. Those carry no folio and b_data is already a kernel address + * which is always mapped, so there is nothing to do for them. Pair with + * kunmap_local_bh(). + * + * Return: A pointer to the buffer's data. + */ +static inline void *kmap_local_bh(const struct buffer_head *bh) +{ + if (!bh->b_folio) + return bh->b_data; + return kmap_local_folio(bh->b_folio, bh_offset(bh)); +} + +/** + * kunmap_local_bh - Unmap the data of a buffer. + * @bh: The buffer. + * @addr: The address returned by kmap_local_bh(). + */ +static inline void kunmap_local_bh(const struct buffer_head *bh, void *addr) +{ + if (bh->b_folio) + kunmap_local(addr); +} + /* If we *know* page->private refers to buffer_heads */ #define page_buffers(page) \ ({ \ From 14cbade7d77348b681631567ec873d7d681b6c7f Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:27 -0400 Subject: [PATCH 225/857] buffer: read the folio's mapping directly in buffer_set_crypto_ctx() folio_mapping() was doing two jobs here. One was to turn a slab folio into NULL, which is what made this safe for jbd2's shadow buffers; the previous patch removed the need for that by giving those buffers no folio at all. The other is a hazard. folio_mapping() maps a folio in the swap cache to its swap_address_space, so if a buffer_head were ever attached to such a folio this would hand fscrypt a swap mapping and dereference ->host on it. There is no reason to want that here: this path wants the file's mapping or nothing. Read ->mapping directly. Buffers with no folio are already handled above. Suggested-by: Matthew Wilcox (Oracle) Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/3fe72ec37bf8491a69031db5f3ba1319da935b97.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/buffer.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/buffer.c b/fs/buffer.c index 4f3b33ae598f00..ecd1a4f0f39953 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -1079,7 +1079,7 @@ static void buffer_set_crypto_ctx(struct bio *bio, const struct buffer_head *bh, */ if (!bh->b_folio) return; - mapping = folio_mapping(bh->b_folio); + mapping = bh->b_folio->mapping; if (!mapping) return; fscrypt_set_bio_crypt_ctx(bio, mapping->host, From d59fe9111f4f035406f423c905e2f1a64ee33871 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:28 -0400 Subject: [PATCH 226/857] buffer: clear BH_Write_EIO when a buffer is forgotten BH_Write_EIO records that the last write of this buffer failed. It is cleared when the buffer is written again, but a filesystem freeing a metadata block never writes it again. It calls bforget() and hands the block back to the allocator, so the flag outlives the block it refers to. That does not matter much today, because the write error is also recorded by clearing BH_Uptodate and the buffer is discarded soon after. It starts to matter in the rest of this series, which stops clearing BH_Uptodate on write error and makes BH_Write_EIO the way a failed metadata write is reported. bforget() is where a filesystem says it no longer cares about this buffer's contents, so clear the error there alongside the dirty flag. Suggested-by: Jan Kara Acked-by: Weidong Zhu Signed-off-by: Chao Shi Reviewed-by: Jan Kara Link: https://patch.msgid.link/ebe0b4f179ccdac7a9400fe2611623ce218c8d87.1785951556.git.coshi036@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/buffer.c | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/buffer.c b/fs/buffer.c index ecd1a4f0f39953..54dbf02b1b88f8 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -1062,6 +1062,7 @@ EXPORT_SYMBOL(__brelse); void __bforget(struct buffer_head *bh) { clear_buffer_dirty(bh); + clear_buffer_write_io_error(bh); remove_assoc_queue(bh); __brelse(bh); } From bbcd6f7456b28feed46f6c244657d20ef713a475 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:29 -0400 Subject: [PATCH 227/857] buffer: discard BH_Write_EIO along with the rest of the buffer state discard_buffer() strips the state that describes where a buffer lives and what has happened to it, because after an invalidate none of it applies any more. BH_Write_EIO belongs in that set for the same reason: it describes a write of the data that is being thrown away. Leaving it set means a buffer_head reused for a different block starts life carrying somebody else's write error. Like the bforget() change, this is mostly theoretical today and becomes load bearing once the rest of the series makes BH_Write_EIO the report of a failed metadata write. Suggested-by: Jan Kara Acked-by: Weidong Zhu Signed-off-by: Chao Shi Reviewed-by: Jan Kara Link: https://patch.msgid.link/c6e9db48d8d0feb83d4ca29306f4bc1e58f1ee0f.1785951556.git.coshi036@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/buffer.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/buffer.c b/fs/buffer.c index 54dbf02b1b88f8..2cb1dc5d6c21ce 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -1492,7 +1492,7 @@ EXPORT_SYMBOL(folio_set_bh); /* Bits that are cleared during an invalidate */ #define BUFFER_FLAGS_DISCARD \ (1 << BH_Mapped | 1 << BH_New | 1 << BH_Req | \ - 1 << BH_Delay | 1 << BH_Unwritten) + 1 << BH_Delay | 1 << BH_Unwritten | 1 << BH_Write_EIO) static void discard_buffer(struct buffer_head * bh) { From 7032ded0a1fae503d3a5339494c3f9cf8d011137 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:30 -0400 Subject: [PATCH 228/857] buffer: detect metadata write errors with buffer_write_io_error() Both places in this file that report a metadata write error to a caller do it by testing !buffer_uptodate() after waiting for the write. That works only because the write completion handlers clear BH_Uptodate when the write fails, which is what this series is removing: a buffer whose write failed still holds the correct data, and saying otherwise makes callers rewrite, re-read or WARN over a buffer that was never wrong. BH_Write_EIO is the flag that actually means "the last write of this buffer failed", and both handlers already set it via mark_buffer_write_io_error(). Test that instead. No behaviour change: today a failed write through bh_end_write() or bh_end_async_write() sets BH_Write_EIO and clears BH_Uptodate together, so the two tests agree. They stop agreeing at the end of the series, and this one stays right. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Reviewed-by: Jan Kara Link: https://patch.msgid.link/2b309196b883cc8979800911a47668e225401b2c.1785951556.git.coshi036@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/buffer.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/buffer.c b/fs/buffer.c index 2cb1dc5d6c21ce..b7c6ea0dbe6b19 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -589,7 +589,7 @@ int mmb_sync(struct mapping_metadata_bhs *mmb) } spin_unlock(&mmb->lock); wait_on_buffer(bh); - if (!buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) err = -EIO; brelse(bh); spin_lock(&mmb->lock); @@ -2720,7 +2720,7 @@ int __sync_dirty_buffer(struct buffer_head *bh, blk_opf_t op_flags) bh_submit(bh, REQ_OP_WRITE | op_flags, bh_end_write); wait_on_buffer(bh); - if (!buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) return -EIO; } else { unlock_buffer(bh); From 613e12102111c211670cbecd3bf3b63e0f14be19 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:31 -0400 Subject: [PATCH 229/857] adfs: check for a directory write error with buffer_write_io_error() adfs_dir_sync() spots a failed write by testing BH_Req together with !BH_Uptodate. That relies on the write completion handler clearing BH_Uptodate on error, which this series removes: a buffer whose write failed still holds the data the filesystem asked to be written, so declaring it not up to date is wrong and makes callers re-read it. BH_Write_EIO says exactly what this code wants to know, and it implies BH_Req, so the pair collapses into one test. No behaviour change today - a failed write sets BH_Write_EIO and clears BH_Uptodate together. It stops being a no-op at the end of the series, where the new test is the one that still works. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/05cf8ad911ad7a6eb68f42e38f1eea39c38e246b.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/adfs/dir.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/adfs/dir.c b/fs/adfs/dir.c index 11afa9e157aa70..b8cc6a697a05dd 100644 --- a/fs/adfs/dir.c +++ b/fs/adfs/dir.c @@ -191,7 +191,7 @@ static int adfs_dir_sync(struct adfs_dir *dir) for (i = dir->nr_buffers - 1; i >= 0; i--) { struct buffer_head *bh = dir->bhs[i]; sync_dirty_buffer(bh); - if (buffer_req(bh) && !buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) err = -EIO; } From 583913a40f2ea45276b3689e9a1d55754debe5cd Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:32 -0400 Subject: [PATCH 230/857] ext2: check for an xattr block write error with buffer_write_io_error() ext2_xattr_set2() spots a failed synchronous write by testing BH_Req together with !BH_Uptodate. That relies on the write completion handler clearing BH_Uptodate on error, which this series removes: a buffer whose write failed still holds the data the filesystem asked to be written, so declaring it not up to date is wrong and makes callers re-read it. BH_Write_EIO says exactly what this code wants to know, and it implies BH_Req, so the pair collapses into one test. No behaviour change today - a failed write sets BH_Write_EIO and clears BH_Uptodate together. It stops being a no-op at the end of the series, where the new test is the one that still works. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Reviewed-by: Jan Kara Link: https://patch.msgid.link/fcc530fffd0f1e18e8cd1a3974bc0ccc5b4c4b66.1785951556.git.coshi036@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/ext2/xattr.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/ext2/xattr.c b/fs/ext2/xattr.c index 9b68c490ab26de..8f608930a48c64 100644 --- a/fs/ext2/xattr.c +++ b/fs/ext2/xattr.c @@ -769,7 +769,7 @@ ext2_xattr_set2(struct inode *inode, struct buffer_head *old_bh, if (IS_SYNC(inode)) { sync_dirty_buffer(new_bh); error = -EIO; - if (buffer_req(new_bh) && !buffer_uptodate(new_bh)) + if (buffer_write_io_error(new_bh)) goto cleanup; } } From a269f6fade99bbcad2a44a88ba8e6e70e842ca3d Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:33 -0400 Subject: [PATCH 231/857] omfs: check for an inode write error with buffer_write_io_error() __omfs_write_inode() spots a failed synchronous write, on both the primary block and each mirror, by testing BH_Req together with !BH_Uptodate. That relies on the write completion handler clearing BH_Uptodate on error, which this series removes: a buffer whose write failed still holds the data the filesystem asked to be written, so declaring it not up to date is wrong and makes callers re-read it. BH_Write_EIO says exactly what this code wants to know, and it implies BH_Req, so each pair collapses into one test. No behaviour change today - a failed write sets BH_Write_EIO and clears BH_Uptodate together. It stops being a no-op at the end of the series, where the new test is the one that still works. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/b33ccb12c29ae743ccfa48cdf1e9140fbf2dabcc.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/omfs/inode.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/omfs/inode.c b/fs/omfs/inode.c index 1d915ef72119fb..bc37029a4afb1f 100644 --- a/fs/omfs/inode.c +++ b/fs/omfs/inode.c @@ -145,7 +145,7 @@ static int __omfs_write_inode(struct inode *inode, int wait) mark_buffer_dirty(bh); if (wait) { sync_dirty_buffer(bh); - if (buffer_req(bh) && !buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) sync_failed = 1; } @@ -159,7 +159,7 @@ static int __omfs_write_inode(struct inode *inode, int wait) mark_buffer_dirty(bh2); if (wait) { sync_dirty_buffer(bh2); - if (buffer_req(bh2) && !buffer_uptodate(bh2)) + if (buffer_write_io_error(bh2)) sync_failed = 1; } brelse(bh2); From 4a53ece574b5cd3f4010604ac0c2877b6e4d9391 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:34 -0400 Subject: [PATCH 232/857] exfat: check for a directory write error with buffer_write_io_error() exfat_update_bhs() waits for the writes it issued and then tests !buffer_uptodate() to find the ones that failed. That relies on the write completion handler clearing BH_Uptodate on error, which this series removes: a buffer whose write failed still holds the data the filesystem asked to be written, so declaring it not up to date is wrong and makes callers re-read it. Test BH_Write_EIO, which is what the completion handler sets and what this code actually wants to know. No behaviour change today - a failed write sets BH_Write_EIO and clears BH_Uptodate together. It stops being a no-op at the end of the series, where the new test is the one that still works. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/0784ef63a525434e7c0aff730eca7b43e043095d.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/exfat/misc.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/exfat/misc.c b/fs/exfat/misc.c index 6f11a96a4ffa84..dfd0bbf31c940a 100644 --- a/fs/exfat/misc.c +++ b/fs/exfat/misc.c @@ -187,7 +187,7 @@ int exfat_update_bhs(struct buffer_head **bhs, int nr_bhs, int sync) for (i = 0; i < nr_bhs && sync; i++) { wait_on_buffer(bhs[i]); - if (!err && !buffer_uptodate(bhs[i])) + if (!err && buffer_write_io_error(bhs[i])) err = -EIO; } return err; From a93ff8e2e398e5b2c252230a6878990808caf360 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:35 -0400 Subject: [PATCH 233/857] fat: check for a metadata write error with buffer_write_io_error() fat_sync_bhs() waits for the writes it issued and then tests !buffer_uptodate() to find the ones that failed. That relies on the write completion handler clearing BH_Uptodate on error, which this series removes: a buffer whose write failed still holds the data the filesystem asked to be written, so declaring it not up to date is wrong and makes callers re-read it. Test BH_Write_EIO, which is what the completion handler sets and what this code actually wants to know. No behaviour change today - a failed write sets BH_Write_EIO and clears BH_Uptodate together. It stops being a no-op at the end of the series, where the new test is the one that still works. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/4b6019b5a48b83c8235918b084248983a57692e1.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/fat/misc.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fat/misc.c b/fs/fat/misc.c index e79762cf19754d..0d04228f916e1e 100644 --- a/fs/fat/misc.c +++ b/fs/fat/misc.c @@ -360,7 +360,7 @@ int fat_sync_bhs(struct buffer_head **bhs, int nr_bhs) for (i = 0; i < nr_bhs; i++) { wait_on_buffer(bhs[i]); - if (!err && !buffer_uptodate(bhs[i])) + if (!err && buffer_write_io_error(bhs[i])) err = -EIO; } return err; From 0f44b420e6c76cbadc07b797cd2a02d4d7bb8824 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:36 -0400 Subject: [PATCH 234/857] ext4: check for a metadata write error with buffer_write_io_error() Two places detect a failed metadata write by testing !buffer_uptodate() after waiting for it. That relies on the write completion handler clearing BH_Uptodate on error, which this series removes: a buffer whose write failed still holds the data the filesystem asked to be written, so declaring it not up to date is wrong and makes callers re-read it. ext4 already does this correctly for the superblock - see ext4_commit_super(), which tests buffer_write_io_error() - so this brings the other two into line. In __ext4_handle_dirty_metadata() the old test also required BH_Req. BH_Write_EIO implies it, so the pair collapses into one test. The new test is also strictly stronger than consuming sync_dirty_buffer()'s return value, because it still fires when the buffer was written by background writeback and that write hit an error, which sync_dirty_buffer() does not report. Note that the failing write does not clear BH_Write_EIO, so an unrepaired itable block now reports on every subsequent sync of that inode rather than only on the write that failed. That is the intended behaviour, and matches what ocfs2 has always done with this flag. No behaviour change today - a failed write sets BH_Write_EIO and clears BH_Uptodate together. It stops being a no-op at the end of the series, where the new test is the one that still works. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Reviewed-by: Jan Kara Link: https://patch.msgid.link/568e57d184da17041872fd4b498f31dd6259a4c1.1785951556.git.coshi036@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/ext4/ext4_jbd2.c | 2 +- fs/ext4/mmp.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/ext4/ext4_jbd2.c b/fs/ext4/ext4_jbd2.c index 53ddedb52a6f73..c241f50b97bc0a 100644 --- a/fs/ext4/ext4_jbd2.c +++ b/fs/ext4/ext4_jbd2.c @@ -421,7 +421,7 @@ int __ext4_handle_dirty_metadata(const char *where, unsigned int line, } if (inode && inode_needs_sync(inode)) { sync_dirty_buffer(bh); - if (buffer_req(bh) && !buffer_uptodate(bh)) { + if (buffer_write_io_error(bh)) { ext4_error_inode_err(inode, where, line, bh->b_blocknr, EIO, "IO error syncing itable block"); diff --git a/fs/ext4/mmp.c b/fs/ext4/mmp.c index 7ce361484b3821..4b18ddef468d71 100644 --- a/fs/ext4/mmp.c +++ b/fs/ext4/mmp.c @@ -49,7 +49,7 @@ static int write_mmp_block_thawed(struct super_block *sb, bh_submit(bh, REQ_OP_WRITE | REQ_SYNC | REQ_META | REQ_PRIO, bh_end_write); wait_on_buffer(bh); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) return -EIO; return 0; } From 3f90f64f818720af930e811190181d9b263e413a Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:37 -0400 Subject: [PATCH 235/857] ocfs2: check for a metadata write error with buffer_write_io_error() ocfs2_write_block() and ocfs2_write_super_or_backup() detect a failed write by looking at BH_Uptodate afterwards. That relies on the write completion handler clearing BH_Uptodate on error, which this series removes: a buffer whose write failed still holds the data the filesystem asked to be written, so declaring it not up to date is wrong and makes callers re-read it. Test BH_Write_EIO instead. Note that ocfs2_write_block()'s test is the positive one, so the sense has to be inverted rather than the flag simply swapped. The comment in ocfs2_write_block()'s error arm needs updating for the same reason. It said the clustered uptodate information did not have to be removed because the buffer was not marked locally uptodate; after this series it is, so the reason no longer holds. Not advertising the block to the cluster is still the right thing to do - the data is in memory but not on disk - so only the justification changes, not the behaviour. No behaviour change today - a failed write sets BH_Write_EIO and clears BH_Uptodate together. It stops being a no-op at the end of the series, where the new test is the one that still works. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Reviewed-by: Jan Kara Link: https://patch.msgid.link/05cd3c34f2dc69b542db7baaef058cd75005d4cb.1785951556.git.coshi036@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/ocfs2/buffer_head_io.c | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/fs/ocfs2/buffer_head_io.c b/fs/ocfs2/buffer_head_io.c index 7bfe377af2dfcf..733ceda79ca1f1 100644 --- a/fs/ocfs2/buffer_head_io.c +++ b/fs/ocfs2/buffer_head_io.c @@ -66,12 +66,14 @@ int ocfs2_write_block(struct ocfs2_super *osb, struct buffer_head *bh, wait_on_buffer(bh); - if (buffer_uptodate(bh)) { + if (!buffer_write_io_error(bh)) { ocfs2_set_buffer_uptodate(ci, bh); } else { - /* We don't need to remove the clustered uptodate - * information for this bh as it's not marked locally - * uptodate. */ + /* + * The buffer still holds what we tried to write, but it did + * not reach the disk, so don't advertise it to the cluster + * as up to date. + */ ret = -EIO; mlog_errno(ret); } @@ -446,7 +448,7 @@ int ocfs2_write_super_or_backup(struct ocfs2_super *osb, wait_on_buffer(bh); - if (!buffer_uptodate(bh)) { + if (buffer_write_io_error(bh)) { ret = -EIO; mlog_errno(ret); } From a87462d7b604423a4165dd1b7f09f89fc29c9ca9 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:38 -0400 Subject: [PATCH 236/857] ocfs2: check for a stale write error before reusing a metadata buffer __ocfs2_journal_access() refuses to journal a buffer whose previous write failed, and turns the filesystem read-only rather than risk metadata inconsistency. That check sits inside an if (!buffer_uptodate(bh)) block, because until now a failed write also cleared BH_Uptodate. This series stops clearing BH_Uptodate on write error, so that outer test would never fire again and ocfs2 would silently start reusing buffers whose last write failed. Hoist the check out of the debug block, where it does not depend on BH_Uptodate any more, and drop the now dead second half of its condition. The mlog() pair keeps its own !buffer_uptodate() guard: it is a separate "we can safely remove this assertion after testing" debug aid about being handed a buffer with no valid contents, which is a different question from whether the last write of that buffer failed. The unlocked test followed by a locked retest is deliberate. BH_Write_EIO is cleared under the buffer lock, so taking the lock and looking a second time avoids turning the filesystem read-only over an error that a concurrent rewrite has already cleared, while keeping the common case lock-free. The code in this patch is Jan's, from the review discussion linked in the cover letter. Suggested-by: Jan Kara Acked-by: Weidong Zhu Signed-off-by: Chao Shi Reviewed-by: Jan Kara Link: https://patch.msgid.link/ad75c40927060cafe5a8ac45ae71c02ae4dfba04.1785951556.git.coshi036@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/ocfs2/journal.c | 25 +++++++++++++------------ 1 file changed, 13 insertions(+), 12 deletions(-) diff --git a/fs/ocfs2/journal.c b/fs/ocfs2/journal.c index d8afbc1a76bb8a..ea6802d894c2b2 100644 --- a/fs/ocfs2/journal.c +++ b/fs/ocfs2/journal.c @@ -676,19 +676,20 @@ static int __ocfs2_journal_access(handle_t *handle, mlog(ML_ERROR, "giving me a buffer that's not uptodate!\n"); mlog(ML_ERROR, "b_blocknr=%llu, b_state=0x%lx\n", (unsigned long long)bh->b_blocknr, bh->b_state); - + } + /* + * A previous transaction with a couple of buffer heads fail + * to checkpoint, so all the bhs are marked as BH_Write_EIO. + * For current transaction, the bh is just among those error + * bhs which previous transaction handle. We can't just clear + * its BH_Write_EIO and reuse directly, since other bhs are + * not written to disk yet and that will cause metadata + * inconsistency. So we should set fs read-only to avoid + * further damage. + */ + if (buffer_write_io_error(bh)) { lock_buffer(bh); - /* - * A previous transaction with a couple of buffer heads fail - * to checkpoint, so all the bhs are marked as BH_Write_EIO. - * For current transaction, the bh is just among those error - * bhs which previous transaction handle. We can't just clear - * its BH_Write_EIO and reuse directly, since other bhs are - * not written to disk yet and that will cause metadata - * inconsistency. So we should set fs read-only to avoid - * further damage. - */ - if (buffer_write_io_error(bh) && !buffer_uptodate(bh)) { + if (buffer_write_io_error(bh)) { unlock_buffer(bh); return ocfs2_error(osb->sb, "A previous attempt to " "write this buffer head failed\n"); From 47552ff5ebf904de054aa7e4463242eec894cad8 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:39 -0400 Subject: [PATCH 237/857] gfs2: check for a metadata write error with buffer_write_io_error() gfs2_ail1_start_one() and gfs2_ail1_empty_one() decide whether a buffer on the ail reached the disk by looking at BH_Uptodate once it is no longer busy. That relies on the write completion handler clearing BH_Uptodate on error, which this series removes: a buffer whose write failed still holds the data the filesystem asked to be written, so declaring it not up to date is wrong and makes callers re-read it. Test BH_Write_EIO instead. In gfs2_ail1_start_one() the test is the positive one, so the sense has to be inverted rather than the flag simply swapped. This is not a pure conversion for gfs2, because gfs2 already has a private write completion handler that behaves the way this series is heading: gfs2_end_log_write_bh() calls mark_buffer_write_io_error() and leaves BH_Uptodate alone. Buffers completed through it are therefore invisible to both tests today, and start being caught once they look at BH_Write_EIO. That is a real behaviour change, and it is the one gfs2 wanted: a failed log write now withdraws the filesystem instead of passing silently. gfs2_pin() is a different case and gets a different treatment. Its !buffer_uptodate() test is not only a proxy for a failed write - a buffer with no valid contents at all is equally a reason to withdraw before pinning it into a transaction - so the write error test is added to it rather than replacing it. Left alone deliberately: the BUG_ON(!buffer_uptodate(bh)) in gfs2_unpin() and the two WARN_ON()s in fs/gfs2/rgrp.c. After this series they simply stop firing for write errors, which is correct; turning them into BUG_ON(buffer_write_io_error(bh)) would newly panic on an I/O error. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/a09b6d08d31d24f69b833e4d855ed88875e87780.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/gfs2/log.c | 4 ++-- fs/gfs2/lops.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/gfs2/log.c b/fs/gfs2/log.c index 78bba8cc10b8fd..e3e0dcb1f56733 100644 --- a/fs/gfs2/log.c +++ b/fs/gfs2/log.c @@ -107,7 +107,7 @@ __acquires(&sdp->sd_ail_lock) gfs2_assert(sdp, bd->bd_tr == tr); if (!buffer_busy(bh)) { - if (buffer_uptodate(bh)) { + if (!buffer_write_io_error(bh)) { list_move(&bd->bd_ail_st_list, &tr->tr_ail2_list); continue; @@ -321,7 +321,7 @@ static int gfs2_ail1_empty_one(struct gfs2_sbd *sdp, struct gfs2_trans *tr, active_count++; continue; } - if (!buffer_uptodate(bh) && + if (buffer_write_io_error(bh) && !cmpxchg(&sdp->sd_log_error, 0, -EIO)) gfs2_io_error_bh(sdp, bh); /* diff --git a/fs/gfs2/lops.c b/fs/gfs2/lops.c index 6dabe73ad790d9..3df6e4b7e8b9eb 100644 --- a/fs/gfs2/lops.c +++ b/fs/gfs2/lops.c @@ -48,7 +48,7 @@ void gfs2_pin(struct gfs2_sbd *sdp, struct buffer_head *bh) clear_buffer_dirty(bh); if (test_set_buffer_pinned(bh)) gfs2_assert_withdraw(sdp, 0); - if (!buffer_uptodate(bh)) + if (!buffer_uptodate(bh) || buffer_write_io_error(bh)) gfs2_io_error_bh(sdp, bh); bd = bh->b_private; /* If this buffer is in the AIL and it has already been written From 0c6c46fbf4d89c94e3f7da3875706e0cbcdf7e15 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:40 -0400 Subject: [PATCH 238/857] jbd2: report journal write errors with BH_Write_EIO The journal's own write completion handler, journal_end_buffer_io_sync(), reports a failed write by clearing BH_Uptodate, and the three places that wait for journal writes look for that. This series is removing that convention: a buffer whose write failed still holds the data that was supposed to reach the disk, and saying it is not up to date makes callers rewrite, re-read or WARN over data that was never wrong. Set BH_Write_EIO instead, with mark_buffer_write_io_error(), and test it in journal_wait_on_commit_record() and in the two commit-phase waits. The handler stops touching BH_Uptodate in either direction. Setting it on success was never needed: every caller marks the buffer up to date before submitting the write, because a buffer with no valid data is not something you can write out. The local flag is renamed to match what it now means. The two changes have to go together, because commit phase 4 waits on a mixed list: descriptor blocks are submitted with journal_end_buffer_io_sync(), while revoke blocks go through write_dirty_buffer() and land in bh_end_write(). bh_end_write() already sets BH_Write_EIO, so converting the consumer alone would keep working for revoke blocks and silently stop detecting failed descriptor writes. With the handler converted, both halves of the list report the same way. mark_buffer_write_io_error() is safe on all of these buffers. The shadow buffers from jbd2_journal_write_metadata_buffer() have no folio and no associated mapping, so it does nothing beyond setting the flag. Descriptor and commit blocks are ordinary buffers on the journal device, and marking the journal's mapping with the error is what write_dirty_buffer() already does for revoke blocks on the same device. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/c700983eea955429d927791fe5fb9ffe6cf11056.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/jbd2/commit.c | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/fs/jbd2/commit.c b/fs/jbd2/commit.c index 0c85af91f9b21f..cd7ef783bd36a2 100644 --- a/fs/jbd2/commit.c +++ b/fs/jbd2/commit.c @@ -32,14 +32,12 @@ static void journal_end_buffer_io_sync(struct bio *bio) { struct buffer_head *bh; - bool uptodate = bio_endio_bh(bio, &bh); + bool success = bio_endio_bh(bio, &bh); struct buffer_head *orig_bh = bh->b_private; BUFFER_TRACE(bh, ""); - if (uptodate) - set_buffer_uptodate(bh); - else - clear_buffer_uptodate(bh); + if (!success) + mark_buffer_write_io_error(bh); if (orig_bh) { clear_and_wake_up_bit(BH_Shadow, &orig_bh->b_state); } @@ -169,7 +167,7 @@ static int journal_wait_on_commit_record(journal_t *journal, clear_buffer_dirty(bh); wait_on_buffer(bh); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) ret = -EIO; put_bh(bh); /* One for getblk() */ @@ -834,7 +832,7 @@ void jbd2_journal_commit_transaction(journal_t *journal) wait_on_buffer(bh); cond_resched(); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) err = -EIO; jbd2_unfile_log_bh(bh); stats.run.rs_blocks_logged++; @@ -877,7 +875,7 @@ void jbd2_journal_commit_transaction(journal_t *journal) wait_on_buffer(bh); cond_resched(); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) err = -EIO; BUFFER_TRACE(bh, "ph5: control buffer writeout done: unfile"); From 50cf8c8e532c9f76d000d69bc152a074894175aa Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:41 -0400 Subject: [PATCH 239/857] jbd2: say what jbd2_freeze_jh_data()'s assertion is actually checking The assertion that the buffer about to be copied out is up to date is correct and stays, but its message - "Possible IO failure" - describes what a buffer that is not up to date used to mean rather than what is being checked. Once this series stops clearing BH_Uptodate on write error, that reading is wrong twice over. A failed write no longer makes a buffer not up to date, and a buffer that does carry BH_Write_EIO is fine here: it still holds valid data and the journal will write it again. What the assertion is really guarding is that there is something valid to copy at all. Say that instead. Suggested-by: Jan Kara Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/713ffb9c0cdfe55499b3cdd2c06a59a2c152c46c.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/jbd2/transaction.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/jbd2/transaction.c b/fs/jbd2/transaction.c index 5cc7d097b2ac8e..85d84d909f785c 100644 --- a/fs/jbd2/transaction.c +++ b/fs/jbd2/transaction.c @@ -920,7 +920,7 @@ static void jbd2_freeze_jh_data(struct journal_head *jh) char *source; struct buffer_head *bh = jh2bh(jh); - J_EXPECT_JH(jh, buffer_uptodate(bh), "Possible IO failure.\n"); + J_EXPECT_JH(jh, buffer_uptodate(bh), "Buffer not uptodate!\n"); source = kmap_local_folio(bh->b_folio, bh_offset(bh)); /* Fire data frozen trigger just before we copy the data */ jbd2_buffer_frozen_trigger(jh, source, jh->b_triggers); From a49d8a9d5d4912567ba6119e928025aa09fefacb Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:42 -0400 Subject: [PATCH 240/857] jbd2: check for a fast commit write error with buffer_write_io_error() jbd2_fc_wait_bufs() decides whether ext4's fast commit blocks reached the disk by looking at BH_Uptodate once the write has completed. That relies on the write completion handler clearing BH_Uptodate on error, which this series removes: a buffer whose write failed still holds the data the filesystem asked to be written, so declaring it not up to date is wrong and makes callers rewrite or re-read it. Test BH_Write_EIO instead. Nothing is needed on the producer side. These buffers used to be submitted by ext4_fc_submit_bh() with a private completion handler that cleared BH_Uptodate, but commit 7f0485dd3017 ("ext4: remove ext4_end_buffer_io_sync()") dropped it in favour of bh_end_write(), which already reports a failed write with mark_buffer_write_io_error(). The wait in jbd2 is the only half left to convert. A stale flag from an earlier write cannot fool the new test. jbd2_fc_get_buf() looks the buffer up with __getblk() at a fixed block on the journal device, so the same buffer comes back commit after commit, but __bh_submit() clears BH_Write_EIO when it resubmits a buffer for writing. The flag jbd2_fc_wait_bufs() sees always belongs to the write it just waited for. No behaviour change today - bh_end_write() sets BH_Write_EIO and clears BH_Uptodate together. It stops being a no-op at the end of the series, where the new test is the one that still works. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/970d8b9603d4620ab73038f49bc36831e37e00a0.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/jbd2/journal.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/jbd2/journal.c b/fs/jbd2/journal.c index a4f63d73833723..cda1ff8851dcc6 100644 --- a/fs/jbd2/journal.c +++ b/fs/jbd2/journal.c @@ -887,7 +887,7 @@ int jbd2_fc_wait_bufs(journal_t *journal, int num_blks) * Update j_fc_off so jbd2_fc_release_bufs can release remain * buffer head. */ - if (unlikely(!buffer_uptodate(bh))) { + if (unlikely(buffer_write_io_error(bh))) { journal->j_fc_off = i + 1; return -EIO; } From 1f2304e87b831729c29125593ed7b0756fcd27d7 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:43 -0400 Subject: [PATCH 241/857] buffer: stop touching BH_Uptodate on write completion A buffer whose write failed still holds exactly the data the filesystem asked to be written. It is the disk that is out of date, not the buffer. Clearing BH_Uptodate says the opposite, and callers act on it: - mark_buffer_dirty() has a WARN_ON_ONCE(!buffer_uptodate(bh)). A filesystem that dirties the buffer again after a failed write - which is the normal way to retry - trips it. That is the warning this series started from. - a buffer that is not up to date gets re-read from disk, which replaces the data the filesystem was trying to write with the stale on-disk copy, silently. - the window between the write completing and the buffer being marked not up to date is visible to anyone holding the folio lock, so the state is not even self consistent while it lasts. BH_Write_EIO already records the failure, and by now every place in the tree that needs to know about it tests that flag instead: the two core helpers in this file, adfs, exfat, ext2, ext4, fat, gfs2, jbd2, ocfs2 and omfs, converted one filesystem at a time in the preceding patches. The private completion handler in jbd2 was converted along with its waiters, and ext4 fast commit needed only its waiter, because commit 7f0485dd3017 ("ext4: remove ext4_end_buffer_io_sync()") had already dropped its handler in favour of bh_end_write(). Nothing is left that reads BH_Uptodate to find out whether a write failed. Setting BH_Uptodate on success goes too. A buffer has to be up to date before it can be written - you cannot write out data you do not have - so the only thing that assignment could do is paper over a caller that got that wrong. Write completion now leaves BH_Uptodate alone in both directions. What this changes for readers. A buffer whose write failed stays up to date, so the read paths stop replacing it with the on-disk copy: __bread_gfp() no longer sends it to __bread_slow(), and bh_uptodate_or_lock() reports it as usable. That is the intent. ocfs2 changes the most, because ocfs2_read_blocks() decides whether to go to disk on its own cluster uptodate cache and only tests BH_Uptodate after the wait, so a block whose write failed makes that read return -EIO today and from here it succeeds and hands back the in-memory data. A caller that needs to know the write failed asks BH_Write_EIO. Found by FuzzNvme. Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/61d7d5737f5773f53ee543f375fcde81aa8d28c2.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/buffer.c | 10 ++-------- 1 file changed, 2 insertions(+), 8 deletions(-) diff --git a/fs/buffer.c b/fs/buffer.c index b7c6ea0dbe6b19..ad5f707e41364b 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -202,12 +202,9 @@ void bh_end_write(struct bio *bio) struct buffer_head *bh; bool success = bio_endio_bh(bio, &bh); - if (success) { - set_buffer_uptodate(bh); - } else { + if (!success) { buffer_io_error(bh, ", lost sync page write"); mark_buffer_write_io_error(bh); - clear_buffer_uptodate(bh); } unlock_buffer(bh); } @@ -407,12 +404,9 @@ void bh_end_async_write(struct bio *bio) BUG_ON(!buffer_async_write(bh)); folio = bh->b_folio; - if (success) { - set_buffer_uptodate(bh); - } else { + if (!success) { buffer_io_error(bh, ", lost async page write"); mark_buffer_write_io_error(bh); - clear_buffer_uptodate(bh); } first = folio_buffers(folio); From e038f1016c6fe61271fc18b8fb274fa26259f7e2 Mon Sep 17 00:00:00 2001 From: Chao Shi Date: Thu, 6 Aug 2026 12:58:44 -0400 Subject: [PATCH 242/857] buffer: clear BH_Write_EIO when a write succeeds, not when one starts BH_Write_EIO is cleared in __bh_submit(), when a buffer that has been written before is submitted for write again. That is early: it says the error is gone at the moment we start trying to fix it, rather than when we have. It also loses errors. A task whose write fails sets the flag and then goes to look at it; if another task redirties the buffer and resubmits it in between, the submission clears the flag and the first task sees no error at all. Neither of them is doing anything wrong. Clear it on successful write completion instead, in the end io handlers - the same three the rest of this series has been converting, plus gfs2's, which already marked errors this way. Then the flag means what it says: the last write of this buffer that finished, failed. A resubmission no longer hides an error that has not been fixed yet, and one that has been fixed clears the flag when the data reaches the disk. __bh_submit() keeps setting BH_Req, which is what the rest of the tree reads it for. Suggested-by: Jan Kara Acked-by: Weidong Zhu Signed-off-by: Chao Shi Link: https://patch.msgid.link/1c976fd191aa6e99dbe65d6a1ec63f8706cc0dfa.1785951556.git.coshi036@gmail.com Reviewed-by: Jan Kara Signed-off-by: Christian Brauner (Amutable) --- fs/buffer.c | 15 +++++++-------- fs/gfs2/lops.c | 2 ++ fs/jbd2/commit.c | 4 +++- 3 files changed, 12 insertions(+), 9 deletions(-) diff --git a/fs/buffer.c b/fs/buffer.c index ad5f707e41364b..4d329c15a2a59c 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -202,7 +202,9 @@ void bh_end_write(struct bio *bio) struct buffer_head *bh; bool success = bio_endio_bh(bio, &bh); - if (!success) { + if (success) { + clear_buffer_write_io_error(bh); + } else { buffer_io_error(bh, ", lost sync page write"); mark_buffer_write_io_error(bh); } @@ -404,7 +406,9 @@ void bh_end_async_write(struct bio *bio) BUG_ON(!buffer_async_write(bh)); folio = bh->b_folio; - if (!success) { + if (success) { + clear_buffer_write_io_error(bh); + } else { buffer_io_error(bh, ", lost async page write"); mark_buffer_write_io_error(bh); } @@ -1085,7 +1089,6 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, enum rw_hint write_hint, struct writeback_control *wbc, bio_end_io_t end_bio) { - const enum req_op op = opf & REQ_OP_MASK; struct bio *bio; BUG_ON(!buffer_locked(bh)); @@ -1093,11 +1096,7 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, BUG_ON(buffer_delay(bh)); BUG_ON(buffer_unwritten(bh)); - /* - * Only clear out a write error when rewriting - */ - if (test_set_buffer_req(bh) && (op == REQ_OP_WRITE)) - clear_buffer_write_io_error(bh); + set_buffer_req(bh); if (buffer_meta(bh)) opf |= REQ_META; diff --git a/fs/gfs2/lops.c b/fs/gfs2/lops.c index 3df6e4b7e8b9eb..7440e5b72f8adb 100644 --- a/fs/gfs2/lops.c +++ b/fs/gfs2/lops.c @@ -179,6 +179,8 @@ static void gfs2_end_log_write_bh(struct gfs2_sbd *sdp, struct folio *folio, do { if (error) mark_buffer_write_io_error(bh); + else + clear_buffer_write_io_error(bh); unlock_buffer(bh); next = bh->b_this_page; size -= bh->b_size; diff --git a/fs/jbd2/commit.c b/fs/jbd2/commit.c index cd7ef783bd36a2..ebf6ba58ff4d0c 100644 --- a/fs/jbd2/commit.c +++ b/fs/jbd2/commit.c @@ -36,7 +36,9 @@ static void journal_end_buffer_io_sync(struct bio *bio) struct buffer_head *orig_bh = bh->b_private; BUFFER_TRACE(bh, ""); - if (!success) + if (success) + clear_buffer_write_io_error(bh); + else mark_buffer_write_io_error(bh); if (orig_bh) { clear_and_wake_up_bit(BH_Shadow, &orig_bh->b_state); From 16671fa0e523c3432db0819f5dd133b8eff38193 Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:06:57 +0200 Subject: [PATCH 243/857] fs/namei.c: use trailing_slashes() There are several places in fs/namei.c that can use the trailing_slashes() function to improve context. To allow this broader use its signature is changed to take a struct qstr instead of a struct nameidata. Reviewed-by: NeilBrown Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-2-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- fs/namei.c | 28 +++++++++++++++------------- 1 file changed, 15 insertions(+), 13 deletions(-) diff --git a/fs/namei.c b/fs/namei.c index 20a6534ea3efff..ab1302b38f460d 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -2781,9 +2781,16 @@ static const char *path_init(struct nameidata *nd, unsigned flags) return s; } +static inline bool trailing_slashes(const struct qstr *last) +{ + /* last->len is set by hash_name() to the length of the current + * component ->name, terminating with '/' or a NUL character. */ + return (bool)last->name[last->len]; +} + static inline const char *lookup_last(struct nameidata *nd) { - if (nd->last_type == LAST_NORM && nd->last.name[nd->last.len]) + if (nd->last_type == LAST_NORM && trailing_slashes(&nd->last)) nd->flags |= LOOKUP_FOLLOW | LOOKUP_DIRECTORY; return walk_component(nd, WALK_TRAILING); @@ -4695,17 +4702,12 @@ struct file *vfs_lookup_open(struct path *parent, struct qstr *last, } EXPORT_SYMBOL_FOR_MODULES(vfs_lookup_open, "nfsd"); -static inline bool trailing_slashes(struct nameidata *nd) -{ - return (bool)nd->last.name[nd->last.len]; -} - static struct dentry *lookup_fast_for_open(struct nameidata *nd, int open_flag) { struct dentry *dentry; if (open_flag & O_CREAT) { - if (trailing_slashes(nd)) + if (trailing_slashes(&nd->last)) return ERR_PTR(-EISDIR); /* Don't bother on an O_EXCL create */ @@ -4713,7 +4715,7 @@ static struct dentry *lookup_fast_for_open(struct nameidata *nd, int open_flag) return NULL; } - if (trailing_slashes(nd)) + if (trailing_slashes(&nd->last)) nd->flags |= LOOKUP_FOLLOW | LOOKUP_DIRECTORY; dentry = lookup_fast(nd); @@ -5087,7 +5089,7 @@ static struct dentry *filename_create(int dfd, struct filename *name, * Do the final lookup. Suppress 'create' if there is a trailing * '/', and a directory wasn't requested. */ - if (last.name[last.len] && !want_dir) + if (trailing_slashes(&last) && !want_dir) create_flags &= ~LOOKUP_CREATE; dentry = start_dirop(path->dentry, &last, reval_flag | create_flags); if (IS_ERR(dentry)) @@ -5703,7 +5705,7 @@ int filename_unlinkat(int dfd, struct filename *name) goto exit_drop_write; /* Why not before? Because we want correct error value */ - if (unlikely(last.name[last.len])) { + if (unlikely(trailing_slashes(&last))) { if (d_is_dir(dentry)) error = -EISDIR; else @@ -6305,16 +6307,16 @@ int filename_renameat2(int olddfd, struct filename *from, if (flags & RENAME_EXCHANGE) { if (!d_is_dir(rd.new_dentry)) { error = -ENOTDIR; - if (new_last.name[new_last.len]) + if (trailing_slashes(&new_last)) goto exit_unlock; } } /* unless the source is a directory trailing slashes give -ENOTDIR */ if (!d_is_dir(rd.old_dentry)) { error = -ENOTDIR; - if (old_last.name[old_last.len]) + if (trailing_slashes(&old_last)) goto exit_unlock; - if (!(flags & RENAME_EXCHANGE) && new_last.name[new_last.len]) + if (!(flags & RENAME_EXCHANGE) && trailing_slashes(&new_last)) goto exit_unlock; } From ae92a7b05339fbee6190213d6cdf1442a587919a Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:06:58 +0200 Subject: [PATCH 244/857] vfs: prepare vfs_creat|mkdir_no_perm for reuse in lookup_open() To implement O_CREAT|O_DIRECTORY we will have to repeat some of the logic that is now in vfs_mkdir() (e.g. do error checks in the same order). Separate this out in vfs_mkdir_no_perm(), which does all the non-permission related work of vfs_mkdir(). Permission checking for the lookup_open() path is timed differently because we may just be doing an open and no create. Similar considerations give rise to vfs_create_no_perm(). Reviewed-by: NeilBrown Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-3-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- fs/namei.c | 78 +++++++++++++++++++++++++++++++++++++----------------- 1 file changed, 54 insertions(+), 24 deletions(-) diff --git a/fs/namei.c b/fs/namei.c index ab1302b38f460d..229a5f7329f0e3 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -4166,6 +4166,24 @@ static inline umode_t vfs_prepare_mode(struct mnt_idmap *idmap, return mode; } +static inline +int vfs_create_no_perm(struct mnt_idmap *idmap, struct dentry *dentry, + umode_t mode, struct delegated_inode *di) +{ + struct inode *dir = d_inode(dentry->d_parent); + int error; + + error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, di); + if (error) + return error; + + error = dir->i_op->create(idmap, dir, dentry, mode); + if (!error) + fsnotify_create(dir, dentry); + + return error; +} + /** * vfs_create - create new file * @idmap: idmap of the mount the inode was found from @@ -4198,13 +4216,8 @@ int vfs_create(struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode, error = security_inode_create(dir, dentry, mode); if (error) return error; - error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, di); - if (error) - return error; - error = dir->i_op->create(idmap, dir, dentry, mode); - if (!error) - fsnotify_create(dir, dentry); - return error; + + return vfs_create_no_perm(idmap, dentry, mode, di); } EXPORT_SYMBOL(vfs_create); @@ -4418,6 +4431,7 @@ static struct dentry *atomic_open(const struct path *path, struct dentry *dentry dput(dentry); dentry = ERR_PTR(error); } + return dentry; } @@ -4544,6 +4558,7 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, dentry = res; } } + if (dentry->d_inode || !(op->open_flag & O_CREAT)) { /* * No need to create a file. If lookup returned a positive @@ -5358,6 +5373,34 @@ SYSCALL_DEFINE3(mknod, const char __user *, filename, umode_t, mode, unsigned, d return filename_mknodat(AT_FDCWD, name, mode, dev); } +/* Returns the dentry to use (not NULL) or -E on error */ +static inline +struct dentry *vfs_mkdir_no_perm(struct mnt_idmap *idmap, struct inode *dir, + struct dentry *dentry, umode_t mode, + struct delegated_inode *di) +{ + int error; + struct dentry *de; + unsigned max_links = dir->i_sb->s_max_links; + + if (max_links && dir->i_nlink >= max_links) + return ERR_PTR(-EMLINK); + + error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, di); + if (error) + return ERR_PTR(error); + + de = dir->i_op->mkdir(idmap, dir, dentry, mode); + if (IS_ERR(de)) + return de; + if (de) { + dput(dentry); + dentry = de; + } + fsnotify_mkdir(dir, dentry); + return dentry; +} + /** * vfs_mkdir - create directory returning correct dentry if possible * @idmap: idmap of the mount the inode was found from @@ -5385,7 +5428,6 @@ struct dentry *vfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, struct delegated_inode *delegated_inode) { int error; - unsigned max_links = dir->i_sb->s_max_links; struct dentry *de; error = may_create_dentry(idmap, dir, dentry); @@ -5401,24 +5443,12 @@ struct dentry *vfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, if (error) goto err; - error = -EMLINK; - if (max_links && dir->i_nlink >= max_links) - goto err; - - error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, delegated_inode); - if (error) + de = vfs_mkdir_no_perm(idmap, dir, dentry, mode, delegated_inode); + if (IS_ERR(de)) { + error = PTR_ERR(de); goto err; - - de = dir->i_op->mkdir(idmap, dir, dentry, mode); - error = PTR_ERR(de); - if (IS_ERR(de)) - goto err; - if (de) { - dput(dentry); - dentry = de; } - fsnotify_mkdir(dir, dentry); - return dentry; + return de; err: end_creating(dentry); From 7ec2ef575bbbb613d86b2207155942c6720b5368 Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:06:59 +0200 Subject: [PATCH 245/857] vfs: lookup_open(): move setting FMODE_CREATED down In preparation for using vfs_create_no_perm() in lookup_open() we need to move setting FMODE_CREATED on the file mode to either before or after that call, as currently it is in the middle. If try_break_deleg() fails it is currently not set, but vfs_create_no_perm() includes a try_break_deleg(). Going up the call chain of lookup_open() we see that it is only used in open_last_lookups() if no error is returned from lookup_open(), so we can safely move it to after the filesystem create() call. This also makes more sense when reading the code as you don't have to wonder what the implications are of setting FMODE_CREATED before the create() call. Reviewed-by: NeilBrown Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-4-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- fs/namei.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/namei.c b/fs/namei.c index 229a5f7329f0e3..2aa18efa4e0420 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -4580,7 +4580,6 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, if (error) goto out_dput; - file->f_mode |= FMODE_CREATED; if (!dir_inode->i_op->create) { error = -EACCES; goto out_dput; @@ -4589,6 +4588,8 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, error = dir_inode->i_op->create(idmap, dir_inode, dentry, mode); if (error) goto out_dput; + + file->f_mode |= FMODE_CREATED; out: if (!IS_ERR(dentry)) { if (file->f_mode & FMODE_CREATED) From c8aa81c37bffc577938de34492aedca555491323 Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:07:00 +0200 Subject: [PATCH 246/857] vfs: move ->create check in lookup_open() to before try_break_deleg() The i_op->create check in lookup_open() takes place after the try_break_deleg() call. This does not match the order when doing a regular file create via mknod(2). There the call order is: filename_mknodat() vfs_create() i_op->create check try_break_deleg() Move the i_op->create check to before try_break_deleg() in lookup_open(). Reviewed-by: NeilBrown Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-5-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- fs/namei.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/namei.c b/fs/namei.c index 2aa18efa4e0420..55ba23f95c5b0c 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -4576,15 +4576,15 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, goto out_dput; } - error = try_break_deleg(dir_inode, LEASE_BREAK_DIR_CREATE, &delegated_inode); - if (error) - goto out_dput; - if (!dir_inode->i_op->create) { error = -EACCES; goto out_dput; } + error = try_break_deleg(dir_inode, LEASE_BREAK_DIR_CREATE, &delegated_inode); + if (error) + goto out_dput; + error = dir_inode->i_op->create(idmap, dir_inode, dentry, mode); if (error) goto out_dput; From 652e9a8a732826f20935c87576ebc56436669417 Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:07:01 +0200 Subject: [PATCH 247/857] vfs: lookup_open(): use vfs_create_no_perm() We can replace the code in the no create_error/negative dentry found from lookup case in lookup_open() with the vfs_create_no_perm() helper. Reviewed-by: NeilBrown Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-6-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- fs/namei.c | 20 +++++++------------- 1 file changed, 7 insertions(+), 13 deletions(-) diff --git a/fs/namei.c b/fs/namei.c index 55ba23f95c5b0c..3afae6e878254f 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -4430,8 +4430,14 @@ static struct dentry *atomic_open(const struct path *path, struct dentry *dentry } dput(dentry); dentry = ERR_PTR(error); + } else { + if (file->f_mode & FMODE_CREATED) + fsnotify_create(dir_inode, dentry); + if (file->f_mode & FMODE_OPENED) + fsnotify_open(file); } + return dentry; } @@ -4581,22 +4587,10 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, goto out_dput; } - error = try_break_deleg(dir_inode, LEASE_BREAK_DIR_CREATE, &delegated_inode); - if (error) - goto out_dput; - - error = dir_inode->i_op->create(idmap, dir_inode, dentry, mode); - if (error) - goto out_dput; + error = vfs_create_no_perm(idmap, dentry, mode, &delegated_inode); file->f_mode |= FMODE_CREATED; out: - if (!IS_ERR(dentry)) { - if (file->f_mode & FMODE_CREATED) - fsnotify_create(dir_inode, dentry); - if (file->f_mode & FMODE_OPENED) - fsnotify_open(file); - } if ((open_flag & O_CREAT) || create_error) inode_unlock(dir_inode); else From 449c7265d60d44eece1f83eee2badf2ea61e767d Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:07:02 +0200 Subject: [PATCH 248/857] vfs: add O_CREAT|O_DIRECTORY to open*(2) Currently there is no way to race-freely create and open a directory. For regular files we have open(O_CREAT) for creating a new file inode, and returning a pinning fd to it. The lack of such functionality for directories means that when populating a directory tree there's always a race involved: the inodes first need to be created, and then opened to adjust their permissions/ownership/labels/timestamps/acls/xattrs/..., but in the time window between the creation and the opening they might be replaced by something else. Addressing this race without a proper API is only partially possible: the caller can immediately fstat() what was opened to verify that it has the expected inode type, owner and mode. But besides being easy to get wrong, this cannot establish who created the directory: a directory created by another process with identical credentials is indistinguishable from one the caller created itself, so the caller cannot tell whether the directory is its own to manage. Historically, the O_CREAT|O_DIRECTORY behaviour was to return ENOTDIR if a regular file exists at the open path; EISDIR if a directory exists at the path; and to create a regular file if no file exists at the path. This behaviour changed accidentally with 973d4b73fbaf ("do_last(): rejoin the common path even earlier in FMODE_{OPENED,CREATED} case") causing ENOTDIR to return in the last case while still creating the file. As this change was not detected for a long time, Brauner proposed to adopt the more consistent NetBSD behaviour, i.e. to return EINVAL on the O_CREAT|O_DIRECTORY combination. This change was applied in 43b450632676 ("open: return EINVAL for O_DIRECTORY | O_CREAT") in March, 2023. As the EINVAL behaviour has been in the kernel for about 3 years now, no rollback is expected as a result of userspace reliance on old behaviour, leaving us free to reassign the O_CREAT|O_DIRECTORY semantics. O_CREAT|O_DIRECTORY is made to reduce to a lookup on ->atomic_open() filesystems. These filesystems currenly cannot handle O_CREAT|O_DIRECTORY without protocol extensions and therefore are forced into a fallback mode by stripping the O_CREAT bit. This causes existing directories to be succesfully opened, while for targets that should have been created, -ENOENT is returned. The other option of simply returning -EINVAL leads to inconsistent behaviour: before ->atomic_open() is called in lookup_open(), the dcache is queried. So returning -EINVAL there would make O_CREAT|O_DIRECTORY dependent on the cache state of the dentry. There is no separate sysctl for directory creation implemented currently. Therefore, for the S_ISDIR case, disabling sysctl_protected_regular is not enough to allow creating a directory in a sticky folder, because that may surprise users not expecting that O_CREAT|O_DIRECTORY is possible on newer kernels. This feature idea (and some of its description) is taken from the UAPI group: https://github.com/uapi-group/kernel-features?tab=readme-ov-file#race-free-creation-and-opening-of-non-file-inodes Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-7-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- fs/namei.c | 89 ++++++++++++++++++++++++++++++++++--------- fs/open.c | 25 ++++++------ include/linux/fcntl.h | 6 +++ 3 files changed, 92 insertions(+), 28 deletions(-) diff --git a/fs/namei.c b/fs/namei.c index 3afae6e878254f..6f18a480665bc3 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -1382,13 +1382,13 @@ int may_linkat(struct mnt_idmap *idmap, const struct path *link) /** * may_create_in_sticky - Check whether an O_CREAT open in a sticky directory - * should be allowed, or not, on files that already - * exist. + * should be allowed, or not, on files/directories that + * already exist. * @idmap: idmap of the mount the inode was found from * @nd: nameidata pathwalk data * @inode: the inode of the file to open * - * Block an O_CREAT open of a FIFO (or a regular file) when: + * Block an O_CREAT open of a FIFO (or a regular file/directory) when: * - sysctl_protected_fifos (or sysctl_protected_regular) is enabled * - the file already exists * - we are in a sticky directory @@ -1416,6 +1416,14 @@ static int may_create_in_sticky(struct mnt_idmap *idmap, struct nameidata *nd, if (likely(!(dir_mode & S_ISVTX))) return 0; + /* + * There is no separate sysctl for directory creation in sticky + * folders. Therefore, for the S_ISDIR case, disabling + * sysctl_protected_regular is not enough to allow creating a + * directory in a sticky folder, because that may surprise users + * not expecting that O_CREAT|O_DIRECTORY is possible on newer + * kernels. + */ if (S_ISREG(inode->i_mode) && !sysctl_protected_regular) return 0; @@ -1447,6 +1455,12 @@ static int may_create_in_sticky(struct mnt_idmap *idmap, struct nameidata *nd, "sticky_create_regular"); return -EACCES; } + + if (S_ISDIR(inode->i_mode)) { + audit_log_path_denied(AUDIT_ANOM_CREAT, + "sticky_create_dir"); + return -EACCES; + } } return 0; @@ -4334,21 +4348,41 @@ static inline int open_to_namei_flags(int flag) static int may_o_create(struct mnt_idmap *idmap, const struct path *dir, struct dentry *dentry, - umode_t mode) + int open_flag, umode_t mode) { - int error = security_path_mknod(dir, dentry, mode, 0); + struct inode *dir_inode = dir->dentry->d_inode; + bool create_dir = O_IS_MKDIR(open_flag); + int error; + + if (create_dir) + error = security_path_mkdir(dir, dentry, mode); + else + error = security_path_mknod(dir, dentry, mode, 0); if (error) return error; if (!fsuidgid_has_mapping(dir->dentry->d_sb, idmap)) return -EOVERFLOW; - error = inode_permission(idmap, dir->dentry->d_inode, - MAY_WRITE | MAY_EXEC); + error = inode_permission(idmap, dir_inode, MAY_WRITE | MAY_EXEC); if (error) return error; - return security_inode_create(dir->dentry->d_inode, dentry, mode); + if (create_dir) + error = security_inode_mkdir(dir_inode, dentry, mode); + else + error = security_inode_create(dir_inode, dentry, mode); + + return error; +} + +static inline umode_t o_create_mode(struct mnt_idmap *idmap, + const struct inode *dir, int open_flag, umode_t mode) +{ + if (O_IS_MKDIR(open_flag)) + return vfs_prepare_mode(idmap, dir, mode, S_IRWXUGO | S_ISVTX, S_IFDIR); + else + return vfs_prepare_mode(idmap, dir, mode, S_IALLUGO, S_IFREG); } /** @@ -4384,8 +4418,9 @@ static struct dentry *atomic_open(const struct path *path, struct dentry *dentry file->__f_path.dentry = DENTRY_NOT_SET; file->__f_path.mnt = path->mnt; + error = dir_inode->i_op->atomic_open(dir_inode, dentry, file, - open_to_namei_flags(open_flag), mode); + open_to_namei_flags(open_flag), mode); d_lookup_done(dentry); if (!error) { @@ -4441,6 +4476,10 @@ static struct dentry *atomic_open(const struct path *path, struct dentry *dentry return dentry; } +static inline +struct dentry *vfs_mkdir_no_perm(struct mnt_idmap *, struct inode *, + struct dentry *, umode_t, + struct delegated_inode *); /* * Look up and maybe create and open the last component. * @@ -4462,6 +4501,7 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, struct mnt_idmap *idmap; struct dentry *dir = nd->path.dentry; struct inode *dir_inode = dir->d_inode; + bool create_dir = O_IS_MKDIR(op->mode); int open_flag; struct dentry *dentry; int error, create_error; @@ -4482,7 +4522,7 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, */ } if (open_flag & O_CREAT) - inode_lock(dir_inode); + inode_lock_nested(dir_inode, I_MUTEX_PARENT); else inode_lock_shared(dir_inode); @@ -4491,6 +4531,9 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, goto out; } + if (create_dir && dir_inode->i_op->atomic_open) + open_flag &= ~O_CREAT; + file->f_mode &= ~FMODE_CREATED; dentry = d_lookup(dir, &nd->last); for (;;) { @@ -4534,10 +4577,10 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, if (open_flag & O_CREAT) { if (open_flag & O_EXCL) open_flag &= ~O_TRUNC; - mode = vfs_prepare_mode(idmap, dir_inode, mode, mode, mode); + mode = o_create_mode(idmap, dir_inode, open_flag, mode); if (likely(got_write)) create_error = may_o_create(idmap, &nd->path, - dentry, mode); + dentry, open_flag, mode); else create_error = -EROFS; } @@ -4582,12 +4625,23 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, goto out_dput; } - if (!dir_inode->i_op->create) { + if ((create_dir && !dir_inode->i_op->mkdir) + || (!create_dir && !dir_inode->i_op->create)) { error = -EACCES; goto out_dput; } - error = vfs_create_no_perm(idmap, dentry, mode, &delegated_inode); + if (create_dir) { + struct dentry *res = vfs_mkdir_no_perm(idmap, dir_inode, dentry, + mode, &delegated_inode); + error = PTR_ERR_OR_ZERO(res); + if (!error) + dentry = res; + } else { + error = vfs_create_no_perm(idmap, dentry, mode, &delegated_inode); + } + if (error) + goto out_dput; file->f_mode |= FMODE_CREATED; out: @@ -4717,7 +4771,7 @@ static struct dentry *lookup_fast_for_open(struct nameidata *nd, int open_flag) struct dentry *dentry; if (open_flag & O_CREAT) { - if (trailing_slashes(&nd->last)) + if (trailing_slashes(&nd->last) && !(open_flag & O_DIRECTORY)) return ERR_PTR(-EISDIR); /* Don't bother on an O_EXCL create */ @@ -4818,8 +4872,9 @@ static int do_open(struct nameidata *nd, if (open_flag & O_CREAT) { if ((open_flag & O_EXCL) && !(file->f_mode & FMODE_CREATED)) return -EEXIST; - if (d_is_dir(nd->path.dentry)) + if (!(open_flag & O_DIRECTORY) && d_is_dir(nd->path.dentry)) return -EISDIR; + error = may_create_in_sticky(idmap, nd, d_backing_inode(nd->path.dentry)); if (unlikely(error)) @@ -5194,7 +5249,7 @@ struct file *dentry_create(struct path *path, int flags, umode_t mode, path->dentry = dir; mode = vfs_prepare_mode(idmap, dir_inode, mode, S_IALLUGO, S_IFREG); - create_error = may_o_create(idmap, path, dentry, mode); + create_error = may_o_create(idmap, path, dentry, flags, mode); if (create_error) flags &= ~O_CREAT; diff --git a/fs/open.c b/fs/open.c index 6b1c14e684a93b..189af02a242505 100644 --- a/fs/open.c +++ b/fs/open.c @@ -1239,29 +1239,30 @@ inline int build_open_flags(const struct open_how *how, struct open_flags *op) if (WILL_CREATE(flags)) { if (how->mode & ~S_IALLUGO) return -EINVAL; - op->mode = how->mode | S_IFREG; + if (O_IS_MKDIR(flags)) + op->mode = how->mode | S_IFDIR; + else + op->mode = how->mode | S_IFREG; } else { if (how->mode != 0) return -EINVAL; op->mode = 0; } - /* - * Block bugs where O_DIRECTORY | O_CREAT created regular files. - * Note, that blocking O_DIRECTORY | O_CREAT here also protects - * O_TMPFILE below which requires O_DIRECTORY being raised. - */ - if ((flags & (O_DIRECTORY | O_CREAT)) == (O_DIRECTORY | O_CREAT)) - return -EINVAL; - /* Now handle the creative implementation of O_TMPFILE. */ if (flags & __O_TMPFILE) { /* * In order to ensure programs get explicit errors when trying * to use O_TMPFILE on old kernels we enforce that O_DIRECTORY - * is raised alongside __O_TMPFILE. + * is raised alongside __O_TMPFILE, but without O_CREAT. The + * reason for disallowing O_CREAT|O_TMPFILE is that + * O_DIRECTORY|O_CREAT used to work and created a regular file + * if nothing existed at the open path. Hence, allowing the + * combination would have caused O_CREAT|O_TMPFILE to create a + * regular (non-temporary) file on old kernels, while the caller + * would believe they created an actual O_TMPFILE. */ - if (!(flags & O_DIRECTORY)) + if (!(flags & O_DIRECTORY) || (flags & O_CREAT)) return -EINVAL; if (!(acc_mode & MAY_WRITE)) return -EINVAL; @@ -1319,6 +1320,8 @@ inline int build_open_flags(const struct open_how *how, struct open_flags *op) op->intent = flags & O_PATH ? 0 : LOOKUP_OPEN; if (flags & O_CREAT) { + if ((flags & O_DIRECTORY) && (acc_mode & MAY_WRITE)) + return -EISDIR; op->intent |= LOOKUP_CREATE; if (flags & O_EXCL) { op->intent |= LOOKUP_EXCL; diff --git a/include/linux/fcntl.h b/include/linux/fcntl.h index 6ad6b9e7a226af..204e16bbe2634c 100644 --- a/include/linux/fcntl.h +++ b/include/linux/fcntl.h @@ -30,6 +30,12 @@ */ #define __O_REGULAR (1 << 30) +#define O_MKDIR_MASK (O_CREAT | O_DIRECTORY) +static inline bool O_IS_MKDIR(unsigned int flags) +{ + return (flags & O_MKDIR_MASK) == O_MKDIR_MASK; +} + /* List of all valid flags for the how->resolve argument: */ #define VALID_RESOLVE_FLAGS \ (RESOLVE_NO_XDEV | RESOLVE_NO_MAGICLINKS | RESOLVE_NO_SYMLINKS | \ From 65c4144bb147aa3f61fb0ad6098514e55b1eace8 Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:07:03 +0200 Subject: [PATCH 249/857] vfs: move O_IS_MKDIR check from lookup_open() into individual filesystems Individual filesystems that implement ->atomic_open() need to get the chance to implement O_CREAT|O_DIRECTORY or not, rather than decide this at the VFS level in lookup_open(). Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-8-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- fs/9p/vfs_inode.c | 3 +++ fs/9p/vfs_inode_dotl.c | 3 +++ fs/ceph/file.c | 3 +++ fs/fuse/dir.c | 3 +++ fs/gfs2/inode.c | 3 +++ fs/namei.c | 3 --- fs/nfs/dir.c | 6 ++++++ fs/smb/client/dir.c | 3 +++ fs/vboxsf/dir.c | 3 +++ 9 files changed, 27 insertions(+), 3 deletions(-) diff --git a/fs/9p/vfs_inode.c b/fs/9p/vfs_inode.c index 3829554ca36929..b1e0823c87b576 100644 --- a/fs/9p/vfs_inode.c +++ b/fs/9p/vfs_inode.c @@ -776,6 +776,9 @@ v9fs_vfs_atomic_open(struct inode *dir, struct dentry *dentry, struct inode *inode; int p9_omode; + if (O_IS_MKDIR(flags)) + flags &= ~O_CREAT; + if (d_in_lookup(dentry)) { struct dentry *res = v9fs_vfs_lookup(dir, dentry, 0); if (res || d_really_is_positive(dentry)) diff --git a/fs/9p/vfs_inode_dotl.c b/fs/9p/vfs_inode_dotl.c index 116b29e95f21ee..64e0aba08da16b 100644 --- a/fs/9p/vfs_inode_dotl.c +++ b/fs/9p/vfs_inode_dotl.c @@ -238,6 +238,9 @@ v9fs_vfs_atomic_open_dotl(struct inode *dir, struct dentry *dentry, struct v9fs_session_info *v9ses; struct posix_acl *pacl = NULL, *dacl = NULL; + if (O_IS_MKDIR(flags)) + flags &= ~O_CREAT; + if (d_in_lookup(dentry)) { struct dentry *res = v9fs_vfs_lookup(dir, dentry, 0); if (res || d_really_is_positive(dentry)) diff --git a/fs/ceph/file.c b/fs/ceph/file.c index bd3e3f5c269e89..78a0b2540f103c 100644 --- a/fs/ceph/file.c +++ b/fs/ceph/file.c @@ -812,6 +812,9 @@ int ceph_atomic_open(struct inode *dir, struct dentry *dentry, dir, ceph_vinop(dir), dentry, dentry, d_unhashed(dentry) ? "unhashed" : "hashed", flags, mode); + if (O_IS_MKDIR(flags)) + flags &= ~O_CREAT; + if (dentry->d_name.len > NAME_MAX) return -ENAMETOOLONG; diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index e49b4e874b15f8..17217d97b9e768 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -944,6 +944,9 @@ static int fuse_atomic_open(struct inode *dir, struct dentry *entry, struct mnt_idmap *idmap = file_mnt_idmap(file); struct fuse_conn *fc = get_fuse_conn(dir); + if (O_IS_MKDIR(flags)) + flags &= ~O_CREAT; + if (fuse_is_bad(dir)) return -EIO; diff --git a/fs/gfs2/inode.c b/fs/gfs2/inode.c index f361876c558335..3ee1360f1bc270 100644 --- a/fs/gfs2/inode.c +++ b/fs/gfs2/inode.c @@ -1386,6 +1386,9 @@ static int gfs2_atomic_open(struct inode *dir, struct dentry *dentry, { bool excl = !!(flags & O_EXCL); + if (O_IS_MKDIR(flags)) + flags &= ~O_CREAT; + if (d_in_lookup(dentry)) { struct dentry *d = __gfs2_lookup(dir, dentry, file); if (file->f_mode & FMODE_OPENED) { diff --git a/fs/namei.c b/fs/namei.c index 6f18a480665bc3..9da91f081c11e6 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -4531,9 +4531,6 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, goto out; } - if (create_dir && dir_inode->i_op->atomic_open) - open_flag &= ~O_CREAT; - file->f_mode &= ~FMODE_CREATED; dentry = d_lookup(dir, &nd->last); for (;;) { diff --git a/fs/nfs/dir.c b/fs/nfs/dir.c index 49394123bd0961..f01f9013202d9b 100644 --- a/fs/nfs/dir.c +++ b/fs/nfs/dir.c @@ -2121,6 +2121,9 @@ int nfs_atomic_open(struct inode *dir, struct dentry *dentry, dfprintk(VFS, "NFS: atomic_open(%s/%llu), %pd\n", dir->i_sb->s_id, dir->i_ino, dentry); + if (O_IS_MKDIR(open_flags)) + open_flags &= ~O_CREAT; + err = nfs_check_flags(open_flags); if (err) return err; @@ -2317,6 +2320,9 @@ int nfs_atomic_open_v23(struct inode *dir, struct dentry *dentry, */ int error = 0; + if (O_IS_MKDIR(open_flags)) + open_flags &= ~O_CREAT; + if (dentry->d_name.len > NFS_SERVER(dir)->namelen) return -ENAMETOOLONG; diff --git a/fs/smb/client/dir.c b/fs/smb/client/dir.c index 6fa6d48fdfd30f..f44defb6435479 100644 --- a/fs/smb/client/dir.c +++ b/fs/smb/client/dir.c @@ -534,6 +534,9 @@ int cifs_atomic_open(struct inode *dir, struct dentry *direntry, if (unlikely(cifs_forced_shutdown(cifs_sb))) return smb_EIO(smb_eio_trace_forced_shutdown); + if (O_IS_MKDIR(oflags)) + oflags &= ~O_CREAT; + /* * Posix open is only called (at lookup time) for file create now. For * opens (rather than creates), because we do not know if it is a file diff --git a/fs/vboxsf/dir.c b/fs/vboxsf/dir.c index 0b9eab157432df..6e306ddd722bfa 100644 --- a/fs/vboxsf/dir.c +++ b/fs/vboxsf/dir.c @@ -318,6 +318,9 @@ static int vboxsf_dir_atomic_open(struct inode *parent, struct dentry *dentry, u64 handle; int err; + if (O_IS_MKDIR(flags)) + flags &= ~O_CREAT; + if (d_in_lookup(dentry)) { struct dentry *res = vboxsf_dir_lookup(parent, dentry, 0); if (res || d_really_is_positive(dentry)) From acc1c34e4526448320e4e1f448e4eb3bf4b4c2d5 Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:07:04 +0200 Subject: [PATCH 250/857] vfs: refuse O_CREAT for directories through a dangling symlink open(O_CREAT) without O_EXCL follows a trailing symlink and, when the symlink target does not exist, creates it. Refuse to create through a dangling symlink for directories. In lookup_open() a negative target reached with nd->depth > 0 was arrived at by following a trailing symlink; since the dentry is negative the symlink is dangling. Set create_error to -EEXIST in that case (matching the errno returned by mkdir(2).) Reusing the existing create_error path strips O_CREAT for both the generic and ->atomic_open create paths and only reports the error when the target is actually negative. Thus opening an existing target through a symlink, interior symlinks, and O_EXCL (which never follows the trailing link) are all unaffected. Suggested-by: Christian Brauner (Amutable) Reviewed-by: NeilBrown Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-9-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- fs/namei.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/fs/namei.c b/fs/namei.c index 9da91f081c11e6..e1a8f18385a529 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -4580,6 +4580,11 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, dentry, open_flag, mode); else create_error = -EROFS; + /* Refuse to create a directory through a dangling (trailing) + * symlink. For regular files this has been allowed historically + * on O_CREAT without O_EXCL. */ + if (unlikely(nd->depth) && create_dir && !create_error) + create_error = -EEXIST; } if (create_error) open_flag &= ~O_CREAT; From 12160a2eff4665ba123a71d59337a739898962c3 Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:07:05 +0200 Subject: [PATCH 251/857] vfs: short-circuit MAY_WRITE access for O_DIRECTORY opens Requesting write access on a directory can never succeed. Rather than performing a path-walk to determine whether the target is actually a directory (-EISDIR) or not (-ENOTDIR), or does not exist (-ENOENT), etc., we short-circuit to -ENOTDIR. Currently O_WRONLY for directories is only blocked in may_open(), which happens after we have the inode for the target, so after any create via O_CREAT|O_DIRECTORY. The advantage of short-circuiting is that we don't have to add even more logic to lookup_open() to differentiate -EISDIR/-ENOTDIR. Also, for filesystems that define atomic_open(), handling this cannot even be done at the VFS level, as we can't know ahead of calling ->atomic_open() what the result of the lookup is. Suggested-by: Christian Brauner (Amutable) Reviewed-by: NeilBrown Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-10-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- fs/open.c | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/fs/open.c b/fs/open.c index 189af02a242505..6cb5e2ad781f30 100644 --- a/fs/open.c +++ b/fs/open.c @@ -1319,9 +1319,16 @@ inline int build_open_flags(const struct open_how *how, struct open_flags *op) op->intent = flags & O_PATH ? 0 : LOOKUP_OPEN; + /* + * Requesting write access on a directory can never succeed. Rather + * than performing a path-walk to determine whether the target is + * actually a directory (-EISDIR) or not (-ENOTDIR), we short-circuit + * to -ENOTDIR. + */ + if ((flags & O_DIRECTORY) && !(flags & __O_TMPFILE) && (acc_mode & MAY_WRITE)) + return -ENOTDIR; + if (flags & O_CREAT) { - if ((flags & O_DIRECTORY) && (acc_mode & MAY_WRITE)) - return -EISDIR; op->intent |= LOOKUP_CREATE; if (flags & O_EXCL) { op->intent |= LOOKUP_EXCL; From f54e04941355e0c97b0ce825964afd2a89a54f6a Mon Sep 17 00:00:00 2001 From: Jori Koolstra Date: Sun, 23 Aug 2026 18:07:06 +0200 Subject: [PATCH 252/857] selftest: add tests for open*(O_CREAT|O_DIRECTORY) Add some tests for the new valid O_CREAT|O_DIRECTORY flag combination for open*(2) to test compliance and to showcase its behaviour. Signed-off-by: Jori Koolstra Link: https://patch.msgid.link/20260823160706.358293-11-jkoolstra@xs4all.nl Signed-off-by: Christian Brauner (Amutable) --- .../testing/selftests/filesystems/.gitignore | 1 + tools/testing/selftests/filesystems/Makefile | 2 +- .../filesystems/open_o_creat_o_dir.c | 201 ++++++++++++++++++ .../testing/selftests/filesystems/wrappers.h | 11 + 4 files changed, 214 insertions(+), 1 deletion(-) create mode 100644 tools/testing/selftests/filesystems/open_o_creat_o_dir.c diff --git a/tools/testing/selftests/filesystems/.gitignore b/tools/testing/selftests/filesystems/.gitignore index 9eb185fb2f9ddb..01c588d4c84f61 100644 --- a/tools/testing/selftests/filesystems/.gitignore +++ b/tools/testing/selftests/filesystems/.gitignore @@ -1,4 +1,5 @@ # SPDX-License-Identifier: GPL-2.0-only +open_o_creat_o_dir dnotify_test devpts_pts fclog diff --git a/tools/testing/selftests/filesystems/Makefile b/tools/testing/selftests/filesystems/Makefile index 03be337c1f351a..0959bd26875ac3 100644 --- a/tools/testing/selftests/filesystems/Makefile +++ b/tools/testing/selftests/filesystems/Makefile @@ -1,7 +1,7 @@ # SPDX-License-Identifier: GPL-2.0 CFLAGS += $(KHDR_INCLUDES) -TEST_GEN_PROGS := devpts_pts file_stressor anon_inode_test kernfs_test fclog ustat_test +TEST_GEN_PROGS := open_o_creat_o_dir devpts_pts file_stressor anon_inode_test kernfs_test fclog ustat_test TEST_GEN_PROGS += idmapped_tmpfile TEST_GEN_PROGS_EXTENDED := dnotify_test diff --git a/tools/testing/selftests/filesystems/open_o_creat_o_dir.c b/tools/testing/selftests/filesystems/open_o_creat_o_dir.c new file mode 100644 index 00000000000000..be0ab34267e1d2 --- /dev/null +++ b/tools/testing/selftests/filesystems/open_o_creat_o_dir.c @@ -0,0 +1,201 @@ +// SPDX-License-Identifier: GPL-2.0 +#include +#include +#include +#include + +#include "kselftest_harness.h" +#include "wrappers.h" + +#define openat_o_mkdir_checked_flags(dfd, pathname, flags) ({ \ + struct stat __st; \ + int __fd = openat_o_mkdir(dfd, pathname, flags, S_IRWXU); \ + ASSERT_GE(__fd, 0); \ + ASSERT_EQ(fstat(__fd, &__st), 0); \ + EXPECT_TRUE(S_ISDIR(__st.st_mode)); \ + __fd; \ +}) + +#define openat_o_mkdir_checked(dfd, pathname) \ + openat_o_mkdir_checked_flags(dfd, pathname, O_RDONLY) + +FIXTURE(open_o_creat_o_dir) { + char dirpath[PATH_MAX]; + int dfd; +}; + +FIXTURE_SETUP(open_o_creat_o_dir) +{ + strcpy(self->dirpath, "/tmp/open_o_creat_o_dir_test.XXXXXX"); + ASSERT_NE(mkdtemp(self->dirpath), NULL); + self->dfd = open(self->dirpath, O_DIRECTORY); + ASSERT_GE(self->dfd, 0); +} + +FIXTURE_TEARDOWN(open_o_creat_o_dir) +{ + close(self->dfd); + rmdir(self->dirpath); +} + +/* Does open_o_creat_o_dir return a fd at all? */ +TEST_F(open_o_creat_o_dir, returns_fd) +{ + int fd = openat_o_mkdir_checked(self->dfd, "newdir"); + EXPECT_EQ(close(fd), 0); + EXPECT_EQ(unlinkat(self->dfd, "newdir", AT_REMOVEDIR), 0); +} + +/* The fd must refer to the directory that was just created. */ +TEST_F(open_o_creat_o_dir, fd_is_created_dir) +{ + int fd; + struct stat st_via_fd, st_via_path; + char path[PATH_MAX]; + + fd = openat_o_mkdir_checked(self->dfd, "checkdir"); + + ASSERT_EQ(fstat(fd, &st_via_fd), 0); + + snprintf(path, sizeof(path), "%s/checkdir", self->dirpath); + ASSERT_EQ(stat(path, &st_via_path), 0); + + EXPECT_EQ(st_via_fd.st_ino, st_via_path.st_ino); + EXPECT_EQ(st_via_fd.st_dev, st_via_path.st_dev); + + EXPECT_EQ(close(fd), 0); + EXPECT_EQ(rmdir(path), 0); +} + +/* Missing parent component must fail with ENOENT. */ +TEST_F(open_o_creat_o_dir, enoent_missing_parent) +{ + EXPECT_EQ(openat_o_mkdir(self->dfd, "nonexistent/child", O_RDONLY, S_IRWXU), -1); + EXPECT_EQ(errno, ENOENT); +} + +/* An invalid dfd must fail with EBADF. */ +TEST_F(open_o_creat_o_dir, ebadf) +{ + EXPECT_EQ(openat_o_mkdir(FD_INVALID, "badfdir", O_RDONLY, S_IRWXU), -1); + EXPECT_EQ(errno, EBADF); +} + +/* A dfd that points to a file (not a directory) must fail with ENOTDIR. */ +TEST_F(open_o_creat_o_dir, enotdir_dfd) +{ + int file_fd; + + file_fd = openat(self->dfd, "file", + O_CREAT | O_RDONLY, S_IRWXU); + ASSERT_GE(file_fd, 0); + + EXPECT_EQ(openat_o_mkdir(file_fd, "subdir", O_RDONLY, S_IRWXU), -1); + EXPECT_EQ(errno, ENOTDIR); + + EXPECT_EQ(close(file_fd), 0); + EXPECT_EQ(unlinkat(self->dfd, "file", 0), 0); +} + +/* + * O_EXCL together with O_CREAT|O_DIRECTORY should succeed if the target + * directory does not yet exist. After directory creation, repeating this + * call must fail with EEXIST. + */ +TEST_F(open_o_creat_o_dir, o_excl_eexist) +{ + int excldir_fd; + + excldir_fd = openat_o_mkdir_checked_flags(self->dfd, "excldir", O_EXCL); + + EXPECT_EQ(openat_o_mkdir(excldir_fd, ".", O_EXCL, S_IRWXU), -1); + EXPECT_EQ(errno, EEXIST); + + EXPECT_EQ(close(excldir_fd), 0); + EXPECT_EQ(unlinkat(self->dfd, "excldir", AT_REMOVEDIR), 0); +} + +/* + * O_CREAT|O_DIRECTORY on a path that already exists as a regular file + * must fail with ENOTDIR. + */ +TEST_F(open_o_creat_o_dir, existing_file_enotdir) +{ + int file_fd; + + file_fd = openat(self->dfd, "regfile", + O_CREAT | O_RDONLY, S_IRWXU); + ASSERT_GE(file_fd, 0); + EXPECT_EQ(close(file_fd), 0); + + EXPECT_EQ(openat_o_mkdir(self->dfd, "regfile", O_RDONLY, S_IRWXU), -1); + EXPECT_EQ(errno, ENOTDIR); + + EXPECT_EQ(unlinkat(self->dfd, "regfile", 0), 0); +} + +/* + * O_CREAT|O_DIRECTORY combined with a writable access mode must be + * rejected: a directory cannot be opened for writing. + */ +TEST_F(open_o_creat_o_dir, rejects_writable_acc_mode) +{ + EXPECT_EQ(openat_o_mkdir(self->dfd, "rdwrdir", O_RDWR, S_IRWXU), -1); + EXPECT_EQ(errno, ENOTDIR); + /* Clean up if the kernel created the directory anyway. */ + unlinkat(self->dfd, "rdwrdir", AT_REMOVEDIR); +} + +/* + * openat(O_CREAT|O_DIRECTORY) with a trailing slash should work. + */ +TEST_F(open_o_creat_o_dir, trailing_slash) +{ + int fd = openat_o_mkdir_checked(self->dfd, "newdir/"); + EXPECT_EQ(close(fd), 0); + EXPECT_EQ(unlinkat(self->dfd, "newdir", AT_REMOVEDIR), 0); +} + +/* + * openat(O_CREAT) with a trailing slash but without O_DIRECTORY + * must fail with EISDIR and must not create anything at the path. + */ +TEST_F(open_o_creat_o_dir, trailing_slash_no_o_dir) +{ + int fd; + struct stat st; + + fd = openat(self->dfd, "trailing/", O_CREAT | O_RDONLY, S_IRWXU); + EXPECT_EQ(fd, -1); + EXPECT_EQ(errno, EISDIR); + + EXPECT_EQ(fstatat(self->dfd, "trailing", &st, 0), -1); + EXPECT_EQ(errno, ENOENT); + + /* Best-effort cleanup in case the kernel left a file behind. */ + if (fd >= 0) + close(fd); + unlinkat(self->dfd, "trailing", 0); +} + +/* + * The returned fd must be usable as a dfd for further *at() calls. + */ +TEST_F(open_o_creat_o_dir, fd_usable_as_dfd) +{ + int parent_fd, child_fd; + char path[PATH_MAX]; + + parent_fd = openat_o_mkdir_checked(self->dfd, "parent"); + child_fd = openat_o_mkdir_checked(parent_fd, "child"); + + EXPECT_EQ(close(child_fd), 0); + EXPECT_EQ(close(parent_fd), 0); + + snprintf(path, sizeof(path), "%s/parent/child", self->dirpath); + EXPECT_EQ(rmdir(path), 0); + snprintf(path, sizeof(path), "%s/parent", self->dirpath); + EXPECT_EQ(rmdir(path), 0); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/wrappers.h b/tools/testing/selftests/filesystems/wrappers.h index 420ae4f908cf2d..abe5b85cebdcdc 100644 --- a/tools/testing/selftests/filesystems/wrappers.h +++ b/tools/testing/selftests/filesystems/wrappers.h @@ -13,6 +13,10 @@ #define STATX_MNT_ID_UNIQUE 0x00004000U /* Want/got extended stx_mount_id */ #endif +#ifndef FD_INVALID +#define FD_INVALID -10009 +#endif + static inline int sys_fsopen(const char *fsname, unsigned int flags) { return syscall(__NR_fsopen, fsname, flags); @@ -105,4 +109,11 @@ static inline int sys_open_tree(int dfd, const char *filename, unsigned int flag return syscall(__NR_open_tree, dfd, filename, flags); } +static inline int openat_o_mkdir(int dfd, const char *pathname, + unsigned int flags, mode_t mode) +{ + return syscall(__NR_openat, dfd, pathname, + flags | O_DIRECTORY | O_CREAT, mode); +} + #endif From 0d1ea1532955fa881424cb01c4599d14b90a22f6 Mon Sep 17 00:00:00 2001 From: Mateusz Guzik Date: Mon, 3 Aug 2026 14:51:38 +0200 Subject: [PATCH 253/857] fs: avoid spurious dentry ref/unref cycle on open Opening a file grabs a reference on the terminal dentry in __legitimize_path(), then another one in do_dentry_open() and finally drops the initial reference in terminate_walk(). That's 2 modifications which don't need to be there -- do_dentry_open() can consume the already held reference instead. When benchmarking on a 20-core vm using will-it-scale to open the same file read-only, the results are (ops/s): before: 4043375 after: 5629378 (+39%) Signed-off-by: Mateusz Guzik Link: https://patch.msgid.link/20260803125138.1937674-1-mjguzik@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/internal.h | 1 + fs/namei.c | 15 ++++++++++++--- fs/open.c | 27 ++++++++++++++++++++++++++- 3 files changed, 39 insertions(+), 4 deletions(-) diff --git a/fs/internal.h b/fs/internal.h index c658c8a5ebd569..9632239036accf 100644 --- a/fs/internal.h +++ b/fs/internal.h @@ -205,6 +205,7 @@ int do_fchownat(int dfd, const char __user *filename, uid_t user, gid_t group, int flag); int chown_common(const struct path *path, uid_t user, gid_t group); extern int vfs_open(const struct path *, struct file *); +int vfs_open_consume(struct path *, struct file *); /* * inode.c diff --git a/fs/namei.c b/fs/namei.c index e1a8f18385a529..9da900f6c11343 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -4857,6 +4857,7 @@ static const char *open_last_lookups(struct nameidata *nd, static int do_open(struct nameidata *nd, struct file *file, const struct open_flags *op) { + struct vfsmount *mnt; struct mnt_idmap *idmap; int open_flag = op->open_flag; bool do_truncate; @@ -4899,11 +4900,17 @@ static int do_open(struct nameidata *nd, error = mnt_want_write(nd->path.mnt); if (error) return error; + /* + * A dedicated reference is needed because after the call to + * vfs_open_consume() we no longer own the reference in nd->path.mnt + * while we need to undo write acess below. + */ + mnt = mntget(nd->path.mnt); do_truncate = true; } error = may_open(idmap, &nd->path, acc_mode, open_flag); if (!error && !(file->f_mode & FMODE_OPENED)) - error = vfs_open(&nd->path, file); + error = vfs_open_consume(&nd->path, file); if (!error) error = security_file_post_open(file, op->acc_mode); if (!error && do_truncate) @@ -4912,8 +4919,10 @@ static int do_open(struct nameidata *nd, WARN_ON(1); error = -EINVAL; } - if (do_truncate) - mnt_drop_write(nd->path.mnt); + if (do_truncate) { + mnt_drop_write(mnt); + mntput(mnt); + } return error; } diff --git a/fs/open.c b/fs/open.c index 6cb5e2ad781f30..a84b55301719ca 100644 --- a/fs/open.c +++ b/fs/open.c @@ -931,6 +931,11 @@ static inline int file_get_write_access(struct file *f) return error; } +/* + * Populate struct file + * + * NOTE: it assumes f_path is populated and consumes the caller's reference. + */ static int do_dentry_open(struct file *f, int (*open)(struct inode *, struct file *)) { @@ -938,7 +943,6 @@ static int do_dentry_open(struct file *f, struct inode *inode = f->f_path.dentry->d_inode; int error; - path_get(&f->f_path); f->f_inode = inode; f->f_mapping = inode->i_mapping; f->f_wb_err = filemap_sample_wb_err(f->f_mapping); @@ -1055,6 +1059,7 @@ int finish_open(struct file *file, struct dentry *dentry, BUG_ON(file->f_mode & FMODE_OPENED); /* once it's opened, it's opened */ file->__f_path.dentry = dentry; + path_get(&file->f_path); return do_dentry_open(file, open); } EXPORT_SYMBOL(finish_open); @@ -1098,6 +1103,7 @@ int vfs_open(const struct path *path, struct file *file) int ret; file->__f_path = *path; + path_get(&file->f_path); ret = do_dentry_open(file, NULL); if (!ret) { /* @@ -1110,6 +1116,25 @@ int vfs_open(const struct path *path, struct file *file) return ret; } +/** + * vfs_open_consume - open the file at the given path and consume the reference + * @path: path to open + * @file: newly allocated file with f_flag initialized + */ +int vfs_open_consume(struct path *path, struct file *file) +{ + int ret; + + file->__f_path = *path; + path->mnt = NULL; + path->dentry = NULL; + ret = do_dentry_open(file, NULL); + if (!ret) { + fsnotify_open(file); + } + return ret; +} + struct file *dentry_open(const struct path *path, int flags, const struct cred *cred) { From bfe43e677bea03049d97c72334adca2af3788501 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Wed, 26 Aug 2026 18:06:40 +0200 Subject: [PATCH 254/857] powerpc: remove coredump support A task that dumps open spufs context adds a bunch of extra elf notes describing the SPU state. It is the only reason do_coredump() unshares the file descriptor table. We could make this conditional on spufs but eh. Nothing can consume those notes anymore. gdb dropped Cell Broadband Engine debugging in 9.1 and binutils removed it in 2.34. That's about 6 years ago. So no program can actually read an SPU note out of a core file and probably never did in recent history. Note that the IBM Cell blades that shipped the Cell processor were removed in commit 05bf59fbeef3 ("powerpc/cell: Remove support for IBM Cell Blades"). The PlayStation 3 is the only platform left and nothing there produces or reads these notes. So remove it. Link: https://patch.msgid.link/20260826-work-spufs-coredump-v1-1-579e72a7ab66@kernel.org Acked-by: Arnd Bergmann Signed-off-by: Christian Brauner (Amutable) --- arch/powerpc/Kconfig | 1 - arch/powerpc/include/asm/elf.h | 6 - arch/powerpc/include/asm/spu.h | 3 - arch/powerpc/platforms/cell/Kconfig | 1 - arch/powerpc/platforms/cell/spu_syscalls.c | 20 -- arch/powerpc/platforms/cell/spufs/Makefile | 1 - arch/powerpc/platforms/cell/spufs/coredump.c | 183 ------------------- arch/powerpc/platforms/cell/spufs/file.c | 114 ------------ arch/powerpc/platforms/cell/spufs/spufs.h | 12 -- arch/powerpc/platforms/cell/spufs/syscalls.c | 4 - 10 files changed, 345 deletions(-) delete mode 100644 arch/powerpc/platforms/cell/spufs/coredump.c diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig index 2580e27e432874..40c874fe2f53ac 100644 --- a/arch/powerpc/Kconfig +++ b/arch/powerpc/Kconfig @@ -160,7 +160,6 @@ config PPC select ARCH_HAS_UBSAN select ARCH_HAS_VDSO_ARCH_DATA select ARCH_HAVE_NMI_SAFE_CMPXCHG - select ARCH_HAVE_EXTRA_ELF_NOTES if SPU_BASE select ARCH_KEEP_MEMBLOCK select ARCH_MHP_MEMMAP_ON_MEMORY_ENABLE if PPC_RADIX_MMU select ARCH_MIGHT_HAVE_PC_PARPORT diff --git a/arch/powerpc/include/asm/elf.h b/arch/powerpc/include/asm/elf.h index bb4b94444d3e8a..5dc8c4923eb1df 100644 --- a/arch/powerpc/include/asm/elf.h +++ b/arch/powerpc/include/asm/elf.h @@ -123,12 +123,6 @@ extern int arch_setup_additional_pages(struct linux_binprm *bprm, (0x7ff >> (PAGE_SHIFT - 12)) : \ (0x3ffff >> (PAGE_SHIFT - 12))) -#ifdef CONFIG_SPU_BASE -/* Notes used in ET_CORE. Note name is "SPU//". */ -#define NT_SPU 1 - -#endif /* CONFIG_SPU_BASE */ - #ifdef CONFIG_PPC64 #define get_cache_geometry(level) \ diff --git a/arch/powerpc/include/asm/spu.h b/arch/powerpc/include/asm/spu.h index 96ad4510c89542..7152285b6268fd 100644 --- a/arch/powerpc/include/asm/spu.h +++ b/arch/powerpc/include/asm/spu.h @@ -210,15 +210,12 @@ extern long spu_sys_callback(struct spu_syscall_block *s); /* syscalls implemented in spufs */ struct file; -struct coredump_params; struct spufs_calls { long (*create_thread)(const char __user *name, unsigned int flags, umode_t mode, struct file *neighbor); long (*spu_run)(struct file *filp, __u32 __user *unpc, __u32 __user *ustatus); - int (*coredump_extra_notes_size)(void); - int (*coredump_extra_notes_write)(struct coredump_params *cprm); void (*notify_spus_active)(void); struct module *owner; }; diff --git a/arch/powerpc/platforms/cell/Kconfig b/arch/powerpc/platforms/cell/Kconfig index db65bfcd1e7498..6bd26815c33103 100644 --- a/arch/powerpc/platforms/cell/Kconfig +++ b/arch/powerpc/platforms/cell/Kconfig @@ -10,7 +10,6 @@ config SPU_FS tristate "SPU file system" default m depends on PPC_CELL - depends on COREDUMP select SPU_BASE help The SPU file system is used to access Synergistic Processing diff --git a/arch/powerpc/platforms/cell/spu_syscalls.c b/arch/powerpc/platforms/cell/spu_syscalls.c index 000894e07b027d..8be81207e886a1 100644 --- a/arch/powerpc/platforms/cell/spu_syscalls.c +++ b/arch/powerpc/platforms/cell/spu_syscalls.c @@ -88,26 +88,6 @@ SYSCALL_DEFINE3(spu_run,int, fd, __u32 __user *, unpc, __u32 __user *, ustatus) return calls->spu_run(fd_file(arg), unpc, ustatus); } -#ifdef CONFIG_COREDUMP -int elf_coredump_extra_notes_size(void) -{ - CLASS(spufs_calls, calls)(); - if (!calls) - return 0; - - return calls->coredump_extra_notes_size(); -} - -int elf_coredump_extra_notes_write(struct coredump_params *cprm) -{ - CLASS(spufs_calls, calls)(); - if (!calls) - return 0; - - return calls->coredump_extra_notes_write(cprm); -} -#endif - void notify_spus_active(void) { struct spufs_calls *calls; diff --git a/arch/powerpc/platforms/cell/spufs/Makefile b/arch/powerpc/platforms/cell/spufs/Makefile index 52e4c80ec8d031..60319d4ff25a6a 100644 --- a/arch/powerpc/platforms/cell/spufs/Makefile +++ b/arch/powerpc/platforms/cell/spufs/Makefile @@ -4,7 +4,6 @@ obj-$(CONFIG_SPU_FS) += spufs.o spufs-y += inode.o file.o context.o syscalls.o spufs-y += sched.o backing_ops.o hw_ops.o run.o gang.o spufs-y += switch.o fault.o lscsa_alloc.o -spufs-$(CONFIG_COREDUMP) += coredump.o # magic for the trace events CFLAGS_sched.o := -I$(src) diff --git a/arch/powerpc/platforms/cell/spufs/coredump.c b/arch/powerpc/platforms/cell/spufs/coredump.c deleted file mode 100644 index 301ee7d8b7df02..00000000000000 --- a/arch/powerpc/platforms/cell/spufs/coredump.c +++ /dev/null @@ -1,183 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-or-later -/* - * SPU core dump code - * - * (C) Copyright 2006 IBM Corp. - * - * Author: Dwayne Grant McConnell - */ - -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#include - -#include "spufs.h" - -static int spufs_ctx_note_size(struct spu_context *ctx, int dfd) -{ - int i, sz, total = 0; - char *name; - char fullname[80]; - - for (i = 0; spufs_coredump_read[i].name != NULL; i++) { - name = spufs_coredump_read[i].name; - sz = spufs_coredump_read[i].size; - - sprintf(fullname, "SPU/%d/%s", dfd, name); - - total += sizeof(struct elf_note); - total += roundup(strlen(fullname) + 1, 4); - total += roundup(sz, 4); - } - - return total; -} - -static int match_context(const void *v, struct file *file, unsigned fd) -{ - struct spu_context *ctx; - if (file->f_op != &spufs_context_fops) - return 0; - ctx = SPUFS_I(file_inode(file))->i_ctx; - if (ctx->flags & SPU_CREATE_NOSCHED) - return 0; - return fd + 1; -} - -/* - * The additional architecture-specific notes for Cell are various - * context files in the spu context. - * - * This function iterates over all open file descriptors and sees - * if they are a directory in spufs. In that case we use spufs - * internal functionality to dump them without needing to actually - * open the files. - */ -/* - * descriptor table is not shared, so files can't change or go away. - */ -static struct spu_context *coredump_next_context(int *fd) -{ - struct spu_context *ctx = NULL; - struct file *file; - int n = iterate_fd(current->files, *fd, match_context, NULL); - if (!n) - return NULL; - *fd = n - 1; - - file = fget_raw(*fd); - if (file) { - ctx = SPUFS_I(file_inode(file))->i_ctx; - get_spu_context(ctx); - fput(file); - } - - return ctx; -} - -int spufs_coredump_extra_notes_size(void) -{ - struct spu_context *ctx; - int size = 0, rc, fd; - - fd = 0; - while ((ctx = coredump_next_context(&fd)) != NULL) { - rc = spu_acquire_saved(ctx); - if (rc) { - put_spu_context(ctx); - break; - } - - rc = spufs_ctx_note_size(ctx, fd); - spu_release_saved(ctx); - if (rc < 0) { - put_spu_context(ctx); - break; - } - - size += rc; - - /* start searching the next fd next time */ - fd++; - put_spu_context(ctx); - } - - return size; -} - -static int spufs_arch_write_note(struct spu_context *ctx, int i, - struct coredump_params *cprm, int dfd) -{ - size_t sz = spufs_coredump_read[i].size; - char fullname[80]; - struct elf_note en; - int ret; - - sprintf(fullname, "SPU/%d/%s", dfd, spufs_coredump_read[i].name); - en.n_namesz = strlen(fullname) + 1; - en.n_descsz = sz; - en.n_type = NT_SPU; - - if (!dump_emit(cprm, &en, sizeof(en))) - return -EIO; - if (!dump_emit(cprm, fullname, en.n_namesz)) - return -EIO; - if (!dump_align(cprm, 4)) - return -EIO; - - if (spufs_coredump_read[i].dump) { - ret = spufs_coredump_read[i].dump(ctx, cprm); - if (ret < 0) - return ret; - } else { - char buf[32]; - - ret = snprintf(buf, sizeof(buf), "0x%.16llx", - spufs_coredump_read[i].get(ctx)); - if (ret >= sizeof(buf)) - return sizeof(buf); - - /* count trailing the NULL: */ - if (!dump_emit(cprm, buf, ret + 1)) - return -EIO; - } - - dump_skip_to(cprm, roundup(cprm->pos - ret + sz, 4)); - return 0; -} - -int spufs_coredump_extra_notes_write(struct coredump_params *cprm) -{ - struct spu_context *ctx; - int fd, j, rc; - - fd = 0; - while ((ctx = coredump_next_context(&fd)) != NULL) { - rc = spu_acquire_saved(ctx); - if (rc) - return rc; - - for (j = 0; spufs_coredump_read[j].name != NULL; j++) { - rc = spufs_arch_write_note(ctx, j, cprm, fd); - if (rc) { - spu_release_saved(ctx); - return rc; - } - } - - spu_release_saved(ctx); - - /* start searching the next fd next time */ - fd++; - } - - return 0; -} diff --git a/arch/powerpc/platforms/cell/spufs/file.c b/arch/powerpc/platforms/cell/spufs/file.c index de7494748fecd6..98c47bafaf6781 100644 --- a/arch/powerpc/platforms/cell/spufs/file.c +++ b/arch/powerpc/platforms/cell/spufs/file.c @@ -9,7 +9,6 @@ #undef DEBUG -#include #include #include #include @@ -130,14 +129,6 @@ static ssize_t spufs_attr_write(struct file *file, const char __user *buf, return ret; } -static ssize_t spufs_dump_emit(struct coredump_params *cprm, void *buf, - size_t size) -{ - if (!dump_emit(cprm, buf, size)) - return -EIO; - return size; -} - #define DEFINE_SPUFS_SIMPLE_ATTRIBUTE(__fops, __get, __set, __fmt) \ static int __fops ## _open(struct inode *inode, struct file *file) \ { \ @@ -180,12 +171,6 @@ spufs_mem_release(struct inode *inode, struct file *file) return 0; } -static ssize_t -spufs_mem_dump(struct spu_context *ctx, struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, ctx->ops->get_ls(ctx), LS_SIZE); -} - static ssize_t spufs_mem_read(struct file *file, char __user *buffer, size_t size, loff_t *pos) @@ -466,13 +451,6 @@ spufs_regs_open(struct inode *inode, struct file *file) return 0; } -static ssize_t -spufs_regs_dump(struct spu_context *ctx, struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, ctx->csa.lscsa->gprs, - sizeof(ctx->csa.lscsa->gprs)); -} - static ssize_t spufs_regs_read(struct file *file, char __user *buffer, size_t size, loff_t *pos) @@ -523,13 +501,6 @@ static const struct file_operations spufs_regs_fops = { .llseek = generic_file_llseek, }; -static ssize_t -spufs_fpcr_dump(struct spu_context *ctx, struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, &ctx->csa.lscsa->fpcr, - sizeof(ctx->csa.lscsa->fpcr)); -} - static ssize_t spufs_fpcr_read(struct file *file, char __user * buffer, size_t size, loff_t * pos) @@ -953,15 +924,6 @@ spufs_signal1_release(struct inode *inode, struct file *file) return 0; } -static ssize_t spufs_signal1_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!ctx->csa.spu_chnlcnt_RW[3]) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.spu_chnldata_RW[3], - sizeof(ctx->csa.spu_chnldata_RW[3])); -} - static ssize_t __spufs_signal1_read(struct spu_context *ctx, char __user *buf, size_t len) { @@ -1086,15 +1048,6 @@ spufs_signal2_release(struct inode *inode, struct file *file) return 0; } -static ssize_t spufs_signal2_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!ctx->csa.spu_chnlcnt_RW[4]) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.spu_chnldata_RW[4], - sizeof(ctx->csa.spu_chnldata_RW[4])); -} - static ssize_t __spufs_signal2_read(struct spu_context *ctx, char __user *buf, size_t len) { @@ -1924,15 +1877,6 @@ static const struct file_operations spufs_caps_fops = { .release = single_release, }; -static ssize_t spufs_mbox_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!(ctx->csa.prob.mb_stat_R & 0x0000ff)) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.prob.pu_mb_R, - sizeof(ctx->csa.prob.pu_mb_R)); -} - static ssize_t spufs_mbox_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -1962,15 +1906,6 @@ static const struct file_operations spufs_mbox_info_fops = { .llseek = generic_file_llseek, }; -static ssize_t spufs_ibox_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!(ctx->csa.prob.mb_stat_R & 0xff0000)) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.priv2.puint_mb_R, - sizeof(ctx->csa.priv2.puint_mb_R)); -} - static ssize_t spufs_ibox_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2005,13 +1940,6 @@ static size_t spufs_wbox_info_cnt(struct spu_context *ctx) return (4 - ((ctx->csa.prob.mb_stat_R & 0x00ff00) >> 8)) * sizeof(u32); } -static ssize_t spufs_wbox_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, &ctx->csa.spu_mailbox_data, - spufs_wbox_info_cnt(ctx)); -} - static ssize_t spufs_wbox_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2059,15 +1987,6 @@ static void spufs_get_dma_info(struct spu_context *ctx, } } -static ssize_t spufs_dma_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - struct spu_dma_info info; - - spufs_get_dma_info(ctx, &info); - return spufs_dump_emit(cprm, &info, sizeof(info)); -} - static ssize_t spufs_dma_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2112,15 +2031,6 @@ static void spufs_get_proxydma_info(struct spu_context *ctx, } } -static ssize_t spufs_proxydma_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - struct spu_proxydma_info info; - - spufs_get_proxydma_info(ctx, &info); - return spufs_dump_emit(cprm, &info, sizeof(info)); -} - static ssize_t spufs_proxydma_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2580,27 +2490,3 @@ const struct spufs_tree_descr spufs_dir_debug_contents[] = { { ".ctx", &spufs_ctx_fops, 0444, }, {}, }; - -const struct spufs_coredump_reader spufs_coredump_read[] = { - { "regs", spufs_regs_dump, NULL, sizeof(struct spu_reg128[128])}, - { "fpcr", spufs_fpcr_dump, NULL, sizeof(struct spu_reg128) }, - { "lslr", NULL, spufs_lslr_get, 19 }, - { "decr", NULL, spufs_decr_get, 19 }, - { "decr_status", NULL, spufs_decr_status_get, 19 }, - { "mem", spufs_mem_dump, NULL, LS_SIZE, }, - { "signal1", spufs_signal1_dump, NULL, sizeof(u32) }, - { "signal1_type", NULL, spufs_signal1_type_get, 19 }, - { "signal2", spufs_signal2_dump, NULL, sizeof(u32) }, - { "signal2_type", NULL, spufs_signal2_type_get, 19 }, - { "event_mask", NULL, spufs_event_mask_get, 19 }, - { "event_status", NULL, spufs_event_status_get, 19 }, - { "mbox_info", spufs_mbox_info_dump, NULL, sizeof(u32) }, - { "ibox_info", spufs_ibox_info_dump, NULL, sizeof(u32) }, - { "wbox_info", spufs_wbox_info_dump, NULL, 4 * sizeof(u32)}, - { "dma_info", spufs_dma_info_dump, NULL, sizeof(struct spu_dma_info)}, - { "proxydma_info", spufs_proxydma_info_dump, - NULL, sizeof(struct spu_proxydma_info)}, - { "object-id", NULL, spufs_object_id_get, 19 }, - { "npc", NULL, spufs_npc_get, 19 }, - { NULL }, -}; diff --git a/arch/powerpc/platforms/cell/spufs/spufs.h b/arch/powerpc/platforms/cell/spufs/spufs.h index d33787c57c39a2..612b5075d0ecb8 100644 --- a/arch/powerpc/platforms/cell/spufs/spufs.h +++ b/arch/powerpc/platforms/cell/spufs/spufs.h @@ -232,13 +232,9 @@ extern const struct spufs_tree_descr spufs_dir_debug_contents[]; /* system call implementation */ extern struct spufs_calls spufs_calls; -struct coredump_params; long spufs_run_spu(struct spu_context *ctx, u32 *npc, u32 *status); long spufs_create(const struct path *nd, struct dentry *dentry, unsigned int flags, umode_t mode, struct file *filp); -/* ELF coredump callbacks for writing SPU ELF notes */ -extern int spufs_coredump_extra_notes_size(void); -extern int spufs_coredump_extra_notes_write(struct coredump_params *cprm); extern const struct file_operations spufs_context_fops; @@ -335,14 +331,6 @@ void spufs_stop_callback(struct spu *spu, int irq); void spufs_mfc_callback(struct spu *spu); void spufs_dma_callback(struct spu *spu, int type); -struct spufs_coredump_reader { - char *name; - ssize_t (*dump)(struct spu_context *ctx, struct coredump_params *cprm); - u64 (*get)(struct spu_context *ctx); - size_t size; -}; -extern const struct spufs_coredump_reader spufs_coredump_read[]; - extern int spu_init_csa(struct spu_state *csa); extern void spu_fini_csa(struct spu_state *csa); extern int spu_save(struct spu_state *prev, struct spu *spu); diff --git a/arch/powerpc/platforms/cell/spufs/syscalls.c b/arch/powerpc/platforms/cell/spufs/syscalls.c index ea4ba1b6ce6a96..b6de37150e734d 100644 --- a/arch/powerpc/platforms/cell/spufs/syscalls.c +++ b/arch/powerpc/platforms/cell/spufs/syscalls.c @@ -82,8 +82,4 @@ struct spufs_calls spufs_calls = { .spu_run = do_spu_run, .notify_spus_active = do_notify_spus_active, .owner = THIS_MODULE, -#ifdef CONFIG_COREDUMP - .coredump_extra_notes_size = spufs_coredump_extra_notes_size, - .coredump_extra_notes_write = spufs_coredump_extra_notes_write, -#endif }; From 433967cab51efba55f9e39d0c9372f4af71560e9 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Wed, 26 Aug 2026 18:06:41 +0200 Subject: [PATCH 255/857] coredump: stop unsharing the file descriptor table We currently unshare the file descriptor table before we write the coredump. The only code that ever looked at the file descriptor table was the cell spufs coredump support. It walked all file descriptors to find the spufs contexts it would have to dump as extra elf notes. Now that we killed spufs coredumping stop doing that and update the comments referencing spufs as they ave become stale. Link: https://patch.msgid.link/20260826-work-spufs-coredump-v1-2-579e72a7ab66@kernel.org Acked-by: Arnd Bergmann Signed-off-by: Christian Brauner (Amutable) --- fs/binfmt_elf.c | 4 ++-- fs/coredump.c | 5 ----- 2 files changed, 2 insertions(+), 7 deletions(-) diff --git a/fs/binfmt_elf.c b/fs/binfmt_elf.c index 06d0df1053823e..db32bb40a86704 100644 --- a/fs/binfmt_elf.c +++ b/fs/binfmt_elf.c @@ -2029,7 +2029,7 @@ static int elf_core_dump(struct coredump_params *cprm) { size_t sz = info.size; - /* For cell spufs and x86 xstate */ + /* For x86 xstate */ sz += elf_coredump_extra_notes_size(); phdr4note = kmalloc_obj(*phdr4note); @@ -2093,7 +2093,7 @@ static int elf_core_dump(struct coredump_params *cprm) if (!write_note_info(&info, cprm)) goto end_coredump; - /* For cell spufs and x86 xstate */ + /* For x86 xstate */ if (elf_coredump_extra_notes_write(cprm)) goto end_coredump; diff --git a/fs/coredump.c b/fs/coredump.c index ac3cd74808c640..a49f2fd6be280b 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1118,11 +1118,6 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, if (cn->mask & COREDUMP_REJECT) return; - /* get us an unshared descriptor table; almost always a no-op */ - /* The cell spufs coredump code reads the file descriptor tables */ - if (unshare_files()) - return; - if ((cn->mask & COREDUMP_KERNEL) && !coredump_write(cn, cprm, binfmt)) return; From b00231613d051fde2be50e5636eb8746f4cbf59a Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:19 +0200 Subject: [PATCH 256/857] coredump: refuse negative skips The dump_skip_to() helper calculates a relative skip based on the absolute positon of the coredump: cprm->to_skip = pos - cprm->pos; That's easy to mess up for callers and one already did. This risk endless zero PAGE_SIZE loops or overwriting already written coredump data thereby corrupting the dump. I don't think skipping backwards has any meaning. So warn and refuse. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-2-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/coredump.c b/fs/coredump.c index a49f2fd6be280b..bf20fd3146d8fb 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1251,6 +1251,8 @@ EXPORT_SYMBOL(dump_emit); void dump_skip_to(struct coredump_params *cprm, unsigned long pos) { + if (WARN_ON_ONCE(pos < cprm->pos)) + return; cprm->to_skip = pos - cprm->pos; } EXPORT_SYMBOL(dump_skip_to); From 85e642f9927c9fd0ab3a802a38155b91f2c589ca Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:20 +0200 Subject: [PATCH 257/857] coredump: set the minimum send buffer size The send buffer of the coredump socket is subject to the limit in net.core.wmem_default. af_unix uses sk_sndbuf / 2 - 64 bytes for a single skb. That means a send buffer below that would split a page-sized write into multiple skbs. Raise the send buffer to leave room for a page-sized write plus the header. The default value is well above that. So we really change it when it's below our minimum. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-3-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/fs/coredump.c b/fs/coredump.c index bf20fd3146d8fb..d9c45f2af1223a 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -665,6 +665,9 @@ static int umh_coredump_setup(struct subprocess_info *info, struct cred *new) } #ifdef CONFIG_UNIX +/* af_unix halves the send buffer to size a single skb. */ +#define COREDUMP_SOCK_SNDBUF_MIN (3 * PAGE_SIZE) + static bool coredump_sock_connect(struct core_name *cn, struct coredump_params *cprm) { struct file *file __free(fput) = NULL; @@ -690,6 +693,10 @@ static bool coredump_sock_connect(struct core_name *cn, struct coredump_params * if (retval < 0) return false; + /* Don't let a page-sized write split into several skbs. */ + socket->sk->sk_sndbuf = max_t(int, socket->sk->sk_sndbuf, + COREDUMP_SOCK_SNDBUF_MIN); + file = sock_alloc_file(socket, 0, NULL); if (IS_ERR(file)) return false; From 6e9d372171f73696257cbae544f9fd21746f0cf9 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:21 +0200 Subject: [PATCH 258/857] selftests/coredump: discard the right amount after the coredump request read_coredump_req() gets the leftover wrong twice. It takes the absolute difference of the two sizes, so a test binary that knows a larger struct coredump_req than the kernel sends tries to discard bytes that were never sent. And it hands recv() sizeof(buffer) instead of the number of bytes it wants. So MSG_WAITALL waits for a whole page. Either one blocks until the kernel closes the socket. Which it won't because it is waiting for the coredump ack... It's benign today because struct coredump_req hasn't grown. But let's fix it for the future. Compute the leftover as what the kernel sent beyond what was consumed. Fixes: 59cd658eaf40 ("selftests/coredump: add coredump server selftests") Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-4-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- tools/testing/selftests/coredump/coredump_test_helpers.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 2a20faf9cb0ad3..524fa5370593a4 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -235,10 +235,10 @@ bool read_coredump_req(int fd, struct coredump_req *req) fprintf(stderr, "Read coredump request with size %u and mask 0x%llx\n", req->size, (unsigned long long)req->mask); - if (user_size > kernel_size) - remaining_size = user_size - kernel_size; - else + if (kernel_size > user_size) remaining_size = kernel_size - user_size; + else + remaining_size = 0; if (PAGE_SIZE <= remaining_size) return false; @@ -250,7 +250,7 @@ bool read_coredump_req(int fd, struct coredump_req *req) if (remaining_size) { char buffer[PAGE_SIZE]; - ret = recv(fd, buffer, sizeof(buffer), MSG_WAITALL); + ret = recv(fd, buffer, remaining_size, MSG_WAITALL); if (ret != remaining_size) return false; fprintf(stderr, "Discarded %zu bytes of data after coredump request\n", remaining_size); From 780ab12fc70d15ddb570bf6dba793b81c48f5fe0 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:22 +0200 Subject: [PATCH 259/857] selftests/coredump: collapse the expected request check into the helper All nine callers spell the expected flags. So every new feature bit the kernel learns has to be cargo culted. The callers also all pass COREDUMP_ACK_SIZE_VER0 as the minimum request size although what is being validated is coredump_req->size. And read_coredump_req() makes the same mixup twice more. Clean this all up. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-5-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../coredump/coredump_socket_protocol_test.c | 36 +++++-------------- .../selftests/coredump/coredump_test.h | 3 +- .../coredump/coredump_test_helpers.c | 35 +++++++++++------- 3 files changed, 32 insertions(+), 42 deletions(-) diff --git a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c index d9fa6239b5a9fd..60a357e628eb40 100644 --- a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c +++ b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c @@ -151,9 +151,7 @@ TEST_F(coredump, socket_request_kernel) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_kernel: check_coredump_req failed\n"); goto out; } @@ -301,9 +299,7 @@ TEST_F(coredump, socket_request_userspace) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_userspace: check_coredump_req failed\n"); goto out; } @@ -441,9 +437,7 @@ TEST_F(coredump, socket_request_reject) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_reject: check_coredump_req failed\n"); goto out; } @@ -581,9 +575,7 @@ TEST_F(coredump, socket_request_invalid_flag_combination) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_invalid_flag_combination: check_coredump_req failed\n"); goto out; } @@ -702,9 +694,7 @@ TEST_F(coredump, socket_request_unknown_flag) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_unknown_flag: check_coredump_req failed\n"); goto out; } @@ -822,9 +812,7 @@ TEST_F(coredump, socket_request_invalid_size_small) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_invalid_size_small: check_coredump_req failed\n"); goto out; } @@ -944,9 +932,7 @@ TEST_F(coredump, socket_request_invalid_size_large) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_invalid_size_large: check_coredump_req failed\n"); goto out; } @@ -1355,9 +1341,7 @@ TEST_F_TIMEOUT(coredump, socket_multiple_crashing_coredumps, 500) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "check_coredump_req failed for fd %d\n", fd_coredump); goto out; } @@ -1509,9 +1493,7 @@ TEST_F_TIMEOUT(coredump, socket_multiple_crashing_coredumps_epoll_workers, 500) fprintf(stderr, "socket_multiple_crashing_coredumps_epoll_workers: read_coredump_req failed\n"); goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_multiple_crashing_coredumps_epoll_workers: check_coredump_req failed\n"); goto out; } diff --git a/tools/testing/selftests/coredump/coredump_test.h b/tools/testing/selftests/coredump/coredump_test.h index ed47f01fa53c52..a02809145e2d6f 100644 --- a/tools/testing/selftests/coredump/coredump_test.h +++ b/tools/testing/selftests/coredump/coredump_test.h @@ -51,8 +51,7 @@ bool read_marker(int fd, enum coredump_mark mark); bool read_coredump_req(int fd, struct coredump_req *req); bool send_coredump_ack(int fd, const struct coredump_req *req, __u64 mask, size_t size_ack); -bool check_coredump_req(const struct coredump_req *req, size_t min_size, - __u64 required_mask); +bool check_coredump_req(const struct coredump_req *req); int open_coredump_tmpfile(int fd_tmpfs_detached); void process_coredump_worker(int fd_coredump, int fd_peer_pidfd, int fd_core_file); diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 524fa5370593a4..306711e1b24de6 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -200,7 +200,7 @@ bool read_marker(int fd, enum coredump_mark mark) bool read_coredump_req(int fd, struct coredump_req *req) { ssize_t ret; - size_t field_size, user_size, ack_size, kernel_size, remaining_size; + size_t field_size, user_size, known_size, kernel_size, remaining_size; memset(req, 0, sizeof(*req)); field_size = sizeof(req->size); @@ -214,9 +214,9 @@ bool read_coredump_req(int fd, struct coredump_req *req) } kernel_size = req->size; - if (kernel_size < COREDUMP_ACK_SIZE_VER0) { + if (kernel_size < COREDUMP_REQ_SIZE_VER0) { fprintf(stderr, "read_coredump_req: kernel_size %zu < min %d\n", - kernel_size, COREDUMP_ACK_SIZE_VER0); + kernel_size, COREDUMP_REQ_SIZE_VER0); return false; } if (kernel_size >= PAGE_SIZE) { @@ -225,11 +225,11 @@ bool read_coredump_req(int fd, struct coredump_req *req) return false; } - /* Use the minimum of user and kernel size to read the full request. */ + /* Consume as much of the request as we know about. */ user_size = sizeof(struct coredump_req); - ack_size = user_size < kernel_size ? user_size : kernel_size; - ret = recv(fd, req, ack_size, MSG_WAITALL); - if (ret != ack_size) + known_size = user_size < kernel_size ? user_size : kernel_size; + ret = recv(fd, req, known_size, MSG_WAITALL); + if (ret != known_size) return false; fprintf(stderr, "Read coredump request with size %u and mask 0x%llx\n", @@ -287,15 +287,24 @@ bool send_coredump_ack(int fd, const struct coredump_req *req, return true; } -bool check_coredump_req(const struct coredump_req *req, size_t min_size, - __u64 required_mask) +/* Every option the kernel is expected to advertise in coredump_req->mask. */ +#define TEST_REQ_MASK_ALL \ + (COREDUMP_KERNEL | COREDUMP_USERSPACE | \ + COREDUMP_REJECT | COREDUMP_WAIT) + +bool check_coredump_req(const struct coredump_req *req) { - if (req->size < min_size) - return false; - if ((req->mask & required_mask) != required_mask) + if (req->size < COREDUMP_REQ_SIZE_VER0) { + fprintf(stderr, "%s: size %u below minimum %d\n", + __func__, req->size, COREDUMP_REQ_SIZE_VER0); return false; - if (req->mask & ~required_mask) + } + if (req->mask != TEST_REQ_MASK_ALL) { + fprintf(stderr, "%s: mask 0x%llx, expected 0x%llx\n", + __func__, (unsigned long long)req->mask, + (unsigned long long)TEST_REQ_MASK_ALL); return false; + } return true; } From 23f87d112c087dbc319c537115bb8aa743af20c6 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:23 +0200 Subject: [PATCH 260/857] selftests/coredump: add a separate helper header Right now we have coredump_test.h which pulls in the test harness. So it can't be included in coredump_test_helpers.c and it hand-rolls a bunch of stuff that is not needed. Instead of this mess, split everything out into a separate coredump_test_helpers.h header and make both coredump_test.h and coredump_test_helpers.c include it. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-6-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../selftests/coredump/coredump_test.h | 30 +-------------- .../coredump/coredump_test_helpers.c | 17 +-------- .../coredump/coredump_test_helpers.h | 37 +++++++++++++++++++ 3 files changed, 39 insertions(+), 45 deletions(-) create mode 100644 tools/testing/selftests/coredump/coredump_test_helpers.h diff --git a/tools/testing/selftests/coredump/coredump_test.h b/tools/testing/selftests/coredump/coredump_test.h index a02809145e2d6f..8d99b5cb2f12f8 100644 --- a/tools/testing/selftests/coredump/coredump_test.h +++ b/tools/testing/selftests/coredump/coredump_test.h @@ -3,18 +3,9 @@ #ifndef __COREDUMP_TEST_H #define __COREDUMP_TEST_H -#include -#include -#include - #include "../kselftest_harness.h" -#include "../pidfd/pidfd.h" - -#ifndef PAGE_SIZE -#define PAGE_SIZE 4096 -#endif -#define NUM_THREAD_SPAWN 128 +#include "coredump_test_helpers.h" /* Coredump fixture */ FIXTURE(coredump) @@ -24,15 +15,6 @@ FIXTURE(coredump) int fd_tmpfs_detached; }; -/* Shared helper function declarations */ -void *do_nothing(void *arg); -void crashing_child(void); -int create_detached_tmpfs(void); -int create_and_listen_unix_socket(const char *path); -bool set_core_pattern(const char *pattern); -int get_peer_pidfd(int fd); -bool get_pidfd_info(int fd_peer_pidfd, struct pidfd_info *info); - /* Inline helper that uses harness types */ static inline void wait_and_check_coredump_server(pid_t pid_coredump_server, struct __test_metadata *const _metadata, @@ -45,14 +27,4 @@ static inline void wait_and_check_coredump_server(pid_t pid_coredump_server, ASSERT_EQ(WEXITSTATUS(status), 0); } -/* Protocol helper function declarations */ -ssize_t recv_marker(int fd); -bool read_marker(int fd, enum coredump_mark mark); -bool read_coredump_req(int fd, struct coredump_req *req); -bool send_coredump_ack(int fd, const struct coredump_req *req, - __u64 mask, size_t size_ack); -bool check_coredump_req(const struct coredump_req *req); -int open_coredump_tmpfile(int fd_tmpfs_detached); -void process_coredump_worker(int fd_coredump, int fd_peer_pidfd, int fd_core_file); - #endif /* __COREDUMP_TEST_H */ diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 306711e1b24de6..570fc2e005c206 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -20,23 +20,8 @@ #include #include "../filesystems/wrappers.h" -#include "../pidfd/pidfd.h" -/* Forward declarations to avoid including harness header */ -struct __test_metadata; - -/* Match the fixture definition from coredump_test.h */ -struct _fixture_coredump_data { - char original_core_pattern[256]; - pid_t pid_coredump_server; - int fd_tmpfs_detached; -}; - -#ifndef PAGE_SIZE -#define PAGE_SIZE 4096 -#endif - -#define NUM_THREAD_SPAWN 128 +#include "coredump_test_helpers.h" void *do_nothing(void *arg) { diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.h b/tools/testing/selftests/coredump/coredump_test_helpers.h new file mode 100644 index 00000000000000..45904bd177b802 --- /dev/null +++ b/tools/testing/selftests/coredump/coredump_test_helpers.h @@ -0,0 +1,37 @@ +/* SPDX-License-Identifier: GPL-2.0 */ + +#ifndef __COREDUMP_TEST_HELPERS_H +#define __COREDUMP_TEST_HELPERS_H + +#include +#include +#include + +#include "../pidfd/pidfd.h" + +#ifndef PAGE_SIZE +#define PAGE_SIZE 4096 +#endif + +#define NUM_THREAD_SPAWN 128 + +/* Shared helper function declarations */ +void *do_nothing(void *arg); +void crashing_child(void); +int create_detached_tmpfs(void); +int create_and_listen_unix_socket(const char *path); +bool set_core_pattern(const char *pattern); +int get_peer_pidfd(int fd); +bool get_pidfd_info(int fd_peer_pidfd, struct pidfd_info *info); + +/* Protocol helper function declarations */ +ssize_t recv_marker(int fd); +bool read_marker(int fd, enum coredump_mark mark); +bool read_coredump_req(int fd, struct coredump_req *req); +bool send_coredump_ack(int fd, const struct coredump_req *req, + __u64 mask, size_t size_ack); +bool check_coredump_req(const struct coredump_req *req); +int open_coredump_tmpfile(int fd_tmpfs_detached); +void process_coredump_worker(int fd_coredump, int fd_peer_pidfd, int fd_core_file); + +#endif /* __COREDUMP_TEST_HELPERS_H */ From d6af81c127871253fab17ed634ea442ca96f5159 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:24 +0200 Subject: [PATCH 261/857] coredump: pin the protocol struct sizes COREDUMP_REQ_SIZE_VER0 and COREDUMP_ACK_SIZE_VER0 define the initial struct sizes. Assert that both published sizes still match their structs. While at it fix some issues with docs. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-7-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 2 ++ include/uapi/linux/coredump.h | 4 ++-- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index d9c45f2af1223a..f402cac8af0465 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -759,6 +759,8 @@ static inline bool coredump_sock_send(struct file *file, struct coredump_req *re return ret == sizeof(*req); } +static_assert(sizeof(struct coredump_req) == COREDUMP_REQ_SIZE_VER0); +static_assert(sizeof(struct coredump_ack) == COREDUMP_ACK_SIZE_VER0); static_assert(sizeof(enum coredump_mark) == sizeof(__u32)); static inline bool coredump_sock_mark(struct file *file, enum coredump_mark mark) diff --git a/include/uapi/linux/coredump.h b/include/uapi/linux/coredump.h index dc3789b78af021..662e0468da6e76 100644 --- a/include/uapi/linux/coredump.h +++ b/include/uapi/linux/coredump.h @@ -30,11 +30,11 @@ enum { * member is set to the size of struct coredump_req and provides a hint * to userspace how much data can be read. Userspace may use MSG_PEEK to * peek the size of struct coredump_req and then choose to consume it in - * one go. Userspace may also simply read a COREDUMP_ACK_SIZE_VER0 + * one go. Userspace may also simply read a COREDUMP_REQ_SIZE_VER0 * request. If the size the kernel sends is larger userspace simply * discards any remaining data. * - * The coredump_req->mask member is set to the currently know features. + * The coredump_req->mask member is set to the currently known features. * Userspace may only set coredump_ack->mask to the bits raised by the * kernel in coredump_req->mask. * From e9918d6340332b21086d87e6f48c20c7728e111a Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:25 +0200 Subject: [PATCH 262/857] coredump: move the negotiated mask into struct coredump_params The coredump server negotiates a set of COREDUMP_* options with the kernel. The core dump path cannot see them though. Move the mask into struct coredump_params to make them available. No functional change. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-8-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 15 +++++++-------- include/linux/coredump.h | 2 ++ 2 files changed, 9 insertions(+), 8 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index f402cac8af0465..79f7140e5238ef 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -100,7 +100,6 @@ struct core_name { unsigned int core_pipe_limit; bool core_dumped; enum coredump_type_t core_type; - u64 mask; }; static int expand_corename(struct core_name *cn, int size) @@ -245,9 +244,9 @@ static bool coredump_parse(struct core_name *cn, struct coredump_params *cprm, int pid_in_pattern = 0; int err = 0; - cn->mask = COREDUMP_KERNEL; + cprm->mask = COREDUMP_KERNEL; if (core_pipe_limit) - cn->mask |= COREDUMP_WAIT; + cprm->mask |= COREDUMP_WAIT; cn->used = 0; cn->corename = NULL; cn->core_pipe_limit = 0; @@ -860,7 +859,7 @@ static bool coredump_sock_request(struct core_name *cn, struct coredump_params * return false; } - cn->mask = ack.mask; + cprm->mask = ack.mask; return coredump_sock_mark(cprm->file, COREDUMP_MARK_REQACK); } @@ -1124,16 +1123,16 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, } /* Don't even generate the coredump. */ - if (cn->mask & COREDUMP_REJECT) + if (cprm->mask & COREDUMP_REJECT) return; - if ((cn->mask & COREDUMP_KERNEL) && !coredump_write(cn, cprm, binfmt)) + if ((cprm->mask & COREDUMP_KERNEL) && !coredump_write(cn, cprm, binfmt)) return; coredump_sock_shutdown(cprm->file); /* Let the parent know that a coredump was generated. */ - if (cn->mask & COREDUMP_USERSPACE) + if (cprm->mask & COREDUMP_USERSPACE) cn->core_dumped = true; /* @@ -1141,7 +1140,7 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, * or usermodehelper to finish before exiting so it can e.g., * inspect /proc/. */ - if (cn->mask & COREDUMP_WAIT) { + if (cprm->mask & COREDUMP_WAIT) { switch (cn->core_type) { case COREDUMP_PIPE: wait_for_dump_helpers(cprm->file); diff --git a/include/linux/coredump.h b/include/linux/coredump.h index 7b38ee2e7913be..dc7a05b1bb0a96 100644 --- a/include/linux/coredump.h +++ b/include/linux/coredump.h @@ -26,6 +26,8 @@ struct coredump_params { /* Snapshot of dumpable at dump start. */ enum task_dumpable dumpable; int cpu; + /* COREDUMP_* options negotiated with the coredump server. */ + u64 mask; loff_t written; loff_t pos; loff_t to_skip; From 97cd4023289c5aab242fdaf466abb5cce3b31c76 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:26 +0200 Subject: [PATCH 263/857] coredump: deduplicate the to_skip flush dump_emit() and dump_emit_page() open-code the same flush of the accumulated cprm->to_skip. Move it into a helper. No functional change. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-9-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 17 +++++++++++------ 1 file changed, 11 insertions(+), 6 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index 79f7140e5238ef..ed07a2722539a7 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1246,13 +1246,21 @@ static int __dump_skip(struct coredump_params *cprm, size_t nr) return __dump_emit(cprm, zeroes, nr); } -int dump_emit(struct coredump_params *cprm, const void *addr, int nr) +/* Flush the accumulated hole before writing data. */ +static int dump_flush_skip(struct coredump_params *cprm) { if (cprm->to_skip) { if (!__dump_skip(cprm, cprm->to_skip)) return 0; cprm->to_skip = 0; } + return 1; +} + +int dump_emit(struct coredump_params *cprm, const void *addr, int nr) +{ + if (!dump_flush_skip(cprm)) + return 0; return __dump_emit(cprm, addr, nr); } EXPORT_SYMBOL(dump_emit); @@ -1283,11 +1291,8 @@ static int dump_emit_page(struct coredump_params *cprm, struct page *page) if (!page) return 0; - if (cprm->to_skip) { - if (!__dump_skip(cprm, cprm->to_skip)) - return 0; - cprm->to_skip = 0; - } + if (!dump_flush_skip(cprm)) + return 0; if (cprm->written + PAGE_SIZE > cprm->limit) return 0; if (dump_interrupted()) From fc68e5c5aaee608a98fcf7f3f3d7d0ca1cf5b9df Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:27 +0200 Subject: [PATCH 264/857] coredump: make the dump helper return bool The various dump helpers return one and zero. Every caller just does a boolean test. Convert them to return an actual bool. While at it, drop the externs. No functional changes. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-10-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 63 ++++++++++++++++++++-------------------- include/linux/coredump.h | 14 ++++----- 2 files changed, 39 insertions(+), 38 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index ed07a2722539a7..1407afa36f3dfa 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1205,41 +1205,41 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) * do on a core-file: use only these functions to write out all the * necessary info. */ -static int __dump_emit(struct coredump_params *cprm, const void *addr, int nr) +static bool __dump_emit(struct coredump_params *cprm, const void *addr, int nr) { struct file *file = cprm->file; loff_t pos = file->f_pos; ssize_t n; if (cprm->written + nr > cprm->limit) - return 0; + return false; if (dump_interrupted()) - return 0; + return false; n = __kernel_write(file, addr, nr, &pos); if (n != nr) - return 0; + return false; file->f_pos = pos; cprm->written += n; cprm->pos += n; - return 1; + return true; } -static int __dump_skip(struct coredump_params *cprm, size_t nr) +static bool __dump_skip(struct coredump_params *cprm, size_t nr) { static char zeroes[PAGE_SIZE]; struct file *file = cprm->file; if (file->f_mode & FMODE_LSEEK) { if (dump_interrupted() || vfs_llseek(file, nr, SEEK_CUR) < 0) - return 0; + return false; cprm->pos += nr; - return 1; + return true; } while (nr > PAGE_SIZE) { if (!__dump_emit(cprm, zeroes, PAGE_SIZE)) - return 0; + return false; nr -= PAGE_SIZE; } @@ -1247,20 +1247,20 @@ static int __dump_skip(struct coredump_params *cprm, size_t nr) } /* Flush the accumulated hole before writing data. */ -static int dump_flush_skip(struct coredump_params *cprm) +static bool dump_flush_skip(struct coredump_params *cprm) { if (cprm->to_skip) { if (!__dump_skip(cprm, cprm->to_skip)) - return 0; + return false; cprm->to_skip = 0; } - return 1; + return true; } -int dump_emit(struct coredump_params *cprm, const void *addr, int nr) +bool dump_emit(struct coredump_params *cprm, const void *addr, int nr) { if (!dump_flush_skip(cprm)) - return 0; + return false; return __dump_emit(cprm, addr, nr); } EXPORT_SYMBOL(dump_emit); @@ -1280,7 +1280,7 @@ void dump_skip(struct coredump_params *cprm, size_t nr) EXPORT_SYMBOL(dump_skip); #ifdef CONFIG_ELF_CORE -static int dump_emit_page(struct coredump_params *cprm, struct page *page) +static bool dump_emit_page(struct coredump_params *cprm, struct page *page) { struct bio_vec bvec; struct iov_iter iter; @@ -1289,25 +1289,25 @@ static int dump_emit_page(struct coredump_params *cprm, struct page *page) ssize_t n; if (!page) - return 0; + return false; if (!dump_flush_skip(cprm)) - return 0; + return false; if (cprm->written + PAGE_SIZE > cprm->limit) - return 0; + return false; if (dump_interrupted()) - return 0; + return false; pos = file->f_pos; bvec_set_page(&bvec, page, PAGE_SIZE, 0); iov_iter_bvec(&iter, ITER_SOURCE, &bvec, 1, PAGE_SIZE); n = __kernel_write_iter(cprm->file, &iter, &pos); if (n != PAGE_SIZE) - return 0; + return false; file->f_pos = pos; cprm->written += PAGE_SIZE; cprm->pos += PAGE_SIZE; - return 1; + return true; } /* @@ -1339,18 +1339,19 @@ static inline struct page *dump_page_copy(struct page *src, struct page *dst) } #endif -int dump_user_range(struct coredump_params *cprm, unsigned long start, - unsigned long len) +bool dump_user_range(struct coredump_params *cprm, unsigned long start, + unsigned long len) { unsigned long addr; struct page *dump_page; - int locked, ret; + int locked; + bool ret; dump_page = dump_page_alloc(); if (!dump_page) - return 0; + return false; - ret = 0; + ret = false; locked = 0; for (addr = start; addr < start + len; addr += PAGE_SIZE) { struct page *page; @@ -1374,7 +1375,7 @@ int dump_user_range(struct coredump_params *cprm, unsigned long start, mmap_read_unlock(current->mm); locked = 0; } - int stop = !dump_emit_page(cprm, dump_page_copy(page, dump_page)); + bool stop = !dump_emit_page(cprm, dump_page_copy(page, dump_page)); put_page(page); if (stop) goto out; @@ -1393,7 +1394,7 @@ int dump_user_range(struct coredump_params *cprm, unsigned long start, } cond_resched(); } - ret = 1; + ret = true; out: if (locked) mmap_read_unlock(current->mm); @@ -1403,14 +1404,14 @@ int dump_user_range(struct coredump_params *cprm, unsigned long start, } #endif -int dump_align(struct coredump_params *cprm, int align) +bool dump_align(struct coredump_params *cprm, int align) { unsigned mod = (cprm->pos + cprm->to_skip) & (align - 1); if (align & (align - 1)) - return 0; + return false; if (mod) cprm->to_skip += align - mod; - return 1; + return true; } EXPORT_SYMBOL(dump_align); diff --git a/include/linux/coredump.h b/include/linux/coredump.h index dc7a05b1bb0a96..943bddfb22bf8e 100644 --- a/include/linux/coredump.h +++ b/include/linux/coredump.h @@ -43,13 +43,13 @@ extern unsigned int core_file_note_size_limit; * These are the only things you should do on a core-file: use only these * functions to write out all the necessary info. */ -extern void dump_skip_to(struct coredump_params *cprm, unsigned long to); -extern void dump_skip(struct coredump_params *cprm, size_t nr); -extern int dump_emit(struct coredump_params *cprm, const void *addr, int nr); -extern int dump_align(struct coredump_params *cprm, int align); -int dump_user_range(struct coredump_params *cprm, unsigned long start, - unsigned long len); -extern void vfs_coredump(const kernel_siginfo_t *siginfo); +void dump_skip_to(struct coredump_params *cprm, unsigned long to); +void dump_skip(struct coredump_params *cprm, size_t nr); +bool dump_emit(struct coredump_params *cprm, const void *addr, int nr); +bool dump_align(struct coredump_params *cprm, int align); +bool dump_user_range(struct coredump_params *cprm, unsigned long start, + unsigned long len); +void vfs_coredump(const kernel_siginfo_t *siginfo); /* * Logging for the coredump code, ratelimited. From b7b45cfa2ff204b0c0a0f4faa2903a9cc125c64f Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:28 +0200 Subject: [PATCH 265/857] coredump: always chunk writes Right now dump_emit() is the only coredump helper that writes buffers larger than a page in one call. For elf notes that can easily blow past PAGE_SIZE. That's annoying because neither pipes nor af_unix sockets take such writes in one piece. If a signal arrives while the writer is waiting they drop a short write. With the coredump records work coming up that means header and its data are desynchronized. A write that fits in one pipe buffer or one skb doesn't suffer from this. So split all writes up, including elf notes, and cap every write at a page. The coredump socket already raises sk_sndbuf far enough for a page to fit a single skb and pipes always work that way. That means dump_interrupted() is now checked once per page. So a large coredump stops earlier (good). An empty write no longer issues a zero-length write. The rlimit core check stays where it was. It continues refusing whole writes. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-11-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 37 ++++++++++++++++++++++++++++++------- 1 file changed, 30 insertions(+), 7 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index 1407afa36f3dfa..50ab6a1165831a 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1205,19 +1205,21 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) * do on a core-file: use only these functions to write out all the * necessary info. */ -static bool __dump_emit(struct coredump_params *cprm, const void *addr, int nr) +/* One write, never more than a page. See __dump_emit(). */ +static bool dump_emit_chunk(struct coredump_params *cprm, const void *addr, + int nr) { struct file *file = cprm->file; loff_t pos = file->f_pos; ssize_t n; - if (cprm->written + nr > cprm->limit) - return false; if (dump_interrupted()) return false; + n = __kernel_write(file, addr, nr, &pos); if (n != nr) return false; + file->f_pos = pos; cprm->written += n; cprm->pos += n; @@ -1225,6 +1227,24 @@ static bool __dump_emit(struct coredump_params *cprm, const void *addr, int nr) return true; } +static bool __dump_emit(struct coredump_params *cprm, const void *addr, int nr) +{ + if (cprm->written + nr > cprm->limit) + return false; + + while (nr) { + int chunk = min_t(int, nr, PAGE_SIZE); + + if (!dump_emit_chunk(cprm, addr, chunk)) + return false; + + addr += chunk; + nr -= chunk; + } + + return true; +} + static bool __dump_skip(struct coredump_params *cprm, size_t nr) { static char zeroes[PAGE_SIZE]; @@ -1237,13 +1257,16 @@ static bool __dump_skip(struct coredump_params *cprm, size_t nr) return true; } - while (nr > PAGE_SIZE) { - if (!__dump_emit(cprm, zeroes, PAGE_SIZE)) + while (nr) { + size_t chunk = min_t(size_t, nr, PAGE_SIZE); + + if (!__dump_emit(cprm, zeroes, chunk)) return false; - nr -= PAGE_SIZE; + + nr -= chunk; } - return __dump_emit(cprm, zeroes, nr); + return true; } /* Flush the accumulated hole before writing data. */ From 3bdb1bd63b4454c2c123e59125bace250f2d719f Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:29 +0200 Subject: [PATCH 266/857] coredump: clean up coredump state handling Right now coredump state handling is messy. The binfmt->core_dump:: methods return 1 when the coredump method did anything at all which means that a partial write counts as having dumped core. This is fine as the state is really only used to indicate that a coredump event occurred in the exit status of the task. That should obviously be indicated even if the actual writeout of the coredump failed. But it means the coredump method of the binary formats is different from all the other coredump helpers. And there's no way to communicate to userspace that a coredump was truncated. We'll add support for that in a second. For now, clean this up. Add a flag member into struct coredump_params. Let the coredump method raise COREDUMP_STATE_STARTED. This is what coredump_finish() will end up using to splice in the coredump bit into the exit status. This allows us to let the return value mean success or failure and align it with the other coredump helpers. We also start raising COREDUMP_STATE_TRUNCATED. This will be used in the next patches to communicate truncation to userspace via the coredump socket. No functional changes. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-12-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/binfmt_elf.c | 12 +++++++----- fs/binfmt_elf_fdpic.c | 12 +++++++----- fs/coredump.c | 31 +++++++++++++++++-------------- include/linux/binfmts.h | 3 ++- include/linux/coredump.h | 12 ++++++++++++ 5 files changed, 45 insertions(+), 25 deletions(-) diff --git a/fs/binfmt_elf.c b/fs/binfmt_elf.c index db32bb40a86704..6b7ffac5d66558 100644 --- a/fs/binfmt_elf.c +++ b/fs/binfmt_elf.c @@ -74,7 +74,7 @@ static int load_elf_binary(struct linux_binprm *bprm); * don't even try. */ #ifdef CONFIG_ELF_CORE -static int elf_core_dump(struct coredump_params *cprm); +static bool elf_core_dump(struct coredump_params *cprm); #else #define elf_core_dump NULL #endif @@ -1987,9 +1987,9 @@ static void fill_extnum_info(struct elfhdr *elf, struct elf_shdr *shdr4extnum, * and then they are actually written out. If we run out of core limit * we just truncate. */ -static int elf_core_dump(struct coredump_params *cprm) +static bool elf_core_dump(struct coredump_params *cprm) { - int has_dumped = 0; + bool ret = false; int segs, i; struct elfhdr elf; loff_t offset = 0, dataoff; @@ -2020,7 +2020,7 @@ static int elf_core_dump(struct coredump_params *cprm) if (!fill_note_info(&elf, e_phnum, &info, cprm)) goto end_coredump; - has_dumped = 1; + cprm->state |= COREDUMP_STATE_STARTED; offset += sizeof(elf); /* ELF header */ offset += segs * sizeof(struct elf_phdr); /* Program headers */ @@ -2115,11 +2115,13 @@ static int elf_core_dump(struct coredump_params *cprm) goto end_coredump; } + ret = true; + end_coredump: free_note_info(&info); kfree(shdr4extnum); kfree(phdr4note); - return has_dumped; + return ret; } #endif /* CONFIG_ELF_CORE */ diff --git a/fs/binfmt_elf_fdpic.c b/fs/binfmt_elf_fdpic.c index 068c46875c7442..005f0a08448381 100644 --- a/fs/binfmt_elf_fdpic.c +++ b/fs/binfmt_elf_fdpic.c @@ -75,7 +75,7 @@ static int elf_fdpic_map_file_by_direct_mmap(struct elf_fdpic_params *, struct file *, struct mm_struct *); #ifdef CONFIG_ELF_CORE -static int elf_fdpic_core_dump(struct coredump_params *cprm); +static bool elf_fdpic_core_dump(struct coredump_params *cprm); #endif static struct linux_binfmt elf_fdpic_format = { @@ -1477,9 +1477,9 @@ static bool elf_fdpic_dump_segments(struct coredump_params *cprm, * and then they are actually written out. If we run out of core limit * we just truncate. */ -static int elf_fdpic_core_dump(struct coredump_params *cprm) +static bool elf_fdpic_core_dump(struct coredump_params *cprm) { - int has_dumped = 0; + bool ret = false; int segs; int i; struct elfhdr *elf = NULL; @@ -1536,7 +1536,7 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) /* Set up header */ fill_elf_fdpic_header(elf, e_phnum); - has_dumped = 1; + cprm->state |= COREDUMP_STATE_STARTED; /* * Set up the notes in similar form to SVR4 core dumps made * with info from their /proc. @@ -1656,6 +1656,8 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) cprm->file->f_pos, offset); } + ret = true; + end_coredump: while (thread_list) { tmp = thread_list; @@ -1666,7 +1668,7 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) kfree(elf); kfree(psinfo); kfree(shdr4extnum); - return has_dumped; + return ret; } #endif /* CONFIG_ELF_CORE */ diff --git a/fs/coredump.c b/fs/coredump.c index 50ab6a1165831a..3f58fdac6907a2 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -98,7 +98,6 @@ struct core_name { char *corename __counted_by_ptr(size); int used, size; unsigned int core_pipe_limit; - bool core_dumped; enum coredump_type_t core_type; }; @@ -250,7 +249,6 @@ static bool coredump_parse(struct core_name *cn, struct coredump_params *cprm, cn->used = 0; cn->corename = NULL; cn->core_pipe_limit = 0; - cn->core_dumped = false; if (*pat_ptr == '|') cn->core_type = COREDUMP_PIPE; else if (*pat_ptr == '@') @@ -549,13 +547,13 @@ static int coredump_wait(int exit_code, struct core_state *core_state) return core_waiters; } -static void coredump_finish(bool core_dumped) +static void coredump_finish(enum coredump_state state) { struct core_thread *curr, *next; struct task_struct *task; spin_lock_irq(¤t->sighand->siglock); - if (core_dumped && !__fatal_signal_pending(current)) + if ((state & COREDUMP_STATE_STARTED) && !__fatal_signal_pending(current)) current->signal->group_exit_code |= 0x80; next = current->signal->core_state->dumper.next; current->signal->core_state = NULL; @@ -1040,19 +1038,23 @@ static bool coredump_pipe(struct core_name *cn, struct coredump_params *cprm, return true; } -static bool coredump_write(struct core_name *cn, - struct coredump_params *cprm, - const struct linux_binfmt *binfmt) +static bool coredump_write(struct coredump_params *cprm, + const struct linux_binfmt *binfmt) { - if (dump_interrupted()) + if (dump_interrupted()) { + cprm->state |= COREDUMP_STATE_TRUNCATED; return true; + } - if (!dump_vma_snapshot(cprm)) + if (!dump_vma_snapshot(cprm)) { + cprm->state |= COREDUMP_STATE_TRUNCATED; return false; + } file_start_write(cprm->file); - cn->core_dumped = binfmt->core_dump(cprm); + if (!binfmt->core_dump(cprm)) + cprm->state |= COREDUMP_STATE_TRUNCATED; /* * Ensures that file size is big enough to contain the current * file postion. This prevents gdb from complaining about @@ -1061,7 +1063,8 @@ static bool coredump_write(struct core_name *cn, */ if (cprm->to_skip) { cprm->to_skip--; - dump_emit(cprm, "", 1); + if (!dump_emit(cprm, "", 1)) + cprm->state |= COREDUMP_STATE_TRUNCATED; } file_end_write(cprm->file); free_vma_snapshot(cprm); @@ -1077,7 +1080,7 @@ static void coredump_cleanup(struct core_name *cn, struct coredump_params *cprm) atomic_dec(&core_pipe_count); } kfree(cn->corename); - coredump_finish(cn->core_dumped); + coredump_finish(cprm->state); } static inline bool coredump_skip(const struct coredump_params *cprm, @@ -1126,14 +1129,14 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, if (cprm->mask & COREDUMP_REJECT) return; - if ((cprm->mask & COREDUMP_KERNEL) && !coredump_write(cn, cprm, binfmt)) + if ((cprm->mask & COREDUMP_KERNEL) && !coredump_write(cprm, binfmt)) return; coredump_sock_shutdown(cprm->file); /* Let the parent know that a coredump was generated. */ if (cprm->mask & COREDUMP_USERSPACE) - cn->core_dumped = true; + cprm->state |= COREDUMP_STATE_STARTED; /* * When core_pipe_limit is set we wait for the coredump server diff --git a/include/linux/binfmts.h b/include/linux/binfmts.h index f686a37f7a0a04..2e87faf9a8c285 100644 --- a/include/linux/binfmts.h +++ b/include/linux/binfmts.h @@ -128,7 +128,8 @@ struct linux_binfmt { struct module *module; int (*load_binary)(struct linux_binprm *); #ifdef CONFIG_COREDUMP - int (*core_dump)(struct coredump_params *cprm); + /* Returns true if the whole coredump was written. */ + bool (*core_dump)(struct coredump_params *cprm); unsigned long min_coredump; /* minimal dump size */ #endif } __randomize_layout; diff --git a/include/linux/coredump.h b/include/linux/coredump.h index 943bddfb22bf8e..709388dd565914 100644 --- a/include/linux/coredump.h +++ b/include/linux/coredump.h @@ -9,6 +9,16 @@ #include #ifdef CONFIG_COREDUMP +/** + * enum coredump_state - what happened while the coredump was written + * @COREDUMP_STATE_STARTED: the dumper committed to writing a coredump + * @COREDUMP_STATE_TRUNCATED: the dumper stopped before it had written all of it + */ +enum coredump_state { + COREDUMP_STATE_STARTED = (1U << 0), + COREDUMP_STATE_TRUNCATED = (1U << 1), +}; + struct core_vma_metadata { unsigned long start, end; vm_flags_t flags; @@ -28,6 +38,8 @@ struct coredump_params { int cpu; /* COREDUMP_* options negotiated with the coredump server. */ u64 mask; + /* COREDUMP_STATE_* raised while the coredump is written. */ + enum coredump_state state; loff_t written; loff_t pos; loff_t to_skip; From 523b6c57cdf9dfcdc16ca497d227f3cf8b601947 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:30 +0200 Subject: [PATCH 267/857] coredump: add COREDUMP_RECORDS to the coredump socket protocol Currently a coredump sent over a socket is raw data. The kernel knows things about the data it's sending that are useful for a coredump server. For example, it knows where the unpopulated parts of a mapping are. We can't communicate this to userspace currently though. Add a COREDUMP_RECORDS feature bit and a struct coredump_record_header so userspace can negotiate that feature. Instead of a byte stream it gets a header plus data. Reassembling the records yields the same coredump that would have been sent without them. The record itself is also versioned and thus extensible with the same protocol as the ack-req sync. A record stream ends explicitly. A COREDUMP_RECORD_END record closes it, carries no data and reports the size of the coredump. It is only sent once the whole coredump has been written. If it's missing the coredump should be treated as truncated. This just adds the infrastructure. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-13-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- include/uapi/linux/coredump.h | 66 +++++++++++++++++++++++++++++++++++ 1 file changed, 66 insertions(+) diff --git a/include/uapi/linux/coredump.h b/include/uapi/linux/coredump.h index 662e0468da6e76..0bd5c8662ebe3b 100644 --- a/include/uapi/linux/coredump.h +++ b/include/uapi/linux/coredump.h @@ -11,12 +11,16 @@ * @COREDUMP_USERSPACE: userspace writes coredump * @COREDUMP_REJECT: don't generate coredump * @COREDUMP_WAIT: wait for coredump server + * @COREDUMP_RECORDS: send the coredump as a sequence of records instead of + * as a plain byte stream, see struct coredump_record_header; + * requires COREDUMP_KERNEL */ enum { COREDUMP_KERNEL = (1ULL << 0), COREDUMP_USERSPACE = (1ULL << 1), COREDUMP_REJECT = (1ULL << 2), COREDUMP_WAIT = (1ULL << 3), + COREDUMP_RECORDS = (1ULL << 4), }; /** @@ -101,4 +105,66 @@ enum coredump_mark { __COREDUMP_MARK_MAX = (1U << 31), }; +/** + * enum coredump_record_type - Type of a coredump record + * + * @COREDUMP_RECORD_DATA: the header is followed by ->len bytes of data + * @COREDUMP_RECORD_END: the coredump ends here, the header is not followed + * by any data and no further record is sent + * @__COREDUMP_RECORD_TYPE_MAX: the maximum coredump record type value + */ +enum coredump_record_type { + COREDUMP_RECORD_DATA = 0U, + COREDUMP_RECORD_END = 1U, + __COREDUMP_RECORD_TYPE_MAX = (1U << 31), +}; + +/** + * struct coredump_record_header - header of a coredump record + * @size: size of struct coredump_record_header + * @type: one of enum coredump_record_type + * @flags: modifiers for this record + * @offset: offset in the coredump this record starts at + * @len: number of coredump bytes this record accounts for + * + * If the coredump server raises COREDUMP_RECORDS in coredump_ack->mask + * the kernel doesn't send the coredump as a plain byte stream. It sends + * a sequence of records instead. A COREDUMP_RECORD_DATA record is + * followed by @len bytes of actual coredump data. Records arrive in + * order and leave no gaps. So @offset is the sum of the @len of all + * records before it. + * + * The last record is a COREDUMP_RECORD_END record. It is followed by + * nothing. Its @len is zero. Its @offset is the size of the coredump. + * The kernel only sends it once it has written the whole coredump. A + * server that hits end-of-file without having seen an end record must + * treat the coredump as incomplete. + * + * The @size member is set to the size of struct coredump_record_header + * the kernel knows and lets the header grow later. It comes first so it + * can be peeked. Userspace must consume @size bytes and discard + * anything beyond what it knows. It must refuse a @size smaller than + * COREDUMP_RECORD_HEADER_SIZE_VER0. @size covers the header alone. + * @offset and @len count coredump bytes. + * + * The @flags member carries modifiers that change how the record is to + * be interpreted. No flag is defined yet. Userspace must refuse a + * record carrying a flag or a type it doesn't know. Every new record + * type is raised in coredump_req->mask as a feature of its own. A + * server only ever sees the types it asked for. + * + * COREDUMP_RECORDS must be combined with COREDUMP_KERNEL. + */ +struct coredump_record_header { + __u32 size; + __u32 type; + __u64 flags; + __u64 offset; + __u64 len; +}; + +enum { + COREDUMP_RECORD_HEADER_SIZE_VER0 = 32U, /* size of first published struct */ +}; + #endif /* _UAPI_LINUX_COREDUMP_H */ From cc4003a4d2fe58aae2ff18faffbbd01c674b5dc9 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:31 +0200 Subject: [PATCH 268/857] coredump: add COREDUMP_SPARSE to the coredump socket protocol A coredump with a lot of unpopulated mappings sends useless amounts of zero data to userspace. This is nonsensical. While __dump_skip() can seek over them when the target is a regular file a socket cannot do this. COREDUMP_RECORDS put the zeroes in records but it didn't get rid of them. Add a COREDUMP_SPARSE feature bit and a COREDUMP_RECORD_ZERO record type. A zero record is a bare header that tells userspace how many zero bytes were skipped. So a hole crosses the socket as one header no matter how long it is. The coredump server can recreate this sparsely. Zero records only exist inside a record stream. COREDUMP_SPARSE requires COREDUMP_RECORDS. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-14-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- include/uapi/linux/coredump.h | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/include/uapi/linux/coredump.h b/include/uapi/linux/coredump.h index 0bd5c8662ebe3b..f3771861ca4857 100644 --- a/include/uapi/linux/coredump.h +++ b/include/uapi/linux/coredump.h @@ -14,6 +14,8 @@ * @COREDUMP_RECORDS: send the coredump as a sequence of records instead of * as a plain byte stream, see struct coredump_record_header; * requires COREDUMP_KERNEL + * @COREDUMP_SPARSE: describe the holes in the coredump as zero records + * instead of transferring them; requires COREDUMP_RECORDS */ enum { COREDUMP_KERNEL = (1ULL << 0), @@ -21,6 +23,7 @@ enum { COREDUMP_REJECT = (1ULL << 2), COREDUMP_WAIT = (1ULL << 3), COREDUMP_RECORDS = (1ULL << 4), + COREDUMP_SPARSE = (1ULL << 5), }; /** @@ -111,11 +114,14 @@ enum coredump_mark { * @COREDUMP_RECORD_DATA: the header is followed by ->len bytes of data * @COREDUMP_RECORD_END: the coredump ends here, the header is not followed * by any data and no further record is sent + * @COREDUMP_RECORD_ZERO: the header stands for ->len zero bytes and is not + * followed by any data * @__COREDUMP_RECORD_TYPE_MAX: the maximum coredump record type value */ enum coredump_record_type { COREDUMP_RECORD_DATA = 0U, COREDUMP_RECORD_END = 1U, + COREDUMP_RECORD_ZERO = 2U, __COREDUMP_RECORD_TYPE_MAX = (1U << 31), }; @@ -130,9 +136,11 @@ enum coredump_record_type { * If the coredump server raises COREDUMP_RECORDS in coredump_ack->mask * the kernel doesn't send the coredump as a plain byte stream. It sends * a sequence of records instead. A COREDUMP_RECORD_DATA record is - * followed by @len bytes of actual coredump data. Records arrive in - * order and leave no gaps. So @offset is the sum of the @len of all - * records before it. + * followed by @len bytes of actual coredump data. A + * COREDUMP_RECORD_ZERO record is followed by nothing and stands for + * @len zero bytes. A server that didn't raise COREDUMP_SPARSE never + * sees a zero record. Records arrive in order and leave no gaps. So + * @offset is the sum of the @len of all records before it. * * The last record is a COREDUMP_RECORD_END record. It is followed by * nothing. Its @len is zero. Its @offset is the size of the coredump. @@ -153,7 +161,8 @@ enum coredump_record_type { * type is raised in coredump_req->mask as a feature of its own. A * server only ever sees the types it asked for. * - * COREDUMP_RECORDS must be combined with COREDUMP_KERNEL. + * COREDUMP_RECORDS must be combined with COREDUMP_KERNEL, and + * COREDUMP_SPARSE with COREDUMP_RECORDS. */ struct coredump_record_header { __u32 size; From 4e0d7642ffc0ecfb1fec6aaa6ad317ce8f29c7e9 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:32 +0200 Subject: [PATCH 269/857] tools: sync coredump.h header Sync the headers for the selftests. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-15-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- tools/include/uapi/linux/coredump.h | 79 ++++++++++++++++++++++++++++- 1 file changed, 77 insertions(+), 2 deletions(-) diff --git a/tools/include/uapi/linux/coredump.h b/tools/include/uapi/linux/coredump.h index dc3789b78af021..f3771861ca4857 100644 --- a/tools/include/uapi/linux/coredump.h +++ b/tools/include/uapi/linux/coredump.h @@ -11,12 +11,19 @@ * @COREDUMP_USERSPACE: userspace writes coredump * @COREDUMP_REJECT: don't generate coredump * @COREDUMP_WAIT: wait for coredump server + * @COREDUMP_RECORDS: send the coredump as a sequence of records instead of + * as a plain byte stream, see struct coredump_record_header; + * requires COREDUMP_KERNEL + * @COREDUMP_SPARSE: describe the holes in the coredump as zero records + * instead of transferring them; requires COREDUMP_RECORDS */ enum { COREDUMP_KERNEL = (1ULL << 0), COREDUMP_USERSPACE = (1ULL << 1), COREDUMP_REJECT = (1ULL << 2), COREDUMP_WAIT = (1ULL << 3), + COREDUMP_RECORDS = (1ULL << 4), + COREDUMP_SPARSE = (1ULL << 5), }; /** @@ -30,11 +37,11 @@ enum { * member is set to the size of struct coredump_req and provides a hint * to userspace how much data can be read. Userspace may use MSG_PEEK to * peek the size of struct coredump_req and then choose to consume it in - * one go. Userspace may also simply read a COREDUMP_ACK_SIZE_VER0 + * one go. Userspace may also simply read a COREDUMP_REQ_SIZE_VER0 * request. If the size the kernel sends is larger userspace simply * discards any remaining data. * - * The coredump_req->mask member is set to the currently know features. + * The coredump_req->mask member is set to the currently known features. * Userspace may only set coredump_ack->mask to the bits raised by the * kernel in coredump_req->mask. * @@ -101,4 +108,72 @@ enum coredump_mark { __COREDUMP_MARK_MAX = (1U << 31), }; +/** + * enum coredump_record_type - Type of a coredump record + * + * @COREDUMP_RECORD_DATA: the header is followed by ->len bytes of data + * @COREDUMP_RECORD_END: the coredump ends here, the header is not followed + * by any data and no further record is sent + * @COREDUMP_RECORD_ZERO: the header stands for ->len zero bytes and is not + * followed by any data + * @__COREDUMP_RECORD_TYPE_MAX: the maximum coredump record type value + */ +enum coredump_record_type { + COREDUMP_RECORD_DATA = 0U, + COREDUMP_RECORD_END = 1U, + COREDUMP_RECORD_ZERO = 2U, + __COREDUMP_RECORD_TYPE_MAX = (1U << 31), +}; + +/** + * struct coredump_record_header - header of a coredump record + * @size: size of struct coredump_record_header + * @type: one of enum coredump_record_type + * @flags: modifiers for this record + * @offset: offset in the coredump this record starts at + * @len: number of coredump bytes this record accounts for + * + * If the coredump server raises COREDUMP_RECORDS in coredump_ack->mask + * the kernel doesn't send the coredump as a plain byte stream. It sends + * a sequence of records instead. A COREDUMP_RECORD_DATA record is + * followed by @len bytes of actual coredump data. A + * COREDUMP_RECORD_ZERO record is followed by nothing and stands for + * @len zero bytes. A server that didn't raise COREDUMP_SPARSE never + * sees a zero record. Records arrive in order and leave no gaps. So + * @offset is the sum of the @len of all records before it. + * + * The last record is a COREDUMP_RECORD_END record. It is followed by + * nothing. Its @len is zero. Its @offset is the size of the coredump. + * The kernel only sends it once it has written the whole coredump. A + * server that hits end-of-file without having seen an end record must + * treat the coredump as incomplete. + * + * The @size member is set to the size of struct coredump_record_header + * the kernel knows and lets the header grow later. It comes first so it + * can be peeked. Userspace must consume @size bytes and discard + * anything beyond what it knows. It must refuse a @size smaller than + * COREDUMP_RECORD_HEADER_SIZE_VER0. @size covers the header alone. + * @offset and @len count coredump bytes. + * + * The @flags member carries modifiers that change how the record is to + * be interpreted. No flag is defined yet. Userspace must refuse a + * record carrying a flag or a type it doesn't know. Every new record + * type is raised in coredump_req->mask as a feature of its own. A + * server only ever sees the types it asked for. + * + * COREDUMP_RECORDS must be combined with COREDUMP_KERNEL, and + * COREDUMP_SPARSE with COREDUMP_RECORDS. + */ +struct coredump_record_header { + __u32 size; + __u32 type; + __u64 flags; + __u64 offset; + __u64 len; +}; + +enum { + COREDUMP_RECORD_HEADER_SIZE_VER0 = 32U, /* size of first published struct */ +}; + #endif /* _UAPI_LINUX_COREDUMP_H */ From d5834c1ee0ca0bf5339602667c2d7e6ca56a2252 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:33 +0200 Subject: [PATCH 270/857] coredump: send the coredump in records if requested When the coredump server raises COREDUMP_RECORDS send the coredump in records. A record consists of a struct coredump_record_header and data. A header and the bytes it describes go out in one iovec. A hole is flushed through __dump_emit() like before. So zeroes still are sent on the socket as actual data records. Making holes cheap is COREDUMP_SPARSE's job. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-16-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 150 ++++++++++++++---- include/linux/coredump.h | 5 + .../coredump/coredump_test_helpers.c | 2 +- 3 files changed, 127 insertions(+), 30 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index 3f58fdac6907a2..412c7672cc9813 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -51,7 +51,6 @@ #include #include #include -#include #include #include @@ -68,6 +67,7 @@ static bool dump_vma_snapshot(struct coredump_params *cprm); static void free_vma_snapshot(struct coredump_params *cprm); +static void dump_end_record(struct coredump_params *cprm); #define CORE_FILE_NOTE_SIZE_DEFAULT (4*1024*1024) /* Define a reasonable max cap */ @@ -661,6 +661,8 @@ static int umh_coredump_setup(struct subprocess_info *info, struct cred *new) return 0; } +static_assert(sizeof(struct coredump_record_header) == COREDUMP_RECORD_HEADER_SIZE_VER0); + #ifdef CONFIG_UNIX /* af_unix halves the send buffer to size a single skb. */ #define COREDUMP_SOCK_SNDBUF_MIN (3 * PAGE_SIZE) @@ -803,7 +805,8 @@ static bool coredump_sock_request(struct core_name *cn, struct coredump_params * struct coredump_req req = { .size = sizeof(struct coredump_req), .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT, + COREDUMP_REJECT | COREDUMP_WAIT | + COREDUMP_RECORDS, .size_ack = sizeof(struct coredump_ack), }; struct coredump_ack ack = {}; @@ -857,6 +860,19 @@ static bool coredump_sock_request(struct core_name *cn, struct coredump_params * return false; } + /* Records only describe a coredump the kernel writes. */ + if ((ack.mask & COREDUMP_RECORDS) && !(ack.mask & COREDUMP_KERNEL)) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_CONFLICTING); + return false; + } + + /* Record header scratch; a bvec can't point at the stack. */ + if (ack.mask & COREDUMP_RECORDS) { + cprm->record_hdr = kmalloc_obj(*cprm->record_hdr); + if (!cprm->record_hdr) + return false; + } + cprm->mask = ack.mask; return coredump_sock_mark(cprm->file, COREDUMP_MARK_REQACK); } @@ -1041,7 +1057,6 @@ static bool coredump_pipe(struct core_name *cn, struct coredump_params *cprm, static bool coredump_write(struct coredump_params *cprm, const struct linux_binfmt *binfmt) { - if (dump_interrupted()) { cprm->state |= COREDUMP_STATE_TRUNCATED; return true; @@ -1057,15 +1072,17 @@ static bool coredump_write(struct coredump_params *cprm, cprm->state |= COREDUMP_STATE_TRUNCATED; /* * Ensures that file size is big enough to contain the current - * file postion. This prevents gdb from complaining about + * file position. This prevents gdb from complaining about * a truncated file if the last "write" to the file was - * dump_skip. + * dump_skip. A record stream relies on it too: the flush + * emits the records that cover a trailing hole. */ if (cprm->to_skip) { cprm->to_skip--; if (!dump_emit(cprm, "", 1)) cprm->state |= COREDUMP_STATE_TRUNCATED; } + dump_end_record(cprm); file_end_write(cprm->file); free_vma_snapshot(cprm); return true; @@ -1080,6 +1097,7 @@ static void coredump_cleanup(struct core_name *cn, struct coredump_params *cprm) atomic_dec(&core_pipe_count); } kfree(cn->corename); + kfree(cprm->record_hdr); coredump_finish(cprm->state); } @@ -1208,26 +1226,74 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) * do on a core-file: use only these functions to write out all the * necessary info. */ -/* One write, never more than a page. See __dump_emit(). */ -static bool dump_emit_chunk(struct coredump_params *cprm, const void *addr, - int nr) +static bool dump_records(const struct coredump_params *cprm) +{ + return cprm->mask & COREDUMP_RECORDS; +} + +/* Describe the next @len bytes of the coredump. Returns the header size. */ +static size_t dump_record_init(struct coredump_params *cprm, + enum coredump_record_type type, u64 flags, + u64 len) +{ + if (!dump_records(cprm)) + return 0; + + *cprm->record_hdr = (struct coredump_record_header) { + .size = sizeof(*cprm->record_hdr), + .type = type, + .flags = flags, + .offset = cprm->pos, + .len = len, + }; + + return sizeof(*cprm->record_hdr); +} + +/* Write @iter whole or fail. @len is what it advances the coredump by. */ +static bool dump_write_iter(struct coredump_params *cprm, struct iov_iter *iter, + size_t len) { struct file *file = cprm->file; + size_t count = iov_iter_count(iter); loff_t pos = file->f_pos; ssize_t n; - if (dump_interrupted()) + n = __kernel_write_iter(file, iter, &pos); + if (n != (ssize_t)count) return false; + file->f_pos = pos; + cprm->written += count; + cprm->pos += len; + + return true; +} + +/* One record, never more than a page. See __dump_emit(). */ +static bool dump_emit_chunk(struct coredump_params *cprm, const void *addr, + int nr) +{ + struct kvec kvec[2]; + struct iov_iter iter; + unsigned int nseg = 0; + size_t hdrlen; - n = __kernel_write(file, addr, nr, &pos); - if (n != nr) + if (dump_interrupted()) return false; - file->f_pos = pos; - cprm->written += n; - cprm->pos += n; + hdrlen = dump_record_init(cprm, COREDUMP_RECORD_DATA, 0, nr); + if (hdrlen) { + kvec[nseg].iov_base = cprm->record_hdr; + kvec[nseg].iov_len = hdrlen; + nseg++; + } + kvec[nseg].iov_base = (void *)addr; + kvec[nseg].iov_len = nr; + nseg++; - return true; + iov_iter_kvec(&iter, ITER_SOURCE, kvec, nseg, hdrlen + nr); + + return dump_write_iter(cprm, &iter, nr); } static bool __dump_emit(struct coredump_params *cprm, const void *addr, int nr) @@ -1248,6 +1314,34 @@ static bool __dump_emit(struct coredump_params *cprm, const void *addr, int nr) return true; } +/* Send a record that stands on its own: a header and nothing else. */ +static bool dump_emit_record(struct coredump_params *cprm, + enum coredump_record_type type, u64 flags, u64 len) +{ + struct kvec kvec; + struct iov_iter iter; + size_t hdrlen; + + hdrlen = dump_record_init(cprm, type, flags, len); + if (!hdrlen) + return false; + + kvec.iov_base = cprm->record_hdr; + kvec.iov_len = hdrlen; + iov_iter_kvec(&iter, ITER_SOURCE, &kvec, 1, hdrlen); + + return dump_write_iter(cprm, &iter, len); +} + +/* Close the record stream. Only a whole coredump gets an end record. */ +static void dump_end_record(struct coredump_params *cprm) +{ + if (cprm->state & COREDUMP_STATE_TRUNCATED) + return; + + dump_emit_record(cprm, COREDUMP_RECORD_END, 0, 0); +} + static bool __dump_skip(struct coredump_params *cprm, size_t nr) { static char zeroes[PAGE_SIZE]; @@ -1308,11 +1402,10 @@ EXPORT_SYMBOL(dump_skip); #ifdef CONFIG_ELF_CORE static bool dump_emit_page(struct coredump_params *cprm, struct page *page) { - struct bio_vec bvec; + struct bio_vec bvec[2]; struct iov_iter iter; - struct file *file = cprm->file; - loff_t pos; - ssize_t n; + unsigned int nseg = 0; + size_t hdrlen; if (!page) return false; @@ -1323,17 +1416,16 @@ static bool dump_emit_page(struct coredump_params *cprm, struct page *page) return false; if (dump_interrupted()) return false; - pos = file->f_pos; - bvec_set_page(&bvec, page, PAGE_SIZE, 0); - iov_iter_bvec(&iter, ITER_SOURCE, &bvec, 1, PAGE_SIZE); - n = __kernel_write_iter(cprm->file, &iter, &pos); - if (n != PAGE_SIZE) - return false; - file->f_pos = pos; - cprm->written += PAGE_SIZE; - cprm->pos += PAGE_SIZE; - return true; + /* Hand the record header to the same write as the page it describes. */ + hdrlen = dump_record_init(cprm, COREDUMP_RECORD_DATA, 0, PAGE_SIZE); + if (hdrlen) + bvec_set_virt(&bvec[nseg++], cprm->record_hdr, hdrlen); + bvec_set_page(&bvec[nseg++], page, PAGE_SIZE, 0); + + iov_iter_bvec(&iter, ITER_SOURCE, bvec, nseg, hdrlen + PAGE_SIZE); + + return dump_write_iter(cprm, &iter, PAGE_SIZE); } /* diff --git a/include/linux/coredump.h b/include/linux/coredump.h index 709388dd565914..b252bb2843b337 100644 --- a/include/linux/coredump.h +++ b/include/linux/coredump.h @@ -6,6 +6,7 @@ #include #include #include +#include #include #ifdef CONFIG_COREDUMP @@ -40,7 +41,11 @@ struct coredump_params { u64 mask; /* COREDUMP_STATE_* raised while the coredump is written. */ enum coredump_state state; + /* Record header scratch, NULL unless the coredump is a record stream. */ + struct coredump_record_header *record_hdr; + /* Bytes handed to the file, record headers included. */ loff_t written; + /* Offset in the coredump, record headers excluded. */ loff_t pos; loff_t to_skip; int vma_count; diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 570fc2e005c206..1c8658f35735de 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -275,7 +275,7 @@ bool send_coredump_ack(int fd, const struct coredump_req *req, /* Every option the kernel is expected to advertise in coredump_req->mask. */ #define TEST_REQ_MASK_ALL \ (COREDUMP_KERNEL | COREDUMP_USERSPACE | \ - COREDUMP_REJECT | COREDUMP_WAIT) + COREDUMP_REJECT | COREDUMP_WAIT | COREDUMP_RECORDS) bool check_coredump_req(const struct coredump_req *req) { From 009111c54d5a8e3748a163bdb698421b52d7f266 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:34 +0200 Subject: [PATCH 271/857] coredump: describe the holes when COREDUMP_SPARSE is negotiated Make use of COREDUMP_SPARSE. Refuse it without COREDUMP_RECORDS. Actual holes are sent as a record with length indicating how much zero data there was. coredump_write() flushes a trailing hole if the coredump is done. Instead of writing the actual byte for pipes and sockets, collapse it. This stops wasting a header with coredump records for a single byte. So we now only write it when the coredump can be seeked. TL;DR a trailing hole is a zero record like any other and the records still cover the whole coredump. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-17-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 41 +++++++++++++++---- .../coredump/coredump_test_helpers.c | 3 +- 2 files changed, 35 insertions(+), 9 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index 412c7672cc9813..3b3721ea84af45 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -68,6 +68,7 @@ static bool dump_vma_snapshot(struct coredump_params *cprm); static void free_vma_snapshot(struct coredump_params *cprm); static void dump_end_record(struct coredump_params *cprm); +static bool dump_flush_skip(struct coredump_params *cprm); #define CORE_FILE_NOTE_SIZE_DEFAULT (4*1024*1024) /* Define a reasonable max cap */ @@ -806,7 +807,7 @@ static bool coredump_sock_request(struct core_name *cn, struct coredump_params * .size = sizeof(struct coredump_req), .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | COREDUMP_REJECT | COREDUMP_WAIT | - COREDUMP_RECORDS, + COREDUMP_RECORDS | COREDUMP_SPARSE, .size_ack = sizeof(struct coredump_ack), }; struct coredump_ack ack = {}; @@ -866,6 +867,12 @@ static bool coredump_sock_request(struct core_name *cn, struct coredump_params * return false; } + /* Zero records only exist inside a record stream. */ + if ((ack.mask & COREDUMP_SPARSE) && !(ack.mask & COREDUMP_RECORDS)) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_CONFLICTING); + return false; + } + /* Record header scratch; a bvec can't point at the stack. */ if (ack.mask & COREDUMP_RECORDS) { cprm->record_hdr = kmalloc_obj(*cprm->record_hdr); @@ -1071,15 +1078,21 @@ static bool coredump_write(struct coredump_params *cprm, if (!binfmt->core_dump(cprm)) cprm->state |= COREDUMP_STATE_TRUNCATED; /* - * Ensures that file size is big enough to contain the current - * file position. This prevents gdb from complaining about - * a truncated file if the last "write" to the file was - * dump_skip. A record stream relies on it too: the flush - * emits the records that cover a trailing hole. + * A trailing hole still has to land in the coredump. Seeking over + * it doesn't grow the file, so the last byte of it is written + * instead and gdb doesn't see a truncated file. Everything else + * puts the hole on the wire as it flushes it. */ if (cprm->to_skip) { - cprm->to_skip--; - if (!dump_emit(cprm, "", 1)) + bool flushed; + + if (cprm->file->f_mode & FMODE_LSEEK) { + cprm->to_skip--; + flushed = dump_emit(cprm, "", 1); + } else { + flushed = dump_flush_skip(cprm); + } + if (!flushed) cprm->state |= COREDUMP_STATE_TRUNCATED; } dump_end_record(cprm); @@ -1231,6 +1244,11 @@ static bool dump_records(const struct coredump_params *cprm) return cprm->mask & COREDUMP_RECORDS; } +static bool dump_sparse(const struct coredump_params *cprm) +{ + return cprm->mask & COREDUMP_SPARSE; +} + /* Describe the next @len bytes of the coredump. Returns the header size. */ static size_t dump_record_init(struct coredump_params *cprm, enum coredump_record_type type, u64 flags, @@ -1347,6 +1365,13 @@ static bool __dump_skip(struct coredump_params *cprm, size_t nr) static char zeroes[PAGE_SIZE]; struct file *file = cprm->file; + if (dump_sparse(cprm)) { + /* Hand the server the length of the hole instead of the hole itself. */ + if (dump_interrupted()) + return false; + return dump_emit_record(cprm, COREDUMP_RECORD_ZERO, 0, nr); + } + if (file->f_mode & FMODE_LSEEK) { if (dump_interrupted() || vfs_llseek(file, nr, SEEK_CUR) < 0) return false; diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 1c8658f35735de..a5b9cde47239b6 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -275,7 +275,8 @@ bool send_coredump_ack(int fd, const struct coredump_req *req, /* Every option the kernel is expected to advertise in coredump_req->mask. */ #define TEST_REQ_MASK_ALL \ (COREDUMP_KERNEL | COREDUMP_USERSPACE | \ - COREDUMP_REJECT | COREDUMP_WAIT | COREDUMP_RECORDS) + COREDUMP_REJECT | COREDUMP_WAIT | \ + COREDUMP_RECORDS | COREDUMP_SPARSE) bool check_coredump_req(const struct coredump_req *req) { From 038707264470ad160cbab44e98e7b6a08a98acac Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:35 +0200 Subject: [PATCH 272/857] selftests/coredump: test COREDUMP_RECORDS and COREDUMP_SPARSE Test the new COREDUMP_RECORDS and COREDUMP_SPARSE flags. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-18-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../coredump/coredump_socket_protocol_test.c | 407 ++++++++++++++++++ .../coredump/coredump_test_helpers.c | 236 +++++++++- .../coredump/coredump_test_helpers.h | 8 + 3 files changed, 650 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c index 60a357e628eb40..abf6e2c4c35473 100644 --- a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c +++ b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c @@ -1573,4 +1573,411 @@ TEST_F_TIMEOUT(coredump, socket_multiple_crashing_coredumps_epoll_workers, 500) wait_and_check_coredump_server(pid_coredump_server, _metadata, self); } +/* + * Reassemble a record stream and check that what comes out is an ELF + * core file. The records themselves are validated by recv_coredump_records(). + */ +TEST_F(coredump, socket_request_sparse_reassemble) +{ + int fd_core_file, pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + int ipc_sockets[2]; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_file = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + + close(ipc_sockets[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + fd_file = creat("/tmp/coredump.file", 0644); + if (fd_file < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + if (!send_coredump_ack(fd_coredump, &req, + COREDUMP_KERNEL | COREDUMP_RECORDS | + COREDUMP_SPARSE | COREDUMP_WAIT, 0)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + if (recv_coredump_records(fd_coredump, fd_file, NULL, NULL, -1) < 0) + goto out; + + exit_code = EXIT_SUCCESS; +out: + if (fd_file >= 0) + close(fd_file); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child(); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + + /* What the records reassemble into has to be an ELF core file. */ + fd_core_file = open("/tmp/coredump.file", O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd_core_file, 0); + ASSERT_TRUE(is_elf_core(fd_core_file)); + EXPECT_EQ(close(fd_core_file), 0); +} + +/* + * Crash a child with a mostly-unpopulated mapping and reassemble its + * record stream, reporting what crossed the socket and the coredump + * size the records describe. With @kill_peer the server kills the task + * once the coredump is under way so the kernel has to cut it short. + */ +static void check_record_dump(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, __u64 ack_mask, + bool kill_peer, ssize_t *received, + off_t *coredump_size) +{ + bool truncated = false; + int pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + int ipc_sockets[2]; + int pipefds[2]; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_EQ(pipe(pipefds), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_file = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + bool is_truncated = false; + off_t size = 0; + ssize_t ret; + + close(ipc_sockets[0]); + close(pipefds[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + /* + * The reassembled coredump is bigger than the mapping the + * child made, so keep it on the detached tmpfs and sparse. + */ + fd_file = open_coredump_tmpfile(self->fd_tmpfs_detached); + if (fd_file < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + if (!send_coredump_ack(fd_coredump, &req, ack_mask, 0)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + ret = recv_coredump_records(fd_coredump, fd_file, &size, &is_truncated, + kill_peer ? fd_peer_pidfd : -1); + if (ret < 0) + goto out; + + if (write_nointr(pipefds[1], &ret, sizeof(ret)) != sizeof(ret)) + goto out; + if (write_nointr(pipefds[1], &size, sizeof(size)) != sizeof(size)) + goto out; + if (write_nointr(pipefds[1], &is_truncated, + sizeof(is_truncated)) != sizeof(is_truncated)) + goto out; + + exit_code = EXIT_SUCCESS; +out: + close(pipefds[1]); + if (fd_file >= 0) + close(fd_file); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + EXPECT_EQ(close(pipefds[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_sparse(SPARSE_MAPPING_SIZE); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + + ASSERT_EQ(read_nointr(pipefds[0], received, sizeof(*received)), + sizeof(*received)); + ASSERT_EQ(read_nointr(pipefds[0], coredump_size, sizeof(*coredump_size)), + sizeof(*coredump_size)); + ASSERT_EQ(read_nointr(pipefds[0], &truncated, sizeof(truncated)), + sizeof(truncated)); + EXPECT_EQ(close(pipefds[0]), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + + if (kill_peer) { + /* The kernel gave up partway, so no end record closed the stream. */ + ASSERT_TRUE(truncated); + ASSERT_FALSE(WCOREDUMP(status)); + ASSERT_LT(*coredump_size, (off_t)SPARSE_MAPPING_SIZE); + return; + } + + ASSERT_FALSE(truncated); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + /* The mapping is in the coredump, holes included. */ + ASSERT_GT(*coredump_size, (off_t)SPARSE_MAPPING_SIZE); +} + +/* + * A mapping that has been written to is dumped whole, including the parts + * of it that were never faulted in. With COREDUMP_SPARSE the holes stay + * off the wire. + */ +TEST_F(coredump, socket_request_sparse_hole) +{ + off_t coredump_size = 0; + ssize_t received = 0; + + check_record_dump(_metadata, self, + COREDUMP_KERNEL | COREDUMP_RECORDS | + COREDUMP_SPARSE | COREDUMP_WAIT, + false, &received, &coredump_size); + + /* The holes didn't have to go over the socket. */ + ASSERT_LT(received, coredump_size / 8); +} + +/* + * COREDUMP_RECORDS alone splits the stream into records but elides + * nothing: the holes cross the socket as data records. + */ +TEST_F(coredump, socket_request_records_hole) +{ + off_t coredump_size = 0; + ssize_t received = 0; + + check_record_dump(_metadata, self, + COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_WAIT, + false, &received, &coredump_size); + + /* Records alone elide nothing, so everything crossed the socket. */ + ASSERT_GT(received, coredump_size); +} + +/* + * A coredump the kernel gives up on halfway still ends in an end record, + * and that record says the coredump is incomplete. COREDUMP_SPARSE is left + * out on purpose: the holes have to cross the socket so the coredump is + * far larger than the socket buffer and the kernel is still writing it + * when the kill lands. + */ +TEST_F(coredump, socket_request_records_truncated) +{ + off_t coredump_size = 0; + ssize_t received = 0; + + check_record_dump(_metadata, self, + COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_WAIT, + true, &received, &coredump_size); + + /* The end record crossed the socket even though the task was killed. */ + ASSERT_GT(received, 0); +} + +/* Ack @ack_mask, expect the kernel to refuse it as conflicting. */ +static void check_conflicting_ack(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, __u64 ack_mask) +{ + int pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + int ipc_sockets[2]; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + + close(ipc_sockets[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + if (!send_coredump_ack(fd_coredump, &req, ack_mask, 0)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_CONFLICTING)) + goto out; + + exit_code = EXIT_SUCCESS; +out: + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child(); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_FALSE(WCOREDUMP(status)); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); +} + +/* COREDUMP_RECORDS applies to a coredump the kernel writes, nothing else. */ +TEST_F(coredump, socket_request_records_without_kernel) +{ + check_conflicting_ack(_metadata, self, COREDUMP_USERSPACE | COREDUMP_RECORDS); +} + +/* A zero record can't exist outside a record stream. */ +TEST_F(coredump, socket_request_sparse_without_records) +{ + check_conflicting_ack(_metadata, self, COREDUMP_KERNEL | COREDUMP_SPARSE); +} + TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index a5b9cde47239b6..5b2ffe17f7b72f 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -1,9 +1,11 @@ // SPDX-License-Identifier: GPL-2.0 #include +#include #include #include #include +#include #include #include #include @@ -13,6 +15,7 @@ #include #include #include +#include #include #include #include @@ -23,6 +26,12 @@ #include "coredump_test_helpers.h" +#if __ELF_NATIVE_CLASS == 64 +#define COREDUMP_ELFCLASS ELFCLASS64 +#else +#define COREDUMP_ELFCLASS ELFCLASS32 +#endif + void *do_nothing(void *arg) { (void)arg; @@ -44,6 +53,228 @@ void crashing_child(void) i = *(volatile int *)NULL; } +void crashing_child_sparse(size_t size) +{ + char *p; + + /* + * Touch the first page only. The whole mapping is dumped because + * it has been written to, but all of it save that one page is a + * hole. + */ + p = mmap(NULL, size, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0); + if (p != MAP_FAILED) + p[0] = 'x'; + + /* crash on purpose */ + *(volatile int *)NULL = 0; +} + +/* Read @len bytes off the socket, writing them at @offset if @fd_out >= 0. */ +static ssize_t recv_record_bytes(int fd_coredump, __u64 len, int fd_out, + off_t offset) +{ + ssize_t received = 0; + + while (len) { + char buffer[PAGE_SIZE]; + size_t chunk = len < sizeof(buffer) ? len : sizeof(buffer); + ssize_t ret; + + ret = recv(fd_coredump, buffer, chunk, MSG_WAITALL); + if (ret <= 0) { + fprintf(stderr, "%s: short read %zd: %m\n", + __func__, ret); + return -1; + } + + if (fd_out >= 0 && + pwrite(fd_out, buffer, ret, offset + received) != ret) { + fprintf(stderr, "%s: pwrite failed: %m\n", __func__); + return -1; + } + + received += ret; + len -= ret; + } + + return received; +} + +/* + * Reassemble a record stream. If @fd_peer_pidfd is valid the task behind + * it is killed once a data record has arrived, so the kernel has to cut + * the coredump short with the stream already under way. + */ +ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, + off_t *coredump_size, bool *truncated, + int fd_peer_pidfd) +{ + ssize_t received = 0; + off_t size = 0; + bool is_truncated = false; + bool ended = false; + char trailing; + + while (!ended) { + struct coredump_record_header record = {}; + size_t known_size; + ssize_t ret; + + /* Peek the header size the way read_coredump_req() does. */ + ret = recv(fd_coredump, &record, sizeof(record.size), + MSG_PEEK | MSG_WAITALL); + if (ret == 0) { + /* Nothing closed the stream, so the coredump was cut short. */ + if (truncated) { + is_truncated = true; + break; + } + fprintf(stderr, "%s: stream ended without an end record\n", + __func__); + return -1; + } + if (ret != sizeof(record.size)) { + fprintf(stderr, "%s: short record peek %zd: %m\n", + __func__, ret); + return -1; + } + + if (record.size < COREDUMP_RECORD_HEADER_SIZE_VER0) { + fprintf(stderr, "%s: header size %u below minimum %u\n", + __func__, record.size, + COREDUMP_RECORD_HEADER_SIZE_VER0); + return -1; + } + + /* Consume as much of the header as we know about. */ + known_size = record.size < sizeof(record) ? record.size : sizeof(record); + ret = recv(fd_coredump, &record, known_size, MSG_WAITALL); + if (ret != (ssize_t)known_size) { + fprintf(stderr, "%s: short record read %zd: %m\n", + __func__, ret); + return -1; + } + received += ret; + + /* + * A flag changes what the record means, so refuse one we + * don't know rather than guess. + */ + if (record.flags) { + fprintf(stderr, "%s: unknown header flags 0x%llx\n", + __func__, (unsigned long long)record.flags); + return -1; + } + + /* Discard any part of the header we have no use for. */ + ret = recv_record_bytes(fd_coredump, record.size - known_size, -1, 0); + if (ret < 0) + return -1; + received += ret; + + /* Records are sent in order and they don't leave gaps. */ + if (record.offset != (__u64)size) { + fprintf(stderr, "%s: record at %llu, expected %llu\n", + __func__, (unsigned long long)record.offset, + (unsigned long long)size); + return -1; + } + + switch (record.type) { + case COREDUMP_RECORD_ZERO: + /* A hole. It comes with no data and needs none. */ + break; + case COREDUMP_RECORD_DATA: + ret = recv_record_bytes(fd_coredump, record.len, + fd_core_file, size); + if (ret < 0) + return -1; + received += ret; + if (fd_peer_pidfd >= 0) { + if (sys_pidfd_send_signal(fd_peer_pidfd, SIGKILL, + NULL, 0)) { + fprintf(stderr, "%s: kill failed: %m\n", + __func__); + return -1; + } + fd_peer_pidfd = -1; + } + break; + case COREDUMP_RECORD_END: + /* The coredump ends here and nothing follows it. */ + if (record.len) { + fprintf(stderr, "%s: end record covers %llu bytes\n", + __func__, + (unsigned long long)record.len); + return -1; + } + ended = true; + break; + default: + fprintf(stderr, "%s: unknown record type %u\n", + __func__, record.type); + return -1; + } + + size += record.len; + } + + /* The end record is the last thing on the wire. */ + if (recv(fd_coredump, &trailing, sizeof(trailing), MSG_DONTWAIT) > 0) { + fprintf(stderr, "%s: data after the end record\n", __func__); + return -1; + } + + if (truncated) + *truncated = is_truncated; + + /* + * Nothing is written for a hole, so grow the file to the size the + * records describe in case the coredump ended in one. + */ + if (ftruncate(fd_core_file, size) < 0) { + fprintf(stderr, "%s: ftruncate to %llu failed: %m\n", + __func__, (unsigned long long)size); + return -1; + } + + if (coredump_size) + *coredump_size = size; + + fprintf(stderr, "Received %zd bytes for a %s coredump of %llu bytes\n", + received, is_truncated ? "truncated" : "complete", + (unsigned long long)size); + return received; +} + +/* The ELF header of a native core file. */ +static bool is_core_ehdr(const ElfW(Ehdr) *ehdr) +{ + return !memcmp(ehdr->e_ident, ELFMAG, SELFMAG) && + ehdr->e_ident[EI_CLASS] == COREDUMP_ELFCLASS && + ehdr->e_type == ET_CORE; +} + +/* Whatever the server ends up with has to be an ELF core file. */ +bool is_elf_core(int fd) +{ + ElfW(Ehdr) ehdr; + + if (pread(fd, &ehdr, sizeof(ehdr), 0) != sizeof(ehdr)) { + fprintf(stderr, "%s: short read: %m\n", __func__); + return false; + } + + if (!is_core_ehdr(&ehdr)) { + fprintf(stderr, "%s: not an ELF core file\n", __func__); + return false; + } + + return true; +} + int create_detached_tmpfs(void) { int fd_context, fd_tmpfs; @@ -86,6 +317,7 @@ int create_and_listen_unix_socket(const char *path) return fd; out: + fprintf(stderr, "%s: %s: %m\n", __func__, path); if (fd >= 0) close(fd); return -1; @@ -264,8 +496,10 @@ bool send_coredump_ack(int fd, const struct coredump_req *req, large_ack.ack.mask = mask; large_ack.ack.size = size_ack; ret = send(fd, &large_ack, size_ack, MSG_NOSIGNAL); - if (ret != size_ack) + if (ret != size_ack) { + fprintf(stderr, "%s: short send %zd: %m\n", __func__, ret); return false; + } fprintf(stderr, "Sent coredump ack with size %zu and mask 0x%llx\n", size_ack, (unsigned long long)mask); diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.h b/tools/testing/selftests/coredump/coredump_test_helpers.h index 45904bd177b802..fe0a88a71b0510 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.h +++ b/tools/testing/selftests/coredump/coredump_test_helpers.h @@ -15,9 +15,17 @@ #define NUM_THREAD_SPAWN 128 +/* Size of the mostly unpopulated mapping the sparse coredump test maps. */ +#define SPARSE_MAPPING_SIZE (256 * 1024 * 1024) + /* Shared helper function declarations */ void *do_nothing(void *arg); void crashing_child(void); +void crashing_child_sparse(size_t size); +ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, + off_t *coredump_size, bool *truncated, + int fd_peer_pidfd); +bool is_elf_core(int fd); int create_detached_tmpfs(void); int create_and_listen_unix_socket(const char *path); bool set_core_pattern(const char *pattern); From 70bd923cbf0feea40907572aafbd96da01ad52ea Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:36 +0200 Subject: [PATCH 273/857] selftests/coredump: hand the record stream to a sink Currently recv_coredump_records() parses the record stream and dumps it into a file. A coredump server may want to process the data it gets. So split the parsing from the processing. No functional changes. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-19-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../coredump/coredump_test_helpers.c | 91 +++++++++++++++---- 1 file changed, 72 insertions(+), 19 deletions(-) diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 5b2ffe17f7b72f..45d76fa0f469aa 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -71,9 +71,19 @@ void crashing_child_sparse(size_t size) *(volatile int *)NULL = 0; } -/* Read @len bytes off the socket, writing them at @offset if @fd_out >= 0. */ -static ssize_t recv_record_bytes(int fd_coredump, __u64 len, int fd_out, - off_t offset) +/* Sink a reassembled record stream is handed to, record by record. */ +struct coredump_record_sink { + /* @len bytes of coredump data that belong at @offset. */ + int (*data)(void *ctx, const void *buf, size_t len, __u64 offset); + /* @len zero bytes that belong at @offset. */ + int (*zero)(void *ctx, __u64 offset, __u64 len); + void *ctx; +}; + +/* Read @len bytes off the socket and hand them to @sink, if there is one. */ +static ssize_t recv_record_bytes(int fd_coredump, __u64 len, + const struct coredump_record_sink *sink, + __u64 offset) { ssize_t received = 0; @@ -89,11 +99,8 @@ static ssize_t recv_record_bytes(int fd_coredump, __u64 len, int fd_out, return -1; } - if (fd_out >= 0 && - pwrite(fd_out, buffer, ret, offset + received) != ret) { - fprintf(stderr, "%s: pwrite failed: %m\n", __func__); + if (sink && sink->data(sink->ctx, buffer, ret, offset + received)) return -1; - } received += ret; len -= ret; @@ -102,14 +109,34 @@ static ssize_t recv_record_bytes(int fd_coredump, __u64 len, int fd_out, return received; } +/* Put the data where the records say it goes and leave the holes alone. */ +static int file_sink_data(void *ctx, const void *buf, size_t len, __u64 offset) +{ + int fd = *(int *)ctx; + + if (pwrite(fd, buf, len, offset) != (ssize_t)len) { + fprintf(stderr, "%s: pwrite failed: %m\n", __func__); + return -1; + } + + return 0; +} + +static int file_sink_zero(void *ctx, __u64 offset, __u64 len) +{ + /* Nothing has to be written for a hole. */ + return 0; +} + /* - * Reassemble a record stream. If @fd_peer_pidfd is valid the task behind - * it is killed once a data record has arrived, so the kernel has to cut - * the coredump short with the stream already under way. + * Read a coredump strea and funnel it into @sink. Allow to pass in a + * @fd_peer_pidfd to simulate coredump truncation by killing it after having + * received a coredump record. */ -ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, - off_t *coredump_size, bool *truncated, - int fd_peer_pidfd) +static ssize_t __recv_coredump_records(int fd_coredump, + const struct coredump_record_sink *sink, + off_t *coredump_size, bool *truncated, + int fd_peer_pidfd) { ssize_t received = 0; off_t size = 0; @@ -169,7 +196,8 @@ ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, } /* Discard any part of the header we have no use for. */ - ret = recv_record_bytes(fd_coredump, record.size - known_size, -1, 0); + ret = recv_record_bytes(fd_coredump, record.size - known_size, + NULL, 0); if (ret < 0) return -1; received += ret; @@ -185,10 +213,12 @@ ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, switch (record.type) { case COREDUMP_RECORD_ZERO: /* A hole. It comes with no data and needs none. */ + if (sink->zero(sink->ctx, record.offset, record.len)) + return -1; break; case COREDUMP_RECORD_DATA: - ret = recv_record_bytes(fd_coredump, record.len, - fd_core_file, size); + ret = recv_record_bytes(fd_coredump, record.len, sink, + record.offset); if (ret < 0) return -1; received += ret; @@ -230,6 +260,32 @@ ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, if (truncated) *truncated = is_truncated; + *coredump_size = size; + + fprintf(stderr, "Received %zd bytes for a %s coredump of %llu bytes\n", + received, is_truncated ? "truncated" : "complete", + (unsigned long long)size); + return received; +} + +/* Reassemble a record stream into the coredump it describes. */ +ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, + off_t *coredump_size, bool *truncated, + int fd_peer_pidfd) +{ + struct coredump_record_sink sink = { + .data = file_sink_data, + .zero = file_sink_zero, + .ctx = &fd_core_file, + }; + ssize_t received; + off_t size = 0; + + received = __recv_coredump_records(fd_coredump, &sink, &size, truncated, + fd_peer_pidfd); + if (received < 0) + return -1; + /* * Nothing is written for a hole, so grow the file to the size the * records describe in case the coredump ended in one. @@ -243,9 +299,6 @@ ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, if (coredump_size) *coredump_size = size; - fprintf(stderr, "Received %zd bytes for a %s coredump of %llu bytes\n", - received, is_truncated ? "truncated" : "complete", - (unsigned long long)size); return received; } From 269932a2a7a0296ed368f8aa95c72d03270af617 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:37 +0200 Subject: [PATCH 274/857] selftests/coredump: put a hole in the middle of a sparse mapping The crashing_child_sparse() helper touches the first page of the mapping. That forces everything behind it to be a trailing hole. This is easy to handle. Make the test more difficult meaningful by also touchin the last page. This causes the hole to sit between two populated pages. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-20-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../testing/selftests/coredump/coredump_test_helpers.c | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 45d76fa0f469aa..89f3954c560742 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -58,14 +58,16 @@ void crashing_child_sparse(size_t size) char *p; /* - * Touch the first page only. The whole mapping is dumped because - * it has been written to, but all of it save that one page is a - * hole. + * Touch the first and the last page. This will cause the whole mapping + * to be dumped because it has been written to. Everything between + * those two pages is a hole though. */ p = mmap(NULL, size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0); - if (p != MAP_FAILED) + if (p != MAP_FAILED) { p[0] = 'x'; + p[size - 1] = 'x'; + } /* crash on purpose */ *(volatile int *)NULL = 0; From 7e1079a8a37485c79d7f744aefeea32e268f580b Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:38 +0200 Subject: [PATCH 275/857] selftests/coredump: simulate a blob store A coredump server that uploads to a blob store must redescribe the coredump and fixup the phdr. A segment is split wherever a hole was left out and everything a segment covers past p_filesz is zeroes anyway. So the blob store ends up with an ordinary ELF core file that is missing nothing but holes. Nothing downstream of the server has to learn a container format. The coredump with its holes still in it is reassembled alongside the object so the two can be compared. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-21-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../coredump/coredump_socket_protocol_test.c | 156 ++++ .../coredump/coredump_test_helpers.c | 765 ++++++++++++++++++ .../coredump/coredump_test_helpers.h | 3 + 3 files changed, 924 insertions(+) diff --git a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c index abf6e2c4c35473..f33eaf2fa93d3e 100644 --- a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c +++ b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c @@ -1882,6 +1882,162 @@ TEST_F(coredump, socket_request_records_truncated) ASSERT_GT(received, 0); } +/* + * A coredump server that uploads to a blob store can't upload a sparse + * file. It doesn't have to: it streams the data records into the object + * as they arrive, leaves the holes out, and uploads the corrected + * program header table last. What it ends up with is an ordinary ELF + * core file that describes the same memory as the coredump the records + * came from, minus the holes. + */ +TEST_F(coredump, socket_request_sparse_blob_upload) +{ + int fd_core_file, pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + off_t coredump_size = 0; + ssize_t received = 0; + int ipc_sockets[2]; + int pipefds[2]; + struct stat st; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_EQ(pipe(pipefds), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_object = -1, fd_reference = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + off_t size = 0; + ssize_t ret; + + close(ipc_sockets[0]); + close(pipefds[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + /* The object is a plain file. It never sees a hole. */ + fd_object = open("/tmp/coredump.file", + O_RDWR | O_CREAT | O_TRUNC | O_CLOEXEC, 0600); + if (fd_object < 0) + goto out; + + /* + * The coredump with its holes still in it is bigger than + * the mapping the child made, so keep it on the detached + * tmpfs and sparse. + */ + fd_reference = open_coredump_tmpfile(self->fd_tmpfs_detached); + if (fd_reference < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + if (!send_coredump_ack(fd_coredump, &req, + COREDUMP_KERNEL | COREDUMP_RECORDS | + COREDUMP_SPARSE | COREDUMP_WAIT, 0)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + ret = recv_coredump_compact(fd_coredump, fd_object, + fd_reference, &size); + if (ret < 0) + goto out; + + if (check_compact_coredump(fd_object, fd_reference)) + goto out; + + if (write_nointr(pipefds[1], &ret, sizeof(ret)) != sizeof(ret)) + goto out; + if (write_nointr(pipefds[1], &size, sizeof(size)) != sizeof(size)) + goto out; + + exit_code = EXIT_SUCCESS; +out: + close(pipefds[1]); + if (fd_reference >= 0) + close(fd_reference); + if (fd_object >= 0) + close(fd_object); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + EXPECT_EQ(close(pipefds[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_sparse(SPARSE_MAPPING_SIZE); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_EQ(read_nointr(pipefds[0], &received, sizeof(received)), + sizeof(received)); + ASSERT_EQ(read_nointr(pipefds[0], &coredump_size, sizeof(coredump_size)), + sizeof(coredump_size)); + EXPECT_EQ(close(pipefds[0]), 0); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + + /* The mapping is in the coredump, holes included. */ + ASSERT_GT(coredump_size, (off_t)SPARSE_MAPPING_SIZE); + + /* The object isn't sparse and doesn't carry them. */ + ASSERT_EQ(stat("/tmp/coredump.file", &st), 0); + ASSERT_LT(st.st_size, coredump_size / 8); + + /* And a debugger still sees an ordinary ELF core file. */ + fd_core_file = open("/tmp/coredump.file", O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd_core_file, 0); + ASSERT_TRUE(is_elf_core(fd_core_file)); + EXPECT_EQ(close(fd_core_file), 0); +} + /* Ack @ack_mask, expect the kernel to refuse it as conflicting. */ static void check_conflicting_ack(struct __test_metadata *const _metadata, FIXTURE_DATA(coredump) *self, __u64 ack_mask) diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 89f3954c560742..9346b8f688e260 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -330,6 +330,771 @@ bool is_elf_core(int fd) return true; } +/* + * A coredump server that uploads to a blob store can't upload a sparse + * file and can't seek in the object it is uploading. It streams the data + * records into the object as they arrive, remembers the holes it left + * out, and uploads the program header table that describes the result + * last. What comes out is an ordinary ELF core file without the holes. + */ + +/* A run of the coredump the object doesn't carry. */ +struct compact_hole { + __u64 offset; + __u64 len; +}; + +/* A program header of the object and where its bytes sat in the coredump. */ +struct compact_piece { + ElfW(Phdr) phdr; + __u64 src; +}; + +struct compact_ctx { + int fd_body; /* the object's payload, append only */ + int fd_reference; /* the coredump with its holes, for the test */ + unsigned char *head; /* everything ahead of the segment data */ + size_t head_len; + size_t head_cap; + __u64 data_offset; /* where the segment data starts, 0 while unknown */ + struct compact_hole *holes; + size_t nr_holes; + size_t holes_cap; + __u64 body_len; +}; + +/* Write @len bytes out, short writes and all. */ +static int compact_write(int fd, const void *buf, size_t len) +{ + const unsigned char *pos = buf; + + while (len) { + ssize_t ret = write(fd, pos, len); + + if (ret <= 0) { + fprintf(stderr, "%s: write failed: %m\n", __func__); + return -1; + } + + pos += ret; + len -= ret; + } + + return 0; +} + +/* Keep @len bytes of the head, or @len zeroes if @buf is NULL. */ +static int compact_head_append(struct compact_ctx *ctx, const void *buf, + size_t len) +{ + if (ctx->head_len + len > ctx->head_cap) { + size_t cap = ctx->head_cap ? ctx->head_cap : PAGE_SIZE; + unsigned char *head; + + while (cap < ctx->head_len + len) + cap *= 2; + + head = realloc(ctx->head, cap); + if (!head) { + fprintf(stderr, "%s: out of memory\n", __func__); + return -1; + } + ctx->head = head; + ctx->head_cap = cap; + } + + if (buf) + memcpy(ctx->head + ctx->head_len, buf, len); + else + memset(ctx->head + ctx->head_len, 0, len); + ctx->head_len += len; + + return 0; +} + +/* Remember a hole so the program header table can account for it later. */ +static int compact_keep_hole(struct compact_ctx *ctx, __u64 offset, __u64 len) +{ + if (ctx->nr_holes == ctx->holes_cap) { + size_t cap = ctx->holes_cap ? ctx->holes_cap * 2 : 64; + struct compact_hole *holes; + + holes = realloc(ctx->holes, cap * sizeof(*holes)); + if (!holes) { + fprintf(stderr, "%s: out of memory\n", __func__); + return -1; + } + ctx->holes = holes; + ctx->holes_cap = cap; + } + + ctx->holes[ctx->nr_holes].offset = offset; + ctx->holes[ctx->nr_holes].len = len; + ctx->nr_holes++; + + return 0; +} + +/* The segment data starts where the first PT_LOAD points. */ +static int compact_probe(struct compact_ctx *ctx) +{ + const ElfW(Ehdr) *ehdr = (const ElfW(Ehdr) *)ctx->head; + const ElfW(Phdr) *phdr; + size_t i; + + if (ctx->data_offset || ctx->head_len < sizeof(*ehdr)) + return 0; + + if (!is_core_ehdr(ehdr)) { + fprintf(stderr, "%s: not an ELF core file\n", __func__); + return -1; + } + + if (ehdr->e_phoff != sizeof(*ehdr) || + ehdr->e_phentsize != sizeof(ElfW(Phdr)) || + ehdr->e_phnum == 0 || ehdr->e_phnum == PN_XNUM) { + fprintf(stderr, "%s: unhandled program header table\n", __func__); + return -1; + } + + if (ctx->head_len < ehdr->e_phoff + + (size_t)ehdr->e_phnum * ehdr->e_phentsize) + return 0; + + phdr = (const ElfW(Phdr) *)(ctx->head + ehdr->e_phoff); + for (i = 0; i < ehdr->e_phnum; i++) { + if (phdr[i].p_type != PT_LOAD) + continue; + if (!ctx->data_offset || phdr[i].p_offset < ctx->data_offset) + ctx->data_offset = phdr[i].p_offset; + } + + if (!ctx->data_offset) { + fprintf(stderr, "%s: coredump without a single segment\n", + __func__); + return -1; + } + + return 0; +} + +/* + * Take whatever of [@offset, @offset + @len) still belongs to the head. + * @buf is NULL for a hole. Returns how much was taken. + */ +static ssize_t compact_head_take(struct compact_ctx *ctx, const void *buf, + __u64 offset, __u64 len) +{ + __u64 chunk; + + if (!len || (ctx->data_offset && offset >= ctx->data_offset)) + return 0; + + chunk = len; + if (ctx->data_offset && offset + chunk > ctx->data_offset) + chunk = ctx->data_offset - offset; + + if (offset != ctx->head_len) { + fprintf(stderr, "%s: head has a gap at %llu\n", __func__, + (unsigned long long)offset); + return -1; + } + + if (compact_head_append(ctx, buf, chunk)) + return -1; + + return chunk; +} + +static int compact_data(void *arg, const void *buf, size_t len, __u64 offset) +{ + struct compact_ctx *ctx = arg; + const unsigned char *pos = buf; + ssize_t head; + + /* Only the test needs a coredump with the holes still in it. */ + if (pwrite(ctx->fd_reference, pos, len, offset) != (ssize_t)len) { + fprintf(stderr, "%s: pwrite failed: %m\n", __func__); + return -1; + } + + /* The head has to be rewritten at the end, so hold on to it. */ + head = compact_head_take(ctx, pos, offset, len); + if (head < 0) + return -1; + if (head && compact_probe(ctx)) + return -1; + + pos += head; + len -= head; + if (!len) + return 0; + + /* Everything else goes into the object as it arrives. */ + if (compact_write(ctx->fd_body, pos, len)) + return -1; + ctx->body_len += len; + + return 0; +} + +static int compact_zero(void *arg, __u64 offset, __u64 len) +{ + struct compact_ctx *ctx = arg; + ssize_t head; + + /* A hole in the head is alignment padding. Write it out. */ + head = compact_head_take(ctx, NULL, offset, len); + if (head < 0) + return -1; + + offset += head; + len -= head; + if (!len) + return 0; + + /* This is what the object doesn't have to carry. */ + return compact_keep_hole(ctx, offset, len); +} + +/* Where @offset ends up in the object once the holes ahead of it are gone. */ +static __u64 compact_offset(const struct compact_ctx *ctx, __u64 body_start, + __u64 offset) +{ + __u64 elided = 0; + size_t i; + + for (i = 0; i < ctx->nr_holes; i++) { + __u64 len = ctx->holes[i].len; + + if (ctx->holes[i].offset >= offset) + break; + if (ctx->holes[i].offset + len > offset) + len = offset - ctx->holes[i].offset; + elided += len; + } + + return body_start + (offset - ctx->data_offset) - elided; +} + +/* A run of segment data that made it into the object. */ +static void compact_add_data(struct compact_piece *pieces, size_t *nr, + const ElfW(Phdr) *phdr, __u64 start, __u64 end) +{ + struct compact_piece *piece = &pieces[(*nr)++]; + + piece->phdr = *phdr; + piece->phdr.p_vaddr = phdr->p_vaddr + (start - phdr->p_offset); + piece->phdr.p_paddr = 0; + piece->phdr.p_filesz = end - start; + piece->phdr.p_memsz = end - start; + piece->src = start; +} + +/* + * A run of @len bytes the object doesn't carry. It grows the piece in + * front of it if this segment already has one, because everything a + * segment covers past p_filesz is zeroes anyway. + */ +static void compact_add_zero(struct compact_piece *pieces, size_t *nr, + size_t first, const ElfW(Phdr) *phdr, __u64 vaddr, + __u64 len) +{ + struct compact_piece *piece; + + if (*nr > first) { + pieces[*nr - 1].phdr.p_memsz += len; + return; + } + + piece = &pieces[(*nr)++]; + piece->phdr = *phdr; + piece->phdr.p_vaddr = vaddr; + piece->phdr.p_paddr = 0; + piece->phdr.p_filesz = 0; + piece->phdr.p_memsz = len; + piece->src = 0; +} + +/* Split the segments at the holes and write out what the object became. */ +static int compact_build(struct compact_ctx *ctx, int fd_object) +{ + __u64 note_offset = 0, note_len = 0, note_new; + __u64 align = 0, head_len, body_start, pos; + size_t nr_old, nr_new = 0, note_piece = 0, i; + struct compact_piece *pieces; + char buffer[PAGE_SIZE]; + const ElfW(Phdr) *old; + ElfW(Ehdr) ehdr; + int ret = -1; + + if (!ctx->data_offset) { + fprintf(stderr, "%s: coredump without segment data\n", __func__); + return -1; + } + + memcpy(&ehdr, ctx->head, sizeof(ehdr)); + if (ehdr.e_shoff) { + fprintf(stderr, "%s: section headers are not handled\n", + __func__); + return -1; + } + + old = (const ElfW(Phdr) *)(ctx->head + ehdr.e_phoff); + nr_old = ehdr.e_phnum; + + pieces = calloc(nr_old + 2 * ctx->nr_holes + 1, sizeof(*pieces)); + if (!pieces) { + fprintf(stderr, "%s: out of memory\n", __func__); + return -1; + } + + for (i = 0; i < nr_old; i++) { + ElfW(Phdr) phdr = old[i]; + __u64 end = phdr.p_offset + phdr.p_filesz; + __u64 cur = phdr.p_offset; + size_t first = nr_new, h; + + /* The notes move because the table in front of them grows. */ + if (phdr.p_type == PT_NOTE) { + if (note_len) { + fprintf(stderr, "%s: more than one note segment\n", + __func__); + goto out; + } + note_offset = phdr.p_offset; + note_len = phdr.p_filesz; + note_piece = nr_new; + pieces[nr_new].phdr = phdr; + pieces[nr_new++].src = 0; + continue; + } + + if (phdr.p_type != PT_LOAD) { + if (phdr.p_filesz && phdr.p_offset < ctx->data_offset) { + fprintf(stderr, "%s: segment %zu is in the head\n", + __func__, i); + goto out; + } + pieces[nr_new].phdr = phdr; + pieces[nr_new++].src = phdr.p_offset; + continue; + } + + if (!align) + align = phdr.p_align; + + for (h = 0; h < ctx->nr_holes && cur < end; h++) { + __u64 start = ctx->holes[h].offset; + __u64 stop = start + ctx->holes[h].len; + + if (stop <= cur) + continue; + if (start >= end) + break; + + /* A hole can span more than this one segment. */ + if (start < cur) + start = cur; + if (stop > end) + stop = end; + + if (start > cur) { + compact_add_data(pieces, &nr_new, &phdr, cur, + start); + cur = start; + } + compact_add_zero(pieces, &nr_new, first, &phdr, + phdr.p_vaddr + (cur - phdr.p_offset), + stop - cur); + cur = stop; + } + + if (cur < end) + compact_add_data(pieces, &nr_new, &phdr, cur, end); + + /* Whatever the kernel didn't dump of this mapping. */ + if (phdr.p_memsz > phdr.p_filesz) + compact_add_zero(pieces, &nr_new, first, &phdr, + phdr.p_vaddr + phdr.p_filesz, + phdr.p_memsz - phdr.p_filesz); + } + + if (!note_len || note_offset + note_len > ctx->head_len) { + fprintf(stderr, "%s: notes aren't where they should be\n", + __func__); + goto out; + } + + if (nr_new >= PN_XNUM) { + fprintf(stderr, "%s: %zu program headers don't fit\n", __func__, + nr_new); + goto out; + } + + if (!align || (align & (align - 1))) + align = sysconf(_SC_PAGESIZE); + + note_new = sizeof(ehdr) + (__u64)nr_new * sizeof(ElfW(Phdr)); + head_len = note_new + note_len; + body_start = (head_len + align - 1) & ~(align - 1); + + for (i = 0; i < nr_new; i++) { + struct compact_piece *piece = &pieces[i]; + + if (i == note_piece) + piece->phdr.p_offset = note_new; + else if (piece->phdr.p_filesz) + piece->phdr.p_offset = compact_offset(ctx, body_start, + piece->src); + else + piece->phdr.p_offset = 0; + } + + /* Only now is the head known. That's why it is uploaded last. */ + ehdr.e_phnum = nr_new; + if (compact_write(fd_object, &ehdr, sizeof(ehdr))) + goto out; + + for (i = 0; i < nr_new; i++) + if (compact_write(fd_object, &pieces[i].phdr, + sizeof(pieces[i].phdr))) + goto out; + + if (compact_write(fd_object, ctx->head + note_offset, note_len)) + goto out; + + /* Keep the segments aligned the way a debugger expects them. */ + memset(buffer, 0, sizeof(buffer)); + for (pos = head_len; pos < body_start; ) { + __u64 chunk = body_start - pos; + + if (chunk > sizeof(buffer)) + chunk = sizeof(buffer); + if (compact_write(fd_object, buffer, chunk)) + goto out; + pos += chunk; + } + + /* Putting the parts together is the blob store's job. Do it here. */ + for (pos = 0; pos < ctx->body_len; ) { + ssize_t chunk = pread(ctx->fd_body, buffer, sizeof(buffer), pos); + + if (chunk <= 0) { + fprintf(stderr, "%s: short read %zd: %m\n", __func__, + chunk); + goto out; + } + if (compact_write(fd_object, buffer, chunk)) + goto out; + pos += chunk; + } + + fprintf(stderr, "Object is %llu bytes in %zu program headers, %zu holes left out\n", + (unsigned long long)(body_start + ctx->body_len), nr_new, + ctx->nr_holes); + ret = 0; +out: + free(pieces); + return ret; +} + +/* + * Reassemble a record stream into an ELF core file that has no holes in + * it, the way a coredump server that uploads to a blob store has to. If + * @fd_reference is valid it gets the coredump the records describe, + * holes and all, so the test can compare the two. + */ +ssize_t recv_coredump_compact(int fd_coredump, int fd_object, int fd_reference, + off_t *coredump_size) +{ + struct compact_ctx ctx = { + .fd_body = -1, + .fd_reference = fd_reference, + }; + struct coredump_record_sink sink = { + .data = compact_data, + .zero = compact_zero, + .ctx = &ctx, + }; + ssize_t received; + off_t size = 0; + FILE *body; + + body = tmpfile(); + if (!body) { + fprintf(stderr, "%s: tmpfile failed: %m\n", __func__); + return -1; + } + ctx.fd_body = fileno(body); + + /* An upload is appended to. Make sure nothing here can seek. */ + if (fcntl(ctx.fd_body, F_SETFL, O_APPEND)) { + fprintf(stderr, "%s: F_SETFL failed: %m\n", __func__); + received = -1; + goto out; + } + + received = __recv_coredump_records(fd_coredump, &sink, &size, NULL, -1); + if (received < 0) + goto out; + + /* + * Nothing is written for a hole, so grow the reference to the size + * the records describe in case the coredump ended in one. + */ + if (ftruncate(fd_reference, size) < 0) { + fprintf(stderr, "%s: ftruncate to %llu failed: %m\n", + __func__, (unsigned long long)size); + received = -1; + goto out; + } + + if (compact_build(&ctx, fd_object)) { + received = -1; + goto out; + } + + if (coredump_size) + *coredump_size = size; +out: + fclose(body); + free(ctx.head); + free(ctx.holes); + return received; +} + +/* Read the ELF header and the program header table of @fd. */ +static ElfW(Phdr) *read_phdrs(int fd, size_t *nr) +{ + ElfW(Ehdr) ehdr; + ElfW(Phdr) *phdr; + size_t size; + + if (pread(fd, &ehdr, sizeof(ehdr), 0) != sizeof(ehdr)) { + fprintf(stderr, "%s: no ELF header: %m\n", __func__); + return NULL; + } + + if (!is_core_ehdr(&ehdr) || !ehdr.e_phnum || + ehdr.e_phentsize != sizeof(*phdr)) { + fprintf(stderr, "%s: not an ELF core file\n", __func__); + return NULL; + } + + size = (size_t)ehdr.e_phnum * ehdr.e_phentsize; + phdr = malloc(size); + if (!phdr) { + fprintf(stderr, "%s: out of memory\n", __func__); + return NULL; + } + + if (pread(fd, phdr, size, ehdr.e_phoff) != (ssize_t)size) { + fprintf(stderr, "%s: short program header table: %m\n", __func__); + free(phdr); + return NULL; + } + + *nr = ehdr.e_phnum; + return phdr; +} + +/* The segment @vaddr falls into. */ +static const ElfW(Phdr) *find_segment(const ElfW(Phdr) *phdr, size_t nr, + __u64 vaddr) +{ + size_t i; + + for (i = 0; i < nr; i++) { + if (phdr[i].p_type != PT_LOAD) + continue; + if (vaddr >= phdr[i].p_vaddr && + vaddr < phdr[i].p_vaddr + phdr[i].p_memsz) + return &phdr[i]; + } + + return NULL; +} + +/* The next stretch of memory the segments cover, split ones merged back. */ +static bool next_range(const ElfW(Phdr) *phdr, size_t nr, size_t *i, + __u64 *start, __u64 *end) +{ + while (*i < nr && phdr[*i].p_type != PT_LOAD) + (*i)++; + + if (*i >= nr) + return false; + + *start = phdr[*i].p_vaddr; + *end = phdr[*i].p_vaddr + phdr[*i].p_memsz; + (*i)++; + + while (*i < nr) { + if (phdr[*i].p_type != PT_LOAD) { + (*i)++; + continue; + } + if (phdr[*i].p_vaddr != *end) + break; + *end = phdr[*i].p_vaddr + phdr[*i].p_memsz; + (*i)++; + } + + return true; +} + +/* Compare @len bytes at @offset against @len bytes at @offset_ref. */ +static int compare_range(int fd, __u64 offset, int fd_ref, __u64 offset_ref, + __u64 len) +{ + char buffer[PAGE_SIZE], buffer_ref[PAGE_SIZE]; + + while (len) { + size_t chunk = len < sizeof(buffer) ? len : sizeof(buffer); + + if (pread(fd, buffer, chunk, offset) != (ssize_t)chunk || + pread(fd_ref, buffer_ref, chunk, offset_ref) != (ssize_t)chunk) { + fprintf(stderr, "%s: short read at %llu: %m\n", + __func__, (unsigned long long)offset); + return -1; + } + + if (memcmp(buffer, buffer_ref, chunk)) { + fprintf(stderr, "%s: %llu differs from %llu\n", __func__, + (unsigned long long)offset, + (unsigned long long)offset_ref); + return -1; + } + + offset += chunk; + offset_ref += chunk; + len -= chunk; + } + + return 0; +} + +/* The @len bytes at @offset the object left out have to have been zeroes. */ +static int check_zero_range(int fd, __u64 offset, __u64 len) +{ + static const char zeroes[PAGE_SIZE]; + char buffer[PAGE_SIZE]; + + while (len) { + size_t chunk = len < sizeof(buffer) ? len : sizeof(buffer); + + if (pread(fd, buffer, chunk, offset) != (ssize_t)chunk) { + fprintf(stderr, "%s: short read at %llu: %m\n", + __func__, (unsigned long long)offset); + return -1; + } + + if (memcmp(buffer, zeroes, chunk)) { + fprintf(stderr, "%s: %llu isn't a hole\n", __func__, + (unsigned long long)offset); + return -1; + } + + offset += chunk; + len -= chunk; + } + + return 0; +} + +/* + * The object has to describe the same memory as the coredump it was built + * from, and it has to describe it correctly. + */ +int check_compact_coredump(int fd_object, int fd_reference) +{ + ElfW(Phdr) *object = NULL, *reference = NULL; + size_t nr_object, nr_reference, i; + size_t io = 0, ir = 0; + int ret = -1; + + object = read_phdrs(fd_object, &nr_object); + reference = read_phdrs(fd_reference, &nr_reference); + if (!object || !reference) + goto out; + + /* Nothing may have been dropped and nothing may have been added. */ + for (;;) { + __u64 start = 0, end = 0, start_ref = 0, end_ref = 0; + bool has, has_ref; + + has = next_range(object, nr_object, &io, &start, &end); + has_ref = next_range(reference, nr_reference, &ir, &start_ref, + &end_ref); + if (!has && !has_ref) + break; + + if (has != has_ref || start != start_ref || end != end_ref) { + fprintf(stderr, "%s: object covers 0x%llx-0x%llx, coredump 0x%llx-0x%llx\n", + __func__, (unsigned long long)start, + (unsigned long long)end, + (unsigned long long)start_ref, + (unsigned long long)end_ref); + goto out; + } + } + + for (i = 0; i < nr_object; i++) { + const ElfW(Phdr) *segment; + __u64 offset, dumped; + + if (object[i].p_type != PT_LOAD || !object[i].p_memsz) + continue; + + segment = find_segment(reference, nr_reference, + object[i].p_vaddr); + if (!segment) { + fprintf(stderr, "%s: 0x%llx isn't in the coredump\n", + __func__, + (unsigned long long)object[i].p_vaddr); + goto out; + } + + offset = object[i].p_vaddr - segment->p_vaddr; + dumped = offset < segment->p_filesz ? + segment->p_filesz - offset : 0; + + /* What the object carries is what the coredump had. */ + if (object[i].p_filesz > dumped) { + fprintf(stderr, "%s: object carries %llu bytes the coredump doesn't have\n", + __func__, + (unsigned long long)(object[i].p_filesz - dumped)); + goto out; + } + + if (compare_range(fd_object, object[i].p_offset, fd_reference, + segment->p_offset + offset, + object[i].p_filesz)) + goto out; + + /* And what it left out was a hole. */ + if (object[i].p_memsz > object[i].p_filesz && + dumped > object[i].p_filesz) { + __u64 left_out = dumped - object[i].p_filesz; + + if (left_out > object[i].p_memsz - object[i].p_filesz) + left_out = object[i].p_memsz - object[i].p_filesz; + + if (check_zero_range(fd_reference, + segment->p_offset + offset + + object[i].p_filesz, left_out)) + goto out; + } + } + + ret = 0; +out: + free(object); + free(reference); + return ret; +} + int create_detached_tmpfs(void) { int fd_context, fd_tmpfs; diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.h b/tools/testing/selftests/coredump/coredump_test_helpers.h index fe0a88a71b0510..00d695b67b3fc0 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.h +++ b/tools/testing/selftests/coredump/coredump_test_helpers.h @@ -26,6 +26,9 @@ ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, off_t *coredump_size, bool *truncated, int fd_peer_pidfd); bool is_elf_core(int fd); +ssize_t recv_coredump_compact(int fd_coredump, int fd_object, int fd_reference, + off_t *coredump_size); +int check_compact_coredump(int fd_object, int fd_reference); int create_detached_tmpfs(void); int create_and_listen_unix_socket(const char *path); bool set_core_pattern(const char *pattern); From ff3b7e64efd072bec4cf42b838cdb7748e1cc50b Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 20 Aug 2026 01:09:39 +0200 Subject: [PATCH 276/857] selftests/coredump: show how to inspect the task to decide how the coredump should be sent The kernel blocks in the coredump req until the coredump ack is sent by the coredump server. This allows the coredump server to decide how the kernel is supposed to send the coredump. Let's show how that can work: - a task that has a large memory mapping gets sent as a sparse record stream - a task with a trivial memory mapping gets sent as a plain byte stream Since the threads are parked in coredump_task_exit() with their mm around we can look at /proc//statm to figure out what the task has mapped. Link: https://patch.msgid.link/20260820-work-coredump-sparse-v2-22-ba32dd718c51@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../coredump/coredump_socket_protocol_test.c | 184 ++++++++++++++++++ .../coredump/coredump_test_helpers.c | 58 ++++++ .../coredump/coredump_test_helpers.h | 7 +- 3 files changed, 248 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c index f33eaf2fa93d3e..daff908232a228 100644 --- a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c +++ b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c @@ -2136,4 +2136,188 @@ TEST_F(coredump, socket_request_sparse_without_records) check_conflicting_ack(_metadata, self, COREDUMP_KERNEL | COREDUMP_SPARSE); } +/* What the server reports back about the coredump it decided to take. */ +struct stream_choice { + bool sparse; + ssize_t received; + off_t size; + ssize_t vm_size; +}; + +/* + * The kernel blocks in the coredump request until the ack arrives, so a + * coredump server gets to look at the task before it commits to a + * stream. Take the record stream only for a task whose mappings are + * worth it and the plain byte stream for everything else. + */ +static void check_stream_choice(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, bool big, + struct stream_choice *choice) +{ + int pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + int ipc_sockets[2]; + int pipefds[2]; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_EQ(pipe(pipefds), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_file = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + struct stream_choice got = {}; + __u64 mask; + + close(ipc_sockets[0]); + close(pipefds[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + /* + * The reassembled coredump is bigger than the mapping the + * child made, so keep it on the detached tmpfs and sparse. + */ + fd_file = open_coredump_tmpfile(self->fd_tmpfs_detached); + if (fd_file < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + /* + * Nothing is on the wire yet and the kernel is waiting for + * the ack, so there is all the time in the world to look at + * the task and decide what to ask it for. + */ + got.vm_size = peer_vm_size(fd_peer_pidfd); + if (got.vm_size < 0) + goto out; + got.sparse = got.vm_size >= SPARSE_STREAM_THRESHOLD; + + fprintf(stderr, "Peer maps %zd bytes, asking for %s\n", + got.vm_size, + got.sparse ? "a sparse record stream" : "a byte stream"); + + mask = COREDUMP_KERNEL | COREDUMP_WAIT; + if (got.sparse) + mask |= COREDUMP_RECORDS | COREDUMP_SPARSE; + + if (!send_coredump_ack(fd_coredump, &req, mask, 0)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + if (got.sparse) { + got.received = recv_coredump_records(fd_coredump, fd_file, + &got.size, NULL, -1); + } else { + got.received = recv_coredump_bytes(fd_coredump, fd_file); + got.size = got.received; + } + if (got.received < 0) + goto out; + + /* Either way a debugger has to see an ordinary core file. */ + if (!is_elf_core(fd_file)) + goto out; + + if (write_nointr(pipefds[1], &got, sizeof(got)) != sizeof(got)) + goto out; + + exit_code = EXIT_SUCCESS; +out: + close(pipefds[1]); + if (fd_file >= 0) + close(fd_file); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + EXPECT_EQ(close(pipefds[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_sparse(big ? SPARSE_MAPPING_SIZE : PAGE_SIZE); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_EQ(read_nointr(pipefds[0], choice, sizeof(*choice)), + sizeof(*choice)); + EXPECT_EQ(close(pipefds[0]), 0); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); +} + +/* A task with little mapped isn't worth a record stream. */ +TEST_F(coredump, socket_request_stream_choice_small) +{ + struct stream_choice choice = {}; + + check_stream_choice(_metadata, self, false, &choice); + + ASSERT_LT(choice.vm_size, (ssize_t)SPARSE_STREAM_THRESHOLD); + ASSERT_FALSE(choice.sparse); + ASSERT_GT(choice.received, 0); +} + +/* A task sitting on a big mapping is. */ +TEST_F(coredump, socket_request_stream_choice_large) +{ + struct stream_choice choice = {}; + + check_stream_choice(_metadata, self, true, &choice); + + ASSERT_GE(choice.vm_size, (ssize_t)SPARSE_STREAM_THRESHOLD); + ASSERT_TRUE(choice.sparse); + ASSERT_GT(choice.size, (off_t)SPARSE_MAPPING_SIZE); + + /* The holes didn't have to go over the socket. */ + ASSERT_LT(choice.received, choice.size / 8); +} + TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 9346b8f688e260..d7cc448eeaf442 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -1095,6 +1095,33 @@ int check_compact_coredump(int fd_object, int fd_reference) return ret; } +/* Read a plain coredump byte stream to end-of-file. */ +ssize_t recv_coredump_bytes(int fd_coredump, int fd_core_file) +{ + ssize_t received = 0; + + for (;;) { + char buffer[PAGE_SIZE]; + ssize_t ret = read_nointr(fd_coredump, buffer, sizeof(buffer)); + + if (ret < 0) { + fprintf(stderr, "%s: read failed: %m\n", __func__); + return -1; + } + if (ret == 0) + break; + + if (write_nointr(fd_core_file, buffer, ret) != ret) { + fprintf(stderr, "%s: write failed: %m\n", __func__); + return -1; + } + received += ret; + } + + fprintf(stderr, "Received %zd bytes of coredump\n", received); + return received; +} + int create_detached_tmpfs(void) { int fd_context, fd_tmpfs; @@ -1190,6 +1217,37 @@ bool get_pidfd_info(int fd_peer_pidfd, struct pidfd_info *info) return true; } +/* + * How much the peer has mapped. The task is parked in the coredump + * handshake, so its mm is still there to be looked at. + */ +ssize_t peer_vm_size(int fd_peer_pidfd) +{ + struct pidfd_info info = {}; + unsigned long pages; + char path[64]; + FILE *f; + + if (!get_pidfd_info(fd_peer_pidfd, &info)) + return -1; + + snprintf(path, sizeof(path), "/proc/%d/statm", info.pid); + f = fopen(path, "r"); + if (!f) { + fprintf(stderr, "%s: %s: %m\n", __func__, path); + return -1; + } + + if (fscanf(f, "%lu", &pages) != 1) { + fprintf(stderr, "%s: %s: no size\n", __func__, path); + fclose(f); + return -1; + } + fclose(f); + + return (ssize_t)pages * sysconf(_SC_PAGESIZE); +} + /* Protocol helper functions */ ssize_t recv_marker(int fd) diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.h b/tools/testing/selftests/coredump/coredump_test_helpers.h index 00d695b67b3fc0..97ad5cfeae928d 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.h +++ b/tools/testing/selftests/coredump/coredump_test_helpers.h @@ -18,6 +18,9 @@ /* Size of the mostly unpopulated mapping the sparse coredump test maps. */ #define SPARSE_MAPPING_SIZE (256 * 1024 * 1024) +/* A task mapping at least this much is worth a record stream. */ +#define SPARSE_STREAM_THRESHOLD (SPARSE_MAPPING_SIZE / 2) + /* Shared helper function declarations */ void *do_nothing(void *arg); void crashing_child(void); @@ -25,9 +28,11 @@ void crashing_child_sparse(size_t size); ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, off_t *coredump_size, bool *truncated, int fd_peer_pidfd); -bool is_elf_core(int fd); ssize_t recv_coredump_compact(int fd_coredump, int fd_object, int fd_reference, off_t *coredump_size); +ssize_t recv_coredump_bytes(int fd_coredump, int fd_core_file); +ssize_t peer_vm_size(int fd_peer_pidfd); +bool is_elf_core(int fd); int check_compact_coredump(int fd_object, int fd_reference); int create_detached_tmpfs(void); int create_and_listen_unix_socket(const char *path); From 283e68caa4b5de9c264e317e9ef2d2bbd4e184ad Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Fri, 21 Aug 2026 13:52:02 +0200 Subject: [PATCH 277/857] coredump: select memory types to include Currently /proc//coredump_filter determines what types of memory are included in a coredump produced by . This is fairly static. The coredump server has no easy way to configure what memory to dump even though it can figure out all the necessary details to make an informed decision. Add a new COREDUMP_MEMORY_TYPES feature bit. If the coredump server raises it the kernel will dump memory types raised in the coredump_ack->memory_types member. Zero is valid and causes the creation of a coredump that just includes the program headers and notes but no memory apart from the mappings that are always dumped. struct coredump_req gains @memory_types which is set to the default memory types that are included in the coredump. This can be overridden by raising bits in coredump_ack->memory_types. It also gains @memory_types_mask which contains a bitmask of all memory types the kernel knows about. A coredump server may only raise bits in coredump_ack->memory_types that are raised in coredump_req->memory_types_mask. struct coredump_ack grows too. If COREDUMP_MEMORY_TYPES is raised in @mask the kernel dumps the memory types set in the @memory_types mask. Zero is valid and dumps no memory apart from the mappings that are always dumped. A coredump server wanting to add or drop memory types instead of outright replacing it should simply copy coredump_req->memory_types and then mask off or raise types as needed. @memory_types must be zero if COREDUMP_MEMORY_TYPES isn't raised. COREDUMP_MEMORY_TYPES requires COREDUMP_KERNEL and an ack of at least COREDUMP_ACK_SIZE_VER1 bytes. Link: https://patch.msgid.link/20260821-work-coredump-filter-v1-1-91f9a73ef03e@kernel.org Signed-off-by: Christian Brauner (Amutable) --- Documentation/filesystems/proc.rst | 4 + fs/coredump.c | 114 +++++++++++++++++++++++------ include/linux/coredump.h | 4 +- include/uapi/linux/coredump.h | 70 +++++++++++++++++- 4 files changed, 165 insertions(+), 27 deletions(-) diff --git a/Documentation/filesystems/proc.rst b/Documentation/filesystems/proc.rst index c102b62023cdc1..fc59c98acca10a 100644 --- a/Documentation/filesystems/proc.rst +++ b/Documentation/filesystems/proc.rst @@ -1963,6 +1963,10 @@ For example:: $ echo 0x7 > /proc/self/coredump_filter $ ./some_program +If the coredump socket protocol is used a coredump server can select memory +types to include dynamically. See COREDUMP_MEMORY_TYPES in +include/uapi/linux/coredump.h. + 3.5 /proc//mountinfo - Information about mounts -------------------------------------------------------- diff --git a/fs/coredump.c b/fs/coredump.c index 3b3721ea84af45..9addd2d59b7b7e 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -759,10 +759,39 @@ static inline bool coredump_sock_send(struct file *file, struct coredump_req *re return ret == sizeof(*req); } -static_assert(sizeof(struct coredump_req) == COREDUMP_REQ_SIZE_VER0); -static_assert(sizeof(struct coredump_ack) == COREDUMP_ACK_SIZE_VER0); +static_assert(sizeof(struct coredump_req) == COREDUMP_REQ_SIZE_VER1); +static_assert(sizeof(struct coredump_ack) == COREDUMP_ACK_SIZE_VER1); static_assert(sizeof(enum coredump_mark) == sizeof(__u32)); +/* Every memory type this kernel knows. */ +#define COREDUMP_MEMORY_ALL \ + (COREDUMP_MEMORY_ANON_PRIVATE | COREDUMP_MEMORY_ANON_SHARED | \ + COREDUMP_MEMORY_FILE_PRIVATE | COREDUMP_MEMORY_FILE_SHARED | \ + COREDUMP_MEMORY_ELF_HEADERS | \ + COREDUMP_MEMORY_HUGETLB_PRIVATE | COREDUMP_MEMORY_HUGETLB_SHARED | \ + COREDUMP_MEMORY_DAX_PRIVATE | COREDUMP_MEMORY_DAX_SHARED) + +#define COREDUMP_MEMORY_TYPE_BIT(mmf) BIT((mmf) - MMF_DUMP_FILTER_SHIFT) +static_assert(COREDUMP_MEMORY_ALL == (MMF_DUMP_FILTER_MASK >> MMF_DUMP_FILTER_SHIFT)); +static_assert(COREDUMP_MEMORY_ANON_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_ANON_PRIVATE)); +static_assert(COREDUMP_MEMORY_ANON_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_ANON_SHARED)); +static_assert(COREDUMP_MEMORY_FILE_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_MAPPED_PRIVATE)); +static_assert(COREDUMP_MEMORY_FILE_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_MAPPED_SHARED)); +static_assert(COREDUMP_MEMORY_ELF_HEADERS == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_ELF_HEADERS)); +static_assert(COREDUMP_MEMORY_HUGETLB_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_HUGETLB_PRIVATE)); +static_assert(COREDUMP_MEMORY_HUGETLB_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_HUGETLB_SHARED)); +static_assert(COREDUMP_MEMORY_DAX_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_DAX_PRIVATE)); +static_assert(COREDUMP_MEMORY_DAX_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_DAX_SHARED)); + static inline bool coredump_sock_mark(struct file *file, enum coredump_mark mark) { struct msghdr msg = { .msg_flags = MSG_NOSIGNAL }; @@ -804,11 +833,14 @@ static inline void coredump_sock_shutdown(struct file *file) static bool coredump_sock_request(struct core_name *cn, struct coredump_params *cprm) { struct coredump_req req = { - .size = sizeof(struct coredump_req), - .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT | - COREDUMP_RECORDS | COREDUMP_SPARSE, - .size_ack = sizeof(struct coredump_ack), + .size = sizeof(struct coredump_req), + .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | + COREDUMP_REJECT | COREDUMP_WAIT | + COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .size_ack = sizeof(struct coredump_ack), + .memory_types = cprm->memory_types, + .memory_types_mask = COREDUMP_MEMORY_ALL, }; struct coredump_ack ack = {}; ssize_t usize; @@ -873,6 +905,30 @@ static bool coredump_sock_request(struct core_name *cn, struct coredump_params * return false; } + if (ack.mask & COREDUMP_MEMORY_TYPES) { + /* The memory types need the whole field. */ + if (usize < COREDUMP_ACK_SIZE_VER1) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_MINSIZE); + return false; + } + + /* The memory types only select what the kernel writes. */ + if (!(ack.mask & COREDUMP_KERNEL)) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_CONFLICTING); + return false; + } + + /* Refuse unknown memory types. */ + if (ack.memory_types & ~req.memory_types_mask) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_UNSUPPORTED); + return false; + } + } else if (ack.memory_types) { + /* Like @spare the field must be zero when it isn't used. */ + coredump_sock_mark(cprm->file, COREDUMP_MARK_UNSUPPORTED); + return false; + } + /* Record header scratch; a bvec can't point at the stack. */ if (ack.mask & COREDUMP_RECORDS) { cprm->record_hdr = kmalloc_obj(*cprm->record_hdr); @@ -880,6 +936,10 @@ static bool coredump_sock_request(struct core_name *cn, struct coredump_params * return false; } + /* The server's selection replaces the task's entirely. */ + if (ack.mask & COREDUMP_MEMORY_TYPES) + cprm->memory_types = ack.memory_types; + cprm->mask = ack.mask; return coredump_sock_mark(cprm->file, COREDUMP_MARK_REQACK); } @@ -1190,6 +1250,10 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, } } +#define COREDUMP_TASK_MEMORY_TYPES(mm) \ + ((__mm_flags_get_word((mm)) & MMF_DUMP_FILTER_MASK) >> \ + MMF_DUMP_FILTER_SHIFT) + void vfs_coredump(const kernel_siginfo_t *siginfo) { size_t *argv __free(kfree) = NULL; @@ -1201,8 +1265,8 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) struct coredump_params cprm = { .siginfo = siginfo, .limit = rlimit(RLIMIT_CORE), - /* Snapshot MMF_DUMP_FILTER_* (unlocked) and dumpable for the dump. */ - .mm_flags = __mm_flags_get_word(mm), + /* Snapshot the memory types (unlocked) and dumpable for the dump. */ + .memory_types = COREDUMP_TASK_MEMORY_TYPES(mm), .dumpable = task_exec_state_get_dumpable(current), .vma_meta = NULL, .cpu = raw_smp_processor_id(), @@ -1736,15 +1800,15 @@ static bool always_dump_vma(struct vm_area_struct *vma) } #define DUMP_SIZE_MAYBE_ELFHDR_PLACEHOLDER 1 +#define COREDUMP_MEMORY_TYPE_INCLUDE(types, type) \ + ((types) & COREDUMP_MEMORY_##type) /* * Decide how much of @vma's contents should be included in a core dump. */ static unsigned long vma_dump_size(struct vm_area_struct *vma, - unsigned long mm_flags) + u64 memory_types) { -#define FILTER(type) (mm_flags & (1UL << MMF_DUMP_##type)) - /* always dump the vdso and vsyscall sections */ if (always_dump_vma(vma)) goto whole; @@ -1754,18 +1818,22 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, /* support for DAX */ if (vma_is_dax(vma)) { - if ((vma->vm_flags & VM_SHARED) && FILTER(DAX_SHARED)) + if ((vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, DAX_SHARED)) goto whole; - if (!(vma->vm_flags & VM_SHARED) && FILTER(DAX_PRIVATE)) + if (!(vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, DAX_PRIVATE)) goto whole; return 0; } /* Hugetlb memory check */ if (is_vm_hugetlb_page(vma)) { - if ((vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_SHARED)) + if ((vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, HUGETLB_SHARED)) goto whole; - if (!(vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_PRIVATE)) + if (!(vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, HUGETLB_PRIVATE)) goto whole; return 0; } @@ -1777,25 +1845,27 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, /* By default, dump shared memory if mapped from an anonymous file. */ if (vma->vm_flags & VM_SHARED) { if (file_inode(vma->vm_file)->i_nlink == 0 ? - FILTER(ANON_SHARED) : FILTER(MAPPED_SHARED)) + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, ANON_SHARED) : + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, FILE_SHARED)) goto whole; return 0; } /* Dump segments that have been written to. */ - if ((!IS_ENABLED(CONFIG_MMU) || vma->anon_vma) && FILTER(ANON_PRIVATE)) + if ((!IS_ENABLED(CONFIG_MMU) || vma->anon_vma) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, ANON_PRIVATE)) goto whole; if (vma->vm_file == NULL) return 0; - if (FILTER(MAPPED_PRIVATE)) + if (COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, FILE_PRIVATE)) goto whole; /* * If this is the beginning of an executable file mapping, * dump the first page to aid in determining what was mapped here. */ - if (FILTER(ELF_HEADERS) && + if (COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, ELF_HEADERS) && vma->vm_pgoff == 0 && (vma->vm_flags & VM_READ)) { if ((READ_ONCE(file_inode(vma->vm_file)->i_mode) & 0111) != 0) return PAGE_SIZE; @@ -1811,8 +1881,6 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, return DUMP_SIZE_MAYBE_ELFHDR_PLACEHOLDER; } -#undef FILTER - return 0; whole: @@ -1897,7 +1965,7 @@ static bool dump_vma_snapshot(struct coredump_params *cprm) m->start = vma->vm_start; m->end = vma->vm_end; m->flags = vma->vm_flags; - m->dump_size = vma_dump_size(vma, cprm->mm_flags); + m->dump_size = vma_dump_size(vma, cprm->memory_types); m->pgoff = vma->vm_pgoff; m->file = vma->vm_file; if (m->file) diff --git a/include/linux/coredump.h b/include/linux/coredump.h index b252bb2843b337..74af57b9406b8e 100644 --- a/include/linux/coredump.h +++ b/include/linux/coredump.h @@ -32,8 +32,8 @@ struct coredump_params { const kernel_siginfo_t *siginfo; struct file *file; unsigned long limit; - /* MMF_DUMP_FILTER_* bits, snapshot of mm->flags at dump start. */ - unsigned long mm_flags; + /* COREDUMP_MEMORY_* types to dump, the task's or the server's. */ + u64 memory_types; /* Snapshot of dumpable at dump start. */ enum task_dumpable dumpable; int cpu; diff --git a/include/uapi/linux/coredump.h b/include/uapi/linux/coredump.h index f3771861ca4857..6d0c53b534ea8d 100644 --- a/include/uapi/linux/coredump.h +++ b/include/uapi/linux/coredump.h @@ -16,6 +16,9 @@ * requires COREDUMP_KERNEL * @COREDUMP_SPARSE: describe the holes in the coredump as zero records * instead of transferring them; requires COREDUMP_RECORDS + * @COREDUMP_MEMORY_TYPES: dump the memory types in + * coredump_ack->memory_types instead of the ones + * the task selected; requires COREDUMP_KERNEL */ enum { COREDUMP_KERNEL = (1ULL << 0), @@ -24,6 +27,37 @@ enum { COREDUMP_WAIT = (1ULL << 3), COREDUMP_RECORDS = (1ULL << 4), COREDUMP_SPARSE = (1ULL << 5), + COREDUMP_MEMORY_TYPES = (1ULL << 6), +}; + +/** + * coredump memory types + * @COREDUMP_MEMORY_ANON_PRIVATE: anonymous private memory + * @COREDUMP_MEMORY_ANON_SHARED: anonymous shared memory + * @COREDUMP_MEMORY_FILE_PRIVATE: file-backed private memory + * @COREDUMP_MEMORY_FILE_SHARED: file-backed shared memory + * @COREDUMP_MEMORY_ELF_HEADERS: the first page of a file-backed private + * mapping that starts an ELF file + * @COREDUMP_MEMORY_HUGETLB_PRIVATE: hugetlb private memory + * @COREDUMP_MEMORY_HUGETLB_SHARED: hugetlb shared memory + * @COREDUMP_MEMORY_DAX_PRIVATE: DAX private memory + * @COREDUMP_MEMORY_DAX_SHARED: DAX shared memory + * + * A bitmask of memory types a coredump may request to be included. New + * memory type bits must ensure that they do not steal memory from an + * existing one so a coredump server will continue to get the same + * coredumps even if a new bit is introduced. + */ +enum { + COREDUMP_MEMORY_ANON_PRIVATE = (1ULL << 0), + COREDUMP_MEMORY_ANON_SHARED = (1ULL << 1), + COREDUMP_MEMORY_FILE_PRIVATE = (1ULL << 2), + COREDUMP_MEMORY_FILE_SHARED = (1ULL << 3), + COREDUMP_MEMORY_ELF_HEADERS = (1ULL << 4), + COREDUMP_MEMORY_HUGETLB_PRIVATE = (1ULL << 5), + COREDUMP_MEMORY_HUGETLB_SHARED = (1ULL << 6), + COREDUMP_MEMORY_DAX_PRIVATE = (1ULL << 7), + COREDUMP_MEMORY_DAX_SHARED = (1ULL << 8), }; /** @@ -31,6 +65,8 @@ enum { * @size: size of struct coredump_req * @size_ack: known size of struct coredump_ack on this kernel * @mask: supported features + * @memory_types: the memory types the task selected + * @memory_types_mask: the memory types this kernel knows * * When a coredump happens the kernel will connect to the coredump * socket and send a coredump request to the coredump server. The @size @@ -49,15 +85,27 @@ enum { * struct coredump_ack the kernel knows. Userspace may only send up to * coredump_req->size_ack bytes to the kernel and must set * coredump_ack->size accordingly. + * + * @memory_types is set to the default memory types that are included in + * the coredump. This can be overridden by raising bits in + * coredump_ack->memory_types. + * + * @memory_types_mask contains a bitmask of all memory types the kernel + * knows about. A coredump server may only raise bits in + * coredump_ack->memory_types that are raised in + * coredump_req->memory_types_mask. */ struct coredump_req { __u32 size; __u32 size_ack; __u64 mask; + __u64 memory_types; + __u64 memory_types_mask; }; enum { COREDUMP_REQ_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_REQ_SIZE_VER1 = 32U, /* memory_types and memory_types_mask added */ }; /** @@ -65,6 +113,8 @@ enum { * @size: size of the struct * @spare: unused * @mask: features kernel is supposed to use + * @memory_types: memory types to dump, only with COREDUMP_MEMORY_TYPES + * in @mask * * The @size member must be set to the size of struct coredump_ack. It * may never exceed what the kernel returned in coredump_req->size_ack @@ -74,15 +124,30 @@ enum { * The @mask member must be set to the features the coredump server * wants the kernel to use. Only bits the kernel returned in * coredump_req->mask may be set. + * + * If COREDUMP_MEMORY_TYPES is raised in @mask the kernel dumps the + * memory types set in the @memory_types mask. Zero is valid and dumps + * no memory apart from the mappings that are always dumped. + * + * Note that memory a task excluded via MADV_DONTDUMP is always left + * out. A coredump server wanting to add or drop memory types instead of + * outright replacing it should simply copy coredump_req->memory_types + * and then mask off or raise types as needed. + * + * Note that @memory_types must be zero if COREDUMP_MEMORY_TYPES isn't + * raised. COREDUMP_MEMORY_TYPES requires COREDUMP_KERNEL and an ack of + * at least COREDUMP_ACK_SIZE_VER1 bytes. */ struct coredump_ack { __u32 size; __u32 spare; __u64 mask; + __u64 memory_types; }; enum { COREDUMP_ACK_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_ACK_SIZE_VER1 = 24U, /* memory_types added */ }; /** @@ -90,11 +155,12 @@ enum { * * The kernel will place a single byte on the coredump socket. The * markers notify userspace whether the coredump ack succeeded or - * failed. + * failed. After any marker other than COREDUMP_MARK_REQACK the kernel + * closes the connection and no coredump is generated. * * @COREDUMP_MARK_MINSIZE: the provided coredump_ack size was too small * @COREDUMP_MARK_MAXSIZE: the provided coredump_ack size was too big - * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask was invalid + * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask or memory types were invalid * @COREDUMP_MARK_CONFLICTING: the provided coredump_ack mask has conflicting options * @COREDUMP_MARK_REQACK: the coredump request and ack was successful * @__COREDUMP_MARK_MAX: the maximum coredump mark value From 10bffb823bf4ad5d242f8c7945da1c3e2eb8a500 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Fri, 21 Aug 2026 13:52:03 +0200 Subject: [PATCH 278/857] tools: sync coredump.h header Sync the headers for the selftests. Link: https://patch.msgid.link/20260821-work-coredump-filter-v1-2-91f9a73ef03e@kernel.org Signed-off-by: Christian Brauner (Amutable) --- tools/include/uapi/linux/coredump.h | 70 ++++++++++++++++++++++++++++- 1 file changed, 68 insertions(+), 2 deletions(-) diff --git a/tools/include/uapi/linux/coredump.h b/tools/include/uapi/linux/coredump.h index f3771861ca4857..6d0c53b534ea8d 100644 --- a/tools/include/uapi/linux/coredump.h +++ b/tools/include/uapi/linux/coredump.h @@ -16,6 +16,9 @@ * requires COREDUMP_KERNEL * @COREDUMP_SPARSE: describe the holes in the coredump as zero records * instead of transferring them; requires COREDUMP_RECORDS + * @COREDUMP_MEMORY_TYPES: dump the memory types in + * coredump_ack->memory_types instead of the ones + * the task selected; requires COREDUMP_KERNEL */ enum { COREDUMP_KERNEL = (1ULL << 0), @@ -24,6 +27,37 @@ enum { COREDUMP_WAIT = (1ULL << 3), COREDUMP_RECORDS = (1ULL << 4), COREDUMP_SPARSE = (1ULL << 5), + COREDUMP_MEMORY_TYPES = (1ULL << 6), +}; + +/** + * coredump memory types + * @COREDUMP_MEMORY_ANON_PRIVATE: anonymous private memory + * @COREDUMP_MEMORY_ANON_SHARED: anonymous shared memory + * @COREDUMP_MEMORY_FILE_PRIVATE: file-backed private memory + * @COREDUMP_MEMORY_FILE_SHARED: file-backed shared memory + * @COREDUMP_MEMORY_ELF_HEADERS: the first page of a file-backed private + * mapping that starts an ELF file + * @COREDUMP_MEMORY_HUGETLB_PRIVATE: hugetlb private memory + * @COREDUMP_MEMORY_HUGETLB_SHARED: hugetlb shared memory + * @COREDUMP_MEMORY_DAX_PRIVATE: DAX private memory + * @COREDUMP_MEMORY_DAX_SHARED: DAX shared memory + * + * A bitmask of memory types a coredump may request to be included. New + * memory type bits must ensure that they do not steal memory from an + * existing one so a coredump server will continue to get the same + * coredumps even if a new bit is introduced. + */ +enum { + COREDUMP_MEMORY_ANON_PRIVATE = (1ULL << 0), + COREDUMP_MEMORY_ANON_SHARED = (1ULL << 1), + COREDUMP_MEMORY_FILE_PRIVATE = (1ULL << 2), + COREDUMP_MEMORY_FILE_SHARED = (1ULL << 3), + COREDUMP_MEMORY_ELF_HEADERS = (1ULL << 4), + COREDUMP_MEMORY_HUGETLB_PRIVATE = (1ULL << 5), + COREDUMP_MEMORY_HUGETLB_SHARED = (1ULL << 6), + COREDUMP_MEMORY_DAX_PRIVATE = (1ULL << 7), + COREDUMP_MEMORY_DAX_SHARED = (1ULL << 8), }; /** @@ -31,6 +65,8 @@ enum { * @size: size of struct coredump_req * @size_ack: known size of struct coredump_ack on this kernel * @mask: supported features + * @memory_types: the memory types the task selected + * @memory_types_mask: the memory types this kernel knows * * When a coredump happens the kernel will connect to the coredump * socket and send a coredump request to the coredump server. The @size @@ -49,15 +85,27 @@ enum { * struct coredump_ack the kernel knows. Userspace may only send up to * coredump_req->size_ack bytes to the kernel and must set * coredump_ack->size accordingly. + * + * @memory_types is set to the default memory types that are included in + * the coredump. This can be overridden by raising bits in + * coredump_ack->memory_types. + * + * @memory_types_mask contains a bitmask of all memory types the kernel + * knows about. A coredump server may only raise bits in + * coredump_ack->memory_types that are raised in + * coredump_req->memory_types_mask. */ struct coredump_req { __u32 size; __u32 size_ack; __u64 mask; + __u64 memory_types; + __u64 memory_types_mask; }; enum { COREDUMP_REQ_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_REQ_SIZE_VER1 = 32U, /* memory_types and memory_types_mask added */ }; /** @@ -65,6 +113,8 @@ enum { * @size: size of the struct * @spare: unused * @mask: features kernel is supposed to use + * @memory_types: memory types to dump, only with COREDUMP_MEMORY_TYPES + * in @mask * * The @size member must be set to the size of struct coredump_ack. It * may never exceed what the kernel returned in coredump_req->size_ack @@ -74,15 +124,30 @@ enum { * The @mask member must be set to the features the coredump server * wants the kernel to use. Only bits the kernel returned in * coredump_req->mask may be set. + * + * If COREDUMP_MEMORY_TYPES is raised in @mask the kernel dumps the + * memory types set in the @memory_types mask. Zero is valid and dumps + * no memory apart from the mappings that are always dumped. + * + * Note that memory a task excluded via MADV_DONTDUMP is always left + * out. A coredump server wanting to add or drop memory types instead of + * outright replacing it should simply copy coredump_req->memory_types + * and then mask off or raise types as needed. + * + * Note that @memory_types must be zero if COREDUMP_MEMORY_TYPES isn't + * raised. COREDUMP_MEMORY_TYPES requires COREDUMP_KERNEL and an ack of + * at least COREDUMP_ACK_SIZE_VER1 bytes. */ struct coredump_ack { __u32 size; __u32 spare; __u64 mask; + __u64 memory_types; }; enum { COREDUMP_ACK_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_ACK_SIZE_VER1 = 24U, /* memory_types added */ }; /** @@ -90,11 +155,12 @@ enum { * * The kernel will place a single byte on the coredump socket. The * markers notify userspace whether the coredump ack succeeded or - * failed. + * failed. After any marker other than COREDUMP_MARK_REQACK the kernel + * closes the connection and no coredump is generated. * * @COREDUMP_MARK_MINSIZE: the provided coredump_ack size was too small * @COREDUMP_MARK_MAXSIZE: the provided coredump_ack size was too big - * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask was invalid + * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask or memory types were invalid * @COREDUMP_MARK_CONFLICTING: the provided coredump_ack mask has conflicting options * @COREDUMP_MARK_REQACK: the coredump request and ack was successful * @__COREDUMP_MARK_MAX: the maximum coredump mark value From 5c493013e1a2b58eafbc53d55855764a5130b27e Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Fri, 21 Aug 2026 13:52:04 +0200 Subject: [PATCH 279/857] selftests/coredump: simplify the refusal tests A couple of tests send a coredump_ack that the kernel refuses. They then check the marker. Make sure they all use common infrastructure. Link: https://patch.msgid.link/20260821-work-coredump-filter-v1-3-91f9a73ef03e@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../coredump/coredump_socket_protocol_test.c | 558 +++--------------- .../coredump/coredump_test_helpers.c | 37 +- .../coredump/coredump_test_helpers.h | 1 + 3 files changed, 111 insertions(+), 485 deletions(-) diff --git a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c index daff908232a228..a07546e796517e 100644 --- a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c +++ b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c @@ -508,91 +508,79 @@ TEST_F(coredump, socket_request_reject) wait_and_check_coredump_server(pid_coredump_server, _metadata, self); } -TEST_F(coredump, socket_request_invalid_flag_combination) +/* An ack the kernel must refuse and how. */ +struct refused_ack { + /* The ack and how many bytes of it the server sends. */ + struct coredump_ack ack; + size_t bytes; + /* The marker the kernel answers with. */ + enum coredump_mark mark; +}; + +/* Send @refused, expect the kernel to refuse it with the marker. */ +static void check_refused_ack(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, + const struct refused_ack *refused) { - int pidfd, ret, status; + int pidfd, status; pid_t pid, pid_coredump_server; struct pidfd_info info = {}; int ipc_sockets[2]; char c; + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); - ret = socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets); - ASSERT_EQ(ret, 0); - pid_coredump_server = fork(); ASSERT_GE(pid_coredump_server, 0); if (pid_coredump_server == 0) { - struct coredump_req req = {}; int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; close(ipc_sockets[0]); fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); - if (fd_server < 0) { - fprintf(stderr, "socket_request_invalid_flag_combination: create_and_listen_unix_socket failed: %m\n"); + if (fd_server < 0) goto out; - } - if (write_nointr(ipc_sockets[1], "1", 1) < 0) { - fprintf(stderr, "socket_request_invalid_flag_combination: write_nointr to ipc socket failed: %m\n"); + if (write_nointr(ipc_sockets[1], "1", 1) < 0) goto out; - } close(ipc_sockets[1]); fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); - if (fd_coredump < 0) { - fprintf(stderr, "socket_request_invalid_flag_combination: accept4 failed: %m\n"); + if (fd_coredump < 0) goto out; - } fd_peer_pidfd = get_peer_pidfd(fd_coredump); - if (fd_peer_pidfd < 0) { - fprintf(stderr, "socket_request_invalid_flag_combination: get_peer_pidfd failed\n"); + if (fd_peer_pidfd < 0) goto out; - } - if (!get_pidfd_info(fd_peer_pidfd, &info)) { - fprintf(stderr, "socket_request_invalid_flag_combination: get_pidfd_info failed\n"); + /* The task shows as dumping while it waits for the ack. */ + if (!get_pidfd_info(fd_peer_pidfd, &info)) goto out; - } - if (!(info.mask & PIDFD_INFO_COREDUMP)) { - fprintf(stderr, "socket_request_invalid_flag_combination: PIDFD_INFO_COREDUMP not set in mask\n"); + if (!(info.mask & PIDFD_INFO_COREDUMP) || + !(info.coredump_mask & PIDFD_COREDUMPED)) { + fprintf(stderr, "Peer isn't marked as dumping\n"); goto out; } - if (!(info.coredump_mask & PIDFD_COREDUMPED)) { - fprintf(stderr, "socket_request_invalid_flag_combination: PIDFD_COREDUMPED not set in coredump_mask\n"); - goto out; - } - - if (!read_coredump_req(fd_coredump, &req)) { - fprintf(stderr, "socket_request_invalid_flag_combination: read_coredump_req failed\n"); + if (!read_coredump_req(fd_coredump, &req)) goto out; - } - if (!check_coredump_req(&req)) { - fprintf(stderr, "socket_request_invalid_flag_combination: check_coredump_req failed\n"); + if (!check_coredump_req(&req)) goto out; - } - if (!send_coredump_ack(fd_coredump, &req, - COREDUMP_KERNEL | COREDUMP_REJECT | COREDUMP_WAIT, 0)) { - fprintf(stderr, "socket_request_invalid_flag_combination: send_coredump_ack failed\n"); + if (!send_coredump_ack_bytes(fd_coredump, &refused->ack, + refused->bytes)) goto out; - } - if (!read_marker(fd_coredump, COREDUMP_MARK_CONFLICTING)) { - fprintf(stderr, "socket_request_invalid_flag_combination: read_marker COREDUMP_MARK_CONFLICTING failed\n"); + if (!read_marker(fd_coredump, refused->mark)) goto out; - } exit_code = EXIT_SUCCESS; - fprintf(stderr, "socket_request_invalid_flag_combination: completed successfully\n"); out: if (fd_peer_pidfd >= 0) close(fd_peer_pidfd); @@ -627,362 +615,72 @@ TEST_F(coredump, socket_request_invalid_flag_combination) wait_and_check_coredump_server(pid_coredump_server, _metadata, self); } -TEST_F(coredump, socket_request_unknown_flag) +/* Ack @ack_mask, expect the kernel to refuse it as conflicting. */ +static void check_conflicting_ack(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, __u64 ack_mask) { - int pidfd, ret, status; - pid_t pid, pid_coredump_server; - struct pidfd_info info = {}; - int ipc_sockets[2]; - char c; - - ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); - - ret = socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets); - ASSERT_EQ(ret, 0); - - pid_coredump_server = fork(); - ASSERT_GE(pid_coredump_server, 0); - if (pid_coredump_server == 0) { - struct coredump_req req = {}; - int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; - int exit_code = EXIT_FAILURE; - - close(ipc_sockets[0]); - - fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); - if (fd_server < 0) { - fprintf(stderr, "socket_request_unknown_flag: create_and_listen_unix_socket failed: %m\n"); - goto out; - } - - if (write_nointr(ipc_sockets[1], "1", 1) < 0) { - fprintf(stderr, "socket_request_unknown_flag: write_nointr to ipc socket failed: %m\n"); - goto out; - } - - close(ipc_sockets[1]); - - fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); - if (fd_coredump < 0) { - fprintf(stderr, "socket_request_unknown_flag: accept4 failed: %m\n"); - goto out; - } - - fd_peer_pidfd = get_peer_pidfd(fd_coredump); - if (fd_peer_pidfd < 0) { - fprintf(stderr, "socket_request_unknown_flag: get_peer_pidfd failed\n"); - goto out; - } - - if (!get_pidfd_info(fd_peer_pidfd, &info)) { - fprintf(stderr, "socket_request_unknown_flag: get_pidfd_info failed\n"); - goto out; - } - - if (!(info.mask & PIDFD_INFO_COREDUMP)) { - fprintf(stderr, "socket_request_unknown_flag: PIDFD_INFO_COREDUMP not set in mask\n"); - goto out; - } - - if (!(info.coredump_mask & PIDFD_COREDUMPED)) { - fprintf(stderr, "socket_request_unknown_flag: PIDFD_COREDUMPED not set in coredump_mask\n"); - goto out; - } - - if (!read_coredump_req(fd_coredump, &req)) { - fprintf(stderr, "socket_request_unknown_flag: read_coredump_req failed\n"); - goto out; - } - - if (!check_coredump_req(&req)) { - fprintf(stderr, "socket_request_unknown_flag: check_coredump_req failed\n"); - goto out; - } - - if (!send_coredump_ack(fd_coredump, &req, (1ULL << 63), 0)) { - fprintf(stderr, "socket_request_unknown_flag: send_coredump_ack failed\n"); - goto out; - } - - if (!read_marker(fd_coredump, COREDUMP_MARK_UNSUPPORTED)) { - fprintf(stderr, "socket_request_unknown_flag: read_marker COREDUMP_MARK_UNSUPPORTED failed\n"); - goto out; - } - - exit_code = EXIT_SUCCESS; - fprintf(stderr, "socket_request_unknown_flag: completed successfully\n"); -out: - if (fd_peer_pidfd >= 0) - close(fd_peer_pidfd); - if (fd_coredump >= 0) - close(fd_coredump); - if (fd_server >= 0) - close(fd_server); - _exit(exit_code); - } - self->pid_coredump_server = pid_coredump_server; - - EXPECT_EQ(close(ipc_sockets[1]), 0); - ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); - EXPECT_EQ(close(ipc_sockets[0]), 0); - - pid = fork(); - ASSERT_GE(pid, 0); - if (pid == 0) - crashing_child(); - - pidfd = sys_pidfd_open(pid, 0); - ASSERT_GE(pidfd, 0); - - waitpid(pid, &status, 0); - ASSERT_TRUE(WIFSIGNALED(status)); - ASSERT_FALSE(WCOREDUMP(status)); + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = ack_mask, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_CONFLICTING, + }; + + check_refused_ack(_metadata, self, &refused); +} - ASSERT_TRUE(get_pidfd_info(pidfd, &info)); - ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); - ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); +/* More than one of KERNEL, USERSPACE and REJECT. */ +TEST_F(coredump, socket_request_invalid_flag_combination) +{ + check_conflicting_ack(_metadata, self, + COREDUMP_KERNEL | COREDUMP_REJECT | COREDUMP_WAIT); +} - wait_and_check_coredump_server(pid_coredump_server, _metadata, self); +/* A flag the kernel didn't advertise in coredump_req->mask. */ +TEST_F(coredump, socket_request_unknown_flag) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = 1ULL << 63, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_UNSUPPORTED, + }; + + check_refused_ack(_metadata, self, &refused); } +/* An ack smaller than the first published struct. */ TEST_F(coredump, socket_request_invalid_size_small) { - int pidfd, ret, status; - pid_t pid, pid_coredump_server; - struct pidfd_info info = {}; - int ipc_sockets[2]; - char c; - - ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); - - ret = socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets); - ASSERT_EQ(ret, 0); - - pid_coredump_server = fork(); - ASSERT_GE(pid_coredump_server, 0); - if (pid_coredump_server == 0) { - struct coredump_req req = {}; - int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; - int exit_code = EXIT_FAILURE; - - close(ipc_sockets[0]); - - fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); - if (fd_server < 0) { - fprintf(stderr, "socket_request_invalid_size_small: create_and_listen_unix_socket failed: %m\n"); - goto out; - } - - if (write_nointr(ipc_sockets[1], "1", 1) < 0) { - fprintf(stderr, "socket_request_invalid_size_small: write_nointr to ipc socket failed: %m\n"); - goto out; - } - - close(ipc_sockets[1]); - - fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); - if (fd_coredump < 0) { - fprintf(stderr, "socket_request_invalid_size_small: accept4 failed: %m\n"); - goto out; - } - - fd_peer_pidfd = get_peer_pidfd(fd_coredump); - if (fd_peer_pidfd < 0) { - fprintf(stderr, "socket_request_invalid_size_small: get_peer_pidfd failed\n"); - goto out; - } - - if (!get_pidfd_info(fd_peer_pidfd, &info)) { - fprintf(stderr, "socket_request_invalid_size_small: get_pidfd_info failed\n"); - goto out; - } - - if (!(info.mask & PIDFD_INFO_COREDUMP)) { - fprintf(stderr, "socket_request_invalid_size_small: PIDFD_INFO_COREDUMP not set in mask\n"); - goto out; - } - - if (!(info.coredump_mask & PIDFD_COREDUMPED)) { - fprintf(stderr, "socket_request_invalid_size_small: PIDFD_COREDUMPED not set in coredump_mask\n"); - goto out; - } - - if (!read_coredump_req(fd_coredump, &req)) { - fprintf(stderr, "socket_request_invalid_size_small: read_coredump_req failed\n"); - goto out; - } - - if (!check_coredump_req(&req)) { - fprintf(stderr, "socket_request_invalid_size_small: check_coredump_req failed\n"); - goto out; - } - - if (!send_coredump_ack(fd_coredump, &req, - COREDUMP_REJECT | COREDUMP_WAIT, - COREDUMP_ACK_SIZE_VER0 / 2)) { - fprintf(stderr, "socket_request_invalid_size_small: send_coredump_ack failed\n"); - goto out; - } - - if (!read_marker(fd_coredump, COREDUMP_MARK_MINSIZE)) { - fprintf(stderr, "socket_request_invalid_size_small: read_marker COREDUMP_MARK_MINSIZE failed\n"); - goto out; - } - - exit_code = EXIT_SUCCESS; - fprintf(stderr, "socket_request_invalid_size_small: completed successfully\n"); -out: - if (fd_peer_pidfd >= 0) - close(fd_peer_pidfd); - if (fd_coredump >= 0) - close(fd_coredump); - if (fd_server >= 0) - close(fd_server); - _exit(exit_code); - } - self->pid_coredump_server = pid_coredump_server; - - EXPECT_EQ(close(ipc_sockets[1]), 0); - ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); - EXPECT_EQ(close(ipc_sockets[0]), 0); - - pid = fork(); - ASSERT_GE(pid, 0); - if (pid == 0) - crashing_child(); - - pidfd = sys_pidfd_open(pid, 0); - ASSERT_GE(pidfd, 0); - - waitpid(pid, &status, 0); - ASSERT_TRUE(WIFSIGNALED(status)); - ASSERT_FALSE(WCOREDUMP(status)); - - ASSERT_TRUE(get_pidfd_info(pidfd, &info)); - ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); - ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); - - wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + struct refused_ack refused = { + .ack = { + .size = COREDUMP_ACK_SIZE_VER0 / 2, + .mask = COREDUMP_REJECT | COREDUMP_WAIT, + }, + .bytes = COREDUMP_ACK_SIZE_VER0 / 2, + .mark = COREDUMP_MARK_MINSIZE, + }; + + check_refused_ack(_metadata, self, &refused); } +/* An ack bigger than the kernel said it accepts. */ TEST_F(coredump, socket_request_invalid_size_large) { - int pidfd, ret, status; - pid_t pid, pid_coredump_server; - struct pidfd_info info = {}; - int ipc_sockets[2]; - char c; - - ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); - - ret = socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets); - ASSERT_EQ(ret, 0); - - pid_coredump_server = fork(); - ASSERT_GE(pid_coredump_server, 0); - if (pid_coredump_server == 0) { - struct coredump_req req = {}; - int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; - int exit_code = EXIT_FAILURE; - - close(ipc_sockets[0]); - - fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); - if (fd_server < 0) { - fprintf(stderr, "socket_request_invalid_size_large: create_and_listen_unix_socket failed: %m\n"); - goto out; - } - - if (write_nointr(ipc_sockets[1], "1", 1) < 0) { - fprintf(stderr, "socket_request_invalid_size_large: write_nointr to ipc socket failed: %m\n"); - goto out; - } - - close(ipc_sockets[1]); - - fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); - if (fd_coredump < 0) { - fprintf(stderr, "socket_request_invalid_size_large: accept4 failed: %m\n"); - goto out; - } - - fd_peer_pidfd = get_peer_pidfd(fd_coredump); - if (fd_peer_pidfd < 0) { - fprintf(stderr, "socket_request_invalid_size_large: get_peer_pidfd failed\n"); - goto out; - } - - if (!get_pidfd_info(fd_peer_pidfd, &info)) { - fprintf(stderr, "socket_request_invalid_size_large: get_pidfd_info failed\n"); - goto out; - } - - if (!(info.mask & PIDFD_INFO_COREDUMP)) { - fprintf(stderr, "socket_request_invalid_size_large: PIDFD_INFO_COREDUMP not set in mask\n"); - goto out; - } - - if (!(info.coredump_mask & PIDFD_COREDUMPED)) { - fprintf(stderr, "socket_request_invalid_size_large: PIDFD_COREDUMPED not set in coredump_mask\n"); - goto out; - } - - if (!read_coredump_req(fd_coredump, &req)) { - fprintf(stderr, "socket_request_invalid_size_large: read_coredump_req failed\n"); - goto out; - } - - if (!check_coredump_req(&req)) { - fprintf(stderr, "socket_request_invalid_size_large: check_coredump_req failed\n"); - goto out; - } - - if (!send_coredump_ack(fd_coredump, &req, - COREDUMP_REJECT | COREDUMP_WAIT, - COREDUMP_ACK_SIZE_VER0 + PAGE_SIZE)) { - fprintf(stderr, "socket_request_invalid_size_large: send_coredump_ack failed\n"); - goto out; - } - - if (!read_marker(fd_coredump, COREDUMP_MARK_MAXSIZE)) { - fprintf(stderr, "socket_request_invalid_size_large: read_marker COREDUMP_MARK_MAXSIZE failed\n"); - goto out; - } - - exit_code = EXIT_SUCCESS; - fprintf(stderr, "socket_request_invalid_size_large: completed successfully\n"); -out: - if (fd_peer_pidfd >= 0) - close(fd_peer_pidfd); - if (fd_coredump >= 0) - close(fd_coredump); - if (fd_server >= 0) - close(fd_server); - _exit(exit_code); - } - self->pid_coredump_server = pid_coredump_server; - - EXPECT_EQ(close(ipc_sockets[1]), 0); - ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); - EXPECT_EQ(close(ipc_sockets[0]), 0); - - pid = fork(); - ASSERT_GE(pid, 0); - if (pid == 0) - crashing_child(); - - pidfd = sys_pidfd_open(pid, 0); - ASSERT_GE(pidfd, 0); - - waitpid(pid, &status, 0); - ASSERT_TRUE(WIFSIGNALED(status)); - ASSERT_FALSE(WCOREDUMP(status)); - - ASSERT_TRUE(get_pidfd_info(pidfd, &info)); - ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); - ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); - - wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + struct refused_ack refused = { + .ack = { + .size = COREDUMP_ACK_SIZE_VER0 + PAGE_SIZE, + .mask = COREDUMP_REJECT | COREDUMP_WAIT, + }, + .bytes = COREDUMP_ACK_SIZE_VER0 + PAGE_SIZE, + .mark = COREDUMP_MARK_MAXSIZE, + }; + + check_refused_ack(_metadata, self, &refused); } /* @@ -2038,92 +1736,6 @@ TEST_F(coredump, socket_request_sparse_blob_upload) EXPECT_EQ(close(fd_core_file), 0); } -/* Ack @ack_mask, expect the kernel to refuse it as conflicting. */ -static void check_conflicting_ack(struct __test_metadata *const _metadata, - FIXTURE_DATA(coredump) *self, __u64 ack_mask) -{ - int pidfd, status; - pid_t pid, pid_coredump_server; - struct pidfd_info info = {}; - int ipc_sockets[2]; - char c; - - ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); - ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); - - pid_coredump_server = fork(); - ASSERT_GE(pid_coredump_server, 0); - if (pid_coredump_server == 0) { - int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; - int exit_code = EXIT_FAILURE; - struct coredump_req req = {}; - - close(ipc_sockets[0]); - - fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); - if (fd_server < 0) - goto out; - - if (write_nointr(ipc_sockets[1], "1", 1) < 0) - goto out; - - close(ipc_sockets[1]); - - fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); - if (fd_coredump < 0) - goto out; - - fd_peer_pidfd = get_peer_pidfd(fd_coredump); - if (fd_peer_pidfd < 0) - goto out; - - if (!read_coredump_req(fd_coredump, &req)) - goto out; - - if (!check_coredump_req(&req)) - goto out; - - if (!send_coredump_ack(fd_coredump, &req, ack_mask, 0)) - goto out; - - if (!read_marker(fd_coredump, COREDUMP_MARK_CONFLICTING)) - goto out; - - exit_code = EXIT_SUCCESS; -out: - if (fd_peer_pidfd >= 0) - close(fd_peer_pidfd); - if (fd_coredump >= 0) - close(fd_coredump); - if (fd_server >= 0) - close(fd_server); - _exit(exit_code); - } - self->pid_coredump_server = pid_coredump_server; - - EXPECT_EQ(close(ipc_sockets[1]), 0); - ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); - EXPECT_EQ(close(ipc_sockets[0]), 0); - - pid = fork(); - ASSERT_GE(pid, 0); - if (pid == 0) - crashing_child(); - - pidfd = sys_pidfd_open(pid, 0); - ASSERT_GE(pidfd, 0); - - waitpid(pid, &status, 0); - ASSERT_TRUE(WIFSIGNALED(status)); - ASSERT_FALSE(WCOREDUMP(status)); - - ASSERT_TRUE(get_pidfd_info(pidfd, &info)); - ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); - ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); - - wait_and_check_coredump_server(pid_coredump_server, _metadata, self); -} - /* COREDUMP_RECORDS applies to a coredump the kernel writes, nothing else. */ TEST_F(coredump, socket_request_records_without_kernel) { diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index d7cc448eeaf442..9aa901e14f02ee 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -1354,8 +1354,8 @@ bool read_coredump_req(int fd, struct coredump_req *req) return true; } -bool send_coredump_ack(int fd, const struct coredump_req *req, - __u64 mask, size_t size_ack) +/* Send @len bytes of @ack as they are, more than the struct if asked to. */ +bool send_coredump_ack_bytes(int fd, const struct coredump_ack *ack, size_t len) { ssize_t ret; /* @@ -1367,23 +1367,36 @@ bool send_coredump_ack(int fd, const struct coredump_req *req, char buffer[PAGE_SIZE]; } large_ack = {}; - if (!size_ack) - size_ack = sizeof(struct coredump_ack) < req->size_ack ? - sizeof(struct coredump_ack) : - req->size_ack; - large_ack.ack.mask = mask; - large_ack.ack.size = size_ack; - ret = send(fd, &large_ack, size_ack, MSG_NOSIGNAL); - if (ret != size_ack) { + if (len > sizeof(large_ack)) + return false; + + large_ack.ack = *ack; + ret = send(fd, &large_ack, len, MSG_NOSIGNAL); + if (ret != len) { fprintf(stderr, "%s: short send %zd: %m\n", __func__, ret); return false; } - fprintf(stderr, "Sent coredump ack with size %zu and mask 0x%llx\n", - size_ack, (unsigned long long)mask); + fprintf(stderr, "Sent %zu bytes of coredump ack: size %u, mask 0x%llx\n", + len, ack->size, (unsigned long long)ack->mask); return true; } +bool send_coredump_ack(int fd, const struct coredump_req *req, + __u64 mask, size_t size_ack) +{ + struct coredump_ack ack = { + .mask = mask, + }; + + if (!size_ack) + size_ack = sizeof(struct coredump_ack) < req->size_ack ? + sizeof(struct coredump_ack) : + req->size_ack; + ack.size = size_ack; + return send_coredump_ack_bytes(fd, &ack, size_ack); +} + /* Every option the kernel is expected to advertise in coredump_req->mask. */ #define TEST_REQ_MASK_ALL \ (COREDUMP_KERNEL | COREDUMP_USERSPACE | \ diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.h b/tools/testing/selftests/coredump/coredump_test_helpers.h index 97ad5cfeae928d..0970d3550fc1db 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.h +++ b/tools/testing/selftests/coredump/coredump_test_helpers.h @@ -46,6 +46,7 @@ bool read_marker(int fd, enum coredump_mark mark); bool read_coredump_req(int fd, struct coredump_req *req); bool send_coredump_ack(int fd, const struct coredump_req *req, __u64 mask, size_t size_ack); +bool send_coredump_ack_bytes(int fd, const struct coredump_ack *ack, size_t len); bool check_coredump_req(const struct coredump_req *req); int open_coredump_tmpfile(int fd_tmpfs_detached); void process_coredump_worker(int fd_coredump, int fd_peer_pidfd, int fd_core_file); From e50a26789232b8fdd625c2876281c01e8ec1cead Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Fri, 21 Aug 2026 13:52:05 +0200 Subject: [PATCH 280/857] selftests/coredump: test COREDUMP_MEMORY_TYPES Test the new COREDUMP_MEMORY_TYPES flag. Link: https://patch.msgid.link/20260821-work-coredump-filter-v1-4-91f9a73ef03e@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../coredump/coredump_socket_protocol_test.c | 349 ++++++++++++++++++ .../coredump/coredump_test_helpers.c | 203 +++++++++- .../coredump/coredump_test_helpers.h | 23 ++ 3 files changed, 568 insertions(+), 7 deletions(-) diff --git a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c index a07546e796517e..6dcd6c15a5653c 100644 --- a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c +++ b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c @@ -1932,4 +1932,353 @@ TEST_F(coredump, socket_request_stream_choice_large) ASSERT_LT(choice.received, choice.size / 8); } +/* What a memory types test asks of the kernel and what it expects back. */ +struct memory_choice { + /* Memory types the crashing child selects, or FILTER_TASK_INHERIT. */ + __u64 task_filter; + /* The ack. */ + __u64 mask; + __u64 memory_types; + size_t size_ack; + /* The shared mapping is in the coredump with all of its memory. */ + bool shared_dumped; + /* No memory at all. Pull a page from /proc//mem instead. */ + bool skeleton; +}; + +/* A skeleton still carries the vdso and friends, nothing bigger. */ +#define SKELETON_DATA_PAGES 16 + +/* + * The crashing child maps shared anonymous memory and tells the server + * where. The server acks with @choice and checks whether that mapping's + * segment in the coredump carries its memory. + */ +static void check_memory_dump(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, + const struct memory_choice *choice) +{ + int pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + int ipc_sockets[2]; + int addr_pipe[2]; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_EQ(pipe(addr_pipe), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_file = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + __u64 task_filter; + ElfW(Phdr) segment; + ssize_t received; + off_t size; + char *addr; + + close(ipc_sockets[0]); + close(addr_pipe[1]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + fd_file = open_coredump_tmpfile(self->fd_tmpfs_detached); + if (fd_file < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + /* The request reports the memory types the task selected. */ + if (!peer_coredump_filter(fd_peer_pidfd, &task_filter)) + goto out; + + if (req.memory_types != task_filter) { + fprintf(stderr, "Request reports 0x%llx, task selected 0x%llx\n", + (unsigned long long)req.memory_types, + (unsigned long long)task_filter); + goto out; + } + + if (choice->task_filter != FILTER_TASK_INHERIT && + task_filter != choice->task_filter) { + fprintf(stderr, "Task selected 0x%llx, child asked for 0x%llx\n", + (unsigned long long)task_filter, + (unsigned long long)choice->task_filter); + goto out; + } + + /* The child sent the address of its mapping before it crashed. */ + if (read_nointr(addr_pipe[0], &addr, sizeof(addr)) != sizeof(addr)) + goto out; + + if (!send_coredump_ack_types(fd_coredump, &req, choice->mask, + choice->memory_types, + choice->size_ack)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + if (choice->mask & COREDUMP_RECORDS) + received = recv_coredump_records(fd_coredump, fd_file, + &size, NULL, -1); + else + received = recv_coredump_bytes(fd_coredump, fd_file); + if (received < 0) + goto out; + + if (!is_elf_core(fd_file)) + goto out; + + /* A dump ending in holes or empty segments must still be whole. */ + if (!check_coredump_extent(fd_file)) + goto out; + + if (!find_coredump_segment(fd_file, (__u64)(uintptr_t)addr, &segment)) + goto out; + + if (segment.p_memsz != MEMORY_MAPPING_SIZE) { + fprintf(stderr, "Segment spans %llu bytes, the mapping %u\n", + (unsigned long long)segment.p_memsz, + MEMORY_MAPPING_SIZE); + goto out; + } + + if (segment.p_filesz != (choice->shared_dumped ? segment.p_memsz : 0)) { + fprintf(stderr, "Segment carries %llu bytes, expected %s of them\n", + (unsigned long long)segment.p_filesz, + choice->shared_dumped ? "all" : "none"); + goto out; + } + + if (choice->skeleton) { + __u64 data, notes, data_max; + char buf[PAGE_SIZE]; + + if (!sum_coredump_segments(fd_file, &data, ¬es)) + goto out; + + data_max = SKELETON_DATA_PAGES * sysconf(_SC_PAGESIZE); + if (!notes || data > data_max) { + fprintf(stderr, "Skeleton has %llu note and %llu memory bytes\n", + (unsigned long long)notes, + (unsigned long long)data); + goto out; + } + + /* The task is parked in COREDUMP_WAIT with its memory. */ + if (peer_read_mem(fd_peer_pidfd, (__u64)(uintptr_t)addr, + buf, sizeof(buf)) != sizeof(buf)) + goto out; + + if (buf[0] != 'x') { + fprintf(stderr, "Pulled memory lacks the child's mark\n"); + goto out; + } + + fprintf(stderr, "Skeleton of %zd bytes, pulled %zu bytes of memory\n", + received, sizeof(buf)); + } + + exit_code = EXIT_SUCCESS; +out: + close(addr_pipe[0]); + if (fd_file >= 0) + close(fd_file); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + EXPECT_EQ(close(addr_pipe[0]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_memory(choice->task_filter, addr_pipe[1]); + EXPECT_EQ(close(addr_pipe[1]), 0); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); +} + +/* Without COREDUMP_MEMORY_TYPES the task's own selection decides. */ +TEST_F(coredump, socket_request_memory_types_task_includes) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .mask = COREDUMP_KERNEL, + .shared_dumped = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +TEST_F(coredump, socket_request_memory_types_task_excludes) +{ + struct memory_choice choice = { + .task_filter = 0, + .mask = COREDUMP_KERNEL, + .shared_dumped = false, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* The server drops a memory type the task would have dumped. */ +TEST_F(coredump, socket_request_memory_types_restricts) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE, + .shared_dumped = false, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* The server adds a memory type the task had excluded. */ +TEST_F(coredump, socket_request_memory_types_widens) +{ + struct memory_choice choice = { + .task_filter = 0, + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .shared_dumped = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* The memory types decide what goes into a record stream just the same. */ +TEST_F(coredump, socket_request_memory_types_records) +{ + struct memory_choice choice = { + .task_filter = FILTER_TASK_INHERIT, + .mask = COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE, + .shared_dumped = false, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* + * An empty selection leaves a skeleton: every program header and every note + * but no memory. A server that wants to pick the memory itself reads it + * from /proc//mem while the task waits for it to finish. + */ +TEST_F(coredump, socket_request_memory_types_skeleton) +{ + struct memory_choice choice = { + .task_filter = FILTER_TASK_INHERIT, + .mask = COREDUMP_KERNEL | COREDUMP_WAIT | COREDUMP_MEMORY_TYPES, + .memory_types = 0, + .shared_dumped = false, + .skeleton = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* A memory type the kernel didn't advertise in memory_types_mask. */ +TEST_F(coredump, socket_request_memory_types_unknown_bit) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + .memory_types = 1ULL << 63, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_UNSUPPORTED, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* The memory types must be zero unless COREDUMP_MEMORY_TYPES is raised. */ +TEST_F(coredump, socket_request_memory_types_stale_field) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = COREDUMP_KERNEL, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_UNSUPPORTED, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* COREDUMP_MEMORY_TYPES needs an ack that has the memory types. */ +TEST_F(coredump, socket_request_memory_types_short_ack) +{ + struct refused_ack refused = { + .ack = { + .size = COREDUMP_ACK_SIZE_VER0, + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + }, + .bytes = COREDUMP_ACK_SIZE_VER0, + .mark = COREDUMP_MARK_MINSIZE, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* The memory types select what the kernel writes, nothing else. */ +TEST_F(coredump, socket_request_memory_types_without_kernel) +{ + check_conflicting_ack(_metadata, self, COREDUMP_USERSPACE | COREDUMP_MEMORY_TYPES); +} + TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 9aa901e14f02ee..7ae0c6c458aa13 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -17,6 +17,7 @@ #include #include #include +#include #include #include #include @@ -73,6 +74,48 @@ void crashing_child_sparse(size_t size) *(volatile int *)NULL = 0; } +/* Select @types through the caller's own /proc/self/coredump_filter. */ +static bool set_coredump_filter(__u64 types) +{ + char buf[32]; + int fd, len; + bool ok; + + fd = open("/proc/self/coredump_filter", O_WRONLY | O_CLOEXEC); + if (fd < 0) + return false; + + len = snprintf(buf, sizeof(buf), "0x%llx", (unsigned long long)types); + ok = write_nointr(fd, buf, len) == len; + close(fd); + return ok; +} + +/* + * Map shared anonymous memory, touch it, tell the server where it is and + * crash. A @task_filter other than FILTER_TASK_INHERIT is selected first. + */ +void crashing_child_memory(__u64 task_filter, int fd_addr) +{ + char *p; + + if (task_filter != FILTER_TASK_INHERIT && !set_coredump_filter(task_filter)) + _exit(EXIT_FAILURE); + + p = mmap(NULL, MEMORY_MAPPING_SIZE, PROT_READ | PROT_WRITE, + MAP_SHARED | MAP_ANONYMOUS, -1, 0); + if (p == MAP_FAILED) + _exit(EXIT_FAILURE); + p[0] = 'x'; + + if (write_nointr(fd_addr, &p, sizeof(p)) != sizeof(p)) + _exit(EXIT_FAILURE); + close(fd_addr); + + /* crash on purpose */ + *(volatile int *)NULL = 0; +} + /* Sink a reassembled record stream is handed to, record by record. */ struct coredump_record_sink { /* @len bytes of coredump data that belong at @offset. */ @@ -916,6 +959,82 @@ static const ElfW(Phdr) *find_segment(const ElfW(Phdr) *phdr, size_t nr, return NULL; } +/* The PT_LOAD segment @vaddr falls into. */ +bool find_coredump_segment(int fd, __u64 vaddr, ElfW(Phdr) *segment) +{ + const ElfW(Phdr) *found; + ElfW(Phdr) *phdr; + size_t nr; + + phdr = read_phdrs(fd, &nr); + if (!phdr) + return false; + + found = find_segment(phdr, nr, vaddr); + if (found) + *segment = *found; + else + fprintf(stderr, "%s: no segment for 0x%llx\n", __func__, + (unsigned long long)vaddr); + + free(phdr); + return found; +} + +/* How many bytes the PT_LOAD and the PT_NOTE segments of @fd carry. */ +bool sum_coredump_segments(int fd, __u64 *data, __u64 *notes) +{ + ElfW(Phdr) *phdr; + size_t nr, i; + + phdr = read_phdrs(fd, &nr); + if (!phdr) + return false; + + *data = 0; + *notes = 0; + for (i = 0; i < nr; i++) { + if (phdr[i].p_type == PT_LOAD) + *data += phdr[i].p_filesz; + else if (phdr[i].p_type == PT_NOTE) + *notes += phdr[i].p_filesz; + } + + free(phdr); + return true; +} + +/* The coredump in @fd is at least as long as every segment it declares. */ +bool check_coredump_extent(int fd) +{ + ElfW(Phdr) *phdr; + struct stat st; + size_t nr, i; + bool ok = true; + + if (fstat(fd, &st)) { + fprintf(stderr, "%s: fstat: %m\n", __func__); + return false; + } + + phdr = read_phdrs(fd, &nr); + if (!phdr) + return false; + + for (i = 0; i < nr; i++) { + if (phdr[i].p_offset + phdr[i].p_filesz <= (__u64)st.st_size) + continue; + fprintf(stderr, "%s: segment %zu ends at %llu, the coredump at %llu\n", + __func__, i, + (unsigned long long)(phdr[i].p_offset + phdr[i].p_filesz), + (unsigned long long)st.st_size); + ok = false; + } + + free(phdr); + return ok; +} + /* The next stretch of memory the segments cover, split ones merged back. */ static bool next_range(const ElfW(Phdr) *phdr, size_t nr, size_t *i, __u64 *start, __u64 *end) @@ -1250,6 +1369,62 @@ ssize_t peer_vm_size(int fd_peer_pidfd) /* Protocol helper functions */ +/* The peer's /proc//coredump_filter, which is in memory types. */ +bool peer_coredump_filter(int fd_peer_pidfd, __u64 *memory_types) +{ + struct pidfd_info info = {}; + unsigned long value; + char path[64]; + FILE *f; + int ret; + + if (!get_pidfd_info(fd_peer_pidfd, &info)) + return false; + + snprintf(path, sizeof(path), "/proc/%d/coredump_filter", info.pid); + f = fopen(path, "r"); + if (!f) { + fprintf(stderr, "%s: %s: %m\n", __func__, path); + return false; + } + + ret = fscanf(f, "%lx", &value); + fclose(f); + if (ret != 1) { + fprintf(stderr, "%s: %s: no value\n", __func__, path); + return false; + } + + *memory_types = value; + return true; +} + +/* Read @len bytes at @addr from the peer's /proc//mem. */ +ssize_t peer_read_mem(int fd_peer_pidfd, __u64 addr, void *buf, size_t len) +{ + struct pidfd_info info = {}; + char path[64]; + ssize_t ret; + int fd; + + if (!get_pidfd_info(fd_peer_pidfd, &info)) + return -1; + + snprintf(path, sizeof(path), "/proc/%d/mem", info.pid); + fd = open(path, O_RDONLY | O_CLOEXEC); + if (fd < 0) { + fprintf(stderr, "%s: %s: %m\n", __func__, path); + return -1; + } + + ret = pread(fd, buf, len, addr); + if (ret < 0) + fprintf(stderr, "%s: %s at 0x%llx: %m\n", __func__, path, + (unsigned long long)addr); + close(fd); + return ret; +} + ssize_t recv_marker(int fd) { enum coredump_mark mark = COREDUMP_MARK_REQACK; @@ -1377,16 +1552,18 @@ bool send_coredump_ack_bytes(int fd, const struct coredump_ack *ack, size_t len) return false; } - fprintf(stderr, "Sent %zu bytes of coredump ack: size %u, mask 0x%llx\n", - len, ack->size, (unsigned long long)ack->mask); + fprintf(stderr, "Sent %zu bytes of coredump ack: size %u, mask 0x%llx, types 0x%llx\n", + len, ack->size, (unsigned long long)ack->mask, + (unsigned long long)ack->memory_types); return true; } -bool send_coredump_ack(int fd, const struct coredump_req *req, - __u64 mask, size_t size_ack) +bool send_coredump_ack_types(int fd, const struct coredump_req *req, + __u64 mask, __u64 memory_types, size_t size_ack) { struct coredump_ack ack = { .mask = mask, + .memory_types = memory_types, }; if (!size_ack) @@ -1397,17 +1574,23 @@ bool send_coredump_ack(int fd, const struct coredump_req *req, return send_coredump_ack_bytes(fd, &ack, size_ack); } +bool send_coredump_ack(int fd, const struct coredump_req *req, + __u64 mask, size_t size_ack) +{ + return send_coredump_ack_types(fd, req, mask, 0, size_ack); +} + /* Every option the kernel is expected to advertise in coredump_req->mask. */ #define TEST_REQ_MASK_ALL \ (COREDUMP_KERNEL | COREDUMP_USERSPACE | \ COREDUMP_REJECT | COREDUMP_WAIT | \ - COREDUMP_RECORDS | COREDUMP_SPARSE) + COREDUMP_RECORDS | COREDUMP_SPARSE | COREDUMP_MEMORY_TYPES) bool check_coredump_req(const struct coredump_req *req) { - if (req->size < COREDUMP_REQ_SIZE_VER0) { + if (req->size < COREDUMP_REQ_SIZE_VER1) { fprintf(stderr, "%s: size %u below minimum %d\n", - __func__, req->size, COREDUMP_REQ_SIZE_VER0); + __func__, req->size, COREDUMP_REQ_SIZE_VER1); return false; } if (req->mask != TEST_REQ_MASK_ALL) { @@ -1416,6 +1599,12 @@ bool check_coredump_req(const struct coredump_req *req) (unsigned long long)TEST_REQ_MASK_ALL); return false; } + if (req->memory_types_mask != TEST_MEMORY_ALL) { + fprintf(stderr, "%s: memory_types_mask 0x%llx, expected 0x%llx\n", + __func__, (unsigned long long)req->memory_types_mask, + (unsigned long long)TEST_MEMORY_ALL); + return false; + } return true; } diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.h b/tools/testing/selftests/coredump/coredump_test_helpers.h index 0970d3550fc1db..fc21b8620359ed 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.h +++ b/tools/testing/selftests/coredump/coredump_test_helpers.h @@ -3,6 +3,7 @@ #ifndef __COREDUMP_TEST_HELPERS_H #define __COREDUMP_TEST_HELPERS_H +#include #include #include #include @@ -21,10 +22,30 @@ /* A task mapping at least this much is worth a record stream. */ #define SPARSE_STREAM_THRESHOLD (SPARSE_MAPPING_SIZE / 2) +/* Size of the shared anonymous mapping the memory types tests map. */ +#define MEMORY_MAPPING_SIZE (4 * 1024 * 1024) + +/* Leave the coredump_filter the crashing child inherited alone. */ +#define FILTER_TASK_INHERIT ((__u64)-1) + +/* Every memory type the kernel is expected to advertise. */ +#define TEST_MEMORY_ALL \ + (COREDUMP_MEMORY_ANON_PRIVATE | COREDUMP_MEMORY_ANON_SHARED | \ + COREDUMP_MEMORY_FILE_PRIVATE | COREDUMP_MEMORY_FILE_SHARED | \ + COREDUMP_MEMORY_ELF_HEADERS | \ + COREDUMP_MEMORY_HUGETLB_PRIVATE | COREDUMP_MEMORY_HUGETLB_SHARED | \ + COREDUMP_MEMORY_DAX_PRIVATE | COREDUMP_MEMORY_DAX_SHARED) + /* Shared helper function declarations */ void *do_nothing(void *arg); void crashing_child(void); void crashing_child_sparse(size_t size); +void crashing_child_memory(__u64 task_filter, int fd_addr); +bool find_coredump_segment(int fd, __u64 vaddr, ElfW(Phdr) *segment); +bool sum_coredump_segments(int fd, __u64 *data, __u64 *notes); +bool check_coredump_extent(int fd); +bool peer_coredump_filter(int fd_peer_pidfd, __u64 *memory_types); +ssize_t peer_read_mem(int fd_peer_pidfd, __u64 addr, void *buf, size_t len); ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, off_t *coredump_size, bool *truncated, int fd_peer_pidfd); @@ -46,6 +67,8 @@ bool read_marker(int fd, enum coredump_mark mark); bool read_coredump_req(int fd, struct coredump_req *req); bool send_coredump_ack(int fd, const struct coredump_req *req, __u64 mask, size_t size_ack); +bool send_coredump_ack_types(int fd, const struct coredump_req *req, + __u64 mask, __u64 memory_types, size_t size_ack); bool send_coredump_ack_bytes(int fd, const struct coredump_ack *ack, size_t len); bool check_coredump_req(const struct coredump_req *req); int open_coredump_tmpfile(int fd_tmpfs_detached); From 6f539c2a83160ac04bbd5dff210d7ee3a6a370d5 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Fri, 21 Aug 2026 13:52:06 +0200 Subject: [PATCH 281/857] selftests/coredump: improve coredump size negotiation tests Improve the size handling tests when negotiating a coredump through req and ack. Link: https://patch.msgid.link/20260821-work-coredump-filter-v1-5-91f9a73ef03e@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../coredump/coredump_socket_protocol_test.c | 158 ++++++++++++++++-- .../coredump/coredump_test_helpers.c | 17 +- .../coredump/coredump_test_helpers.h | 1 + 3 files changed, 155 insertions(+), 21 deletions(-) diff --git a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c index 6dcd6c15a5653c..6c7327832d444d 100644 --- a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c +++ b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c @@ -1932,11 +1932,74 @@ TEST_F(coredump, socket_request_stream_choice_large) ASSERT_LT(choice.received, choice.size / 8); } +/* What a coredump server was built with. */ +struct server_build { + /* sizeof(struct coredump_req) and sizeof(struct coredump_ack) back then. */ + size_t req_size; + size_t ack_size; + /* The features it raises if the kernel offers them. */ + __u64 wants; + /* Its policy: what it drops from and adds to the task's selection. */ + __u64 drop; + __u64 add; +}; + +/* A server from when the structs were first published: kernel-written dumps. */ +static const struct server_build server_build_ver0 = { + .req_size = COREDUMP_REQ_SIZE_VER0, + .ack_size = COREDUMP_ACK_SIZE_VER0, + .wants = COREDUMP_KERNEL, +}; + +/* A server built against this header: no shared memory, always the ELF headers. */ +static const struct server_build server_build_ver1 = { + .req_size = sizeof(struct coredump_req), + .ack_size = sizeof(struct coredump_ack), + .wants = COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .drop = COREDUMP_MEMORY_ANON_SHARED | COREDUMP_MEMORY_FILE_SHARED, + .add = COREDUMP_MEMORY_ELF_HEADERS, +}; + +/* + * Build the ack the way a server does: from what the kernel offers, what + * this build implements, and what fits in the ack the kernel accepts. + * Fields the build never read are zero and never consulted. + */ +static void negotiate(const struct coredump_req *req, + const struct server_build *build, + struct coredump_ack *ack) +{ + __u64 offered = req->mask & build->wants; + + memset(ack, 0, sizeof(*ack)); + ack->size = build->ack_size < req->size_ack ? build->ack_size : req->size_ack; + /* These builds only ever have the kernel write the coredump. */ + ack->mask = COREDUMP_KERNEL; + + /* Sparse needs records, records need the kernel to write. */ + if (offered & COREDUMP_RECORDS) { + ack->mask |= COREDUMP_RECORDS; + if (offered & COREDUMP_SPARSE) + ack->mask |= COREDUMP_SPARSE; + } + + /* The memory types need an ack that carries them. */ + if ((offered & COREDUMP_MEMORY_TYPES) && ack->size >= COREDUMP_ACK_SIZE_VER1) { + ack->mask |= COREDUMP_MEMORY_TYPES; + /* Start from the task's selection; only advertised types pass. */ + ack->memory_types = (req->memory_types & ~build->drop) | build->add; + ack->memory_types &= req->memory_types_mask; + } +} + /* What a memory types test asks of the kernel and what it expects back. */ struct memory_choice { /* Memory types the crashing child selects, or FILTER_TASK_INHERIT. */ __u64 task_filter; - /* The ack. */ + /* Negotiate the ack as this server build, NULL to send it as given. */ + const struct server_build *build; + /* The ack, or what the negotiation must arrive at. */ __u64 mask; __u64 memory_types; size_t size_ack; @@ -1976,6 +2039,13 @@ static void check_memory_dump(struct __test_metadata *const _metadata, int fd_file = -1; int exit_code = EXIT_FAILURE; struct coredump_req req = {}; + struct coredump_ack ack = { + .size = choice->size_ack, + .mask = choice->mask, + .memory_types = choice->memory_types, + }; + /* How much of the request this server reads. */ + size_t req_size = choice->build ? choice->build->req_size : sizeof(req); __u64 task_filter; ElfW(Phdr) segment; ssize_t received; @@ -2006,21 +2076,24 @@ static void check_memory_dump(struct __test_metadata *const _metadata, if (fd_file < 0) goto out; - if (!read_coredump_req(fd_coredump, &req)) + if (!read_coredump_req_sized(fd_coredump, &req, req_size)) goto out; - if (!check_coredump_req(&req)) - goto out; - - /* The request reports the memory types the task selected. */ if (!peer_coredump_filter(fd_peer_pidfd, &task_filter)) goto out; - if (req.memory_types != task_filter) { - fprintf(stderr, "Request reports 0x%llx, task selected 0x%llx\n", - (unsigned long long)req.memory_types, - (unsigned long long)task_filter); - goto out; + /* A build from before the memory types never read that far. */ + if (req_size >= COREDUMP_REQ_SIZE_VER1) { + if (!check_coredump_req(&req)) + goto out; + + /* The request reports the memory types the task selected. */ + if (req.memory_types != task_filter) { + fprintf(stderr, "Request reports 0x%llx, task selected 0x%llx\n", + (unsigned long long)req.memory_types, + (unsigned long long)task_filter); + goto out; + } } if (choice->task_filter != FILTER_TASK_INHERIT && @@ -2035,15 +2108,28 @@ static void check_memory_dump(struct __test_metadata *const _metadata, if (read_nointr(addr_pipe[0], &addr, sizeof(addr)) != sizeof(addr)) goto out; - if (!send_coredump_ack_types(fd_coredump, &req, choice->mask, - choice->memory_types, - choice->size_ack)) + /* A server build negotiates its ack and must arrive at the choice. */ + if (choice->build) { + negotiate(&req, choice->build, &ack); + + if (ack.size != choice->size_ack || ack.mask != choice->mask || + ack.memory_types != choice->memory_types) { + fprintf(stderr, + "Negotiated %u bytes, mask 0x%llx, types 0x%llx\n", + ack.size, (unsigned long long)ack.mask, + (unsigned long long)ack.memory_types); + goto out; + } + } + + if (!send_coredump_ack_types(fd_coredump, &req, ack.mask, + ack.memory_types, ack.size)) goto out; if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) goto out; - if (choice->mask & COREDUMP_RECORDS) + if (ack.mask & COREDUMP_RECORDS) received = recv_coredump_records(fd_coredump, fd_file, &size, NULL, -1); else @@ -2281,4 +2367,46 @@ TEST_F(coredump, socket_request_memory_types_without_kernel) check_conflicting_ack(_metadata, self, COREDUMP_USERSPACE | COREDUMP_MEMORY_TYPES); } +/* + * A server built with the first structs reads the request it knows, + * discards the rest and acks with the ack it knows. It raises nothing + * it wasn't built for and the kernel dumps what the task selected. + */ +TEST_F(coredump, socket_request_negotiate_ver0) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .build = &server_build_ver0, + .mask = COREDUMP_KERNEL, + .memory_types = 0, + .size_ack = COREDUMP_ACK_SIZE_VER0, + .shared_dumped = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* + * A server built against this header takes every feature the kernel + * offers, drops shared memory from what the task selected and adds the + * ELF headers. + */ +TEST_F(coredump, socket_request_negotiate_ver1) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .build = &server_build_ver1, + .mask = COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ELF_HEADERS, + .size_ack = COREDUMP_ACK_SIZE_VER1, + .shared_dumped = false, + }; + + check_memory_dump(_metadata, self, &choice); +} + TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 7ae0c6c458aa13..ab94c45cd8be61 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -1467,10 +1467,11 @@ bool read_marker(int fd, enum coredump_mark mark) return ret == mark; } -bool read_coredump_req(int fd, struct coredump_req *req) +/* Read the request as a server built with a @user_size byte struct does. */ +bool read_coredump_req_sized(int fd, struct coredump_req *req, size_t user_size) { ssize_t ret; - size_t field_size, user_size, known_size, kernel_size, remaining_size; + size_t field_size, known_size, kernel_size, remaining_size; memset(req, 0, sizeof(*req)); field_size = sizeof(req->size); @@ -1478,25 +1479,24 @@ bool read_coredump_req(int fd, struct coredump_req *req) /* Peek the size of the coredump request. */ ret = recv(fd, req, field_size, MSG_PEEK | MSG_WAITALL); if (ret != field_size) { - fprintf(stderr, "read_coredump_req: peek failed (got %zd, expected %zu): %m\n", + fprintf(stderr, "%s: peek failed (got %zd, expected %zu): %m\n", __func__, ret, field_size); return false; } kernel_size = req->size; if (kernel_size < COREDUMP_REQ_SIZE_VER0) { - fprintf(stderr, "read_coredump_req: kernel_size %zu < min %d\n", + fprintf(stderr, "%s: kernel_size %zu < min %d\n", __func__, kernel_size, COREDUMP_REQ_SIZE_VER0); return false; } if (kernel_size >= PAGE_SIZE) { - fprintf(stderr, "read_coredump_req: kernel_size %zu >= PAGE_SIZE %d\n", + fprintf(stderr, "%s: kernel_size %zu >= PAGE_SIZE %d\n", __func__, kernel_size, PAGE_SIZE); return false; } /* Consume as much of the request as we know about. */ - user_size = sizeof(struct coredump_req); known_size = user_size < kernel_size ? user_size : kernel_size; ret = recv(fd, req, known_size, MSG_WAITALL); if (ret != known_size) @@ -1529,6 +1529,11 @@ bool read_coredump_req(int fd, struct coredump_req *req) return true; } +bool read_coredump_req(int fd, struct coredump_req *req) +{ + return read_coredump_req_sized(fd, req, sizeof(*req)); +} + /* Send @len bytes of @ack as they are, more than the struct if asked to. */ bool send_coredump_ack_bytes(int fd, const struct coredump_ack *ack, size_t len) { diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.h b/tools/testing/selftests/coredump/coredump_test_helpers.h index fc21b8620359ed..8e0187645c930d 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.h +++ b/tools/testing/selftests/coredump/coredump_test_helpers.h @@ -65,6 +65,7 @@ bool get_pidfd_info(int fd_peer_pidfd, struct pidfd_info *info); ssize_t recv_marker(int fd); bool read_marker(int fd, enum coredump_mark mark); bool read_coredump_req(int fd, struct coredump_req *req); +bool read_coredump_req_sized(int fd, struct coredump_req *req, size_t user_size); bool send_coredump_ack(int fd, const struct coredump_req *req, __u64 mask, size_t size_ack); bool send_coredump_ack_types(int fd, const struct coredump_req *req, From 35aa12ea27c46c3b732a776cfa42d85591e75861 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Fri, 21 Aug 2026 13:52:07 +0200 Subject: [PATCH 282/857] selftests/coredump: test failed handshakes Add more coredump refusal tests. Link: https://patch.msgid.link/20260821-work-coredump-filter-v1-6-91f9a73ef03e@kernel.org Signed-off-by: Christian Brauner (Amutable) --- .../coredump/coredump_socket_protocol_test.c | 211 +++++++++++++++++- .../coredump/coredump_test_helpers.c | 32 ++- .../coredump/coredump_test_helpers.h | 1 + 3 files changed, 238 insertions(+), 6 deletions(-) diff --git a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c index 6c7327832d444d..f5c9bad8754634 100644 --- a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c +++ b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c @@ -510,14 +510,15 @@ TEST_F(coredump, socket_request_reject) /* An ack the kernel must refuse and how. */ struct refused_ack { - /* The ack and how many bytes of it the server sends. */ + /* The ack, and how many bytes of it the server sends before it hangs up. */ struct coredump_ack ack; size_t bytes; - /* The marker the kernel answers with. */ + /* The marker the kernel answers with, or none if @no_marker. */ enum coredump_mark mark; + bool no_marker; }; -/* Send @refused, expect the kernel to refuse it with the marker. */ +/* Send @refused, expect the kernel to refuse it and hang up. */ static void check_refused_ack(struct __test_metadata *const _metadata, FIXTURE_DATA(coredump) *self, const struct refused_ack *refused) @@ -577,7 +578,16 @@ static void check_refused_ack(struct __test_metadata *const _metadata, refused->bytes)) goto out; - if (!read_marker(fd_coredump, refused->mark)) + /* Nothing more to say. A server that died looks the same. */ + if (shutdown(fd_coredump, SHUT_WR)) + goto out; + + if (!refused->no_marker && + !read_marker(fd_coredump, refused->mark)) + goto out; + + /* The kernel hangs up after a refusal, marker or not. */ + if (!read_hangup(fd_coredump)) goto out; exit_code = EXIT_SUCCESS; @@ -2409,4 +2419,197 @@ TEST_F(coredump, socket_request_negotiate_ver1) check_memory_dump(_metadata, self, &choice); } +/* An ack that picks none of KERNEL, USERSPACE and REJECT. */ +TEST_F(coredump, socket_request_no_mode) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = COREDUMP_WAIT, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_CONFLICTING, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* @spare must be zero, like every field that isn't in use. */ +TEST_F(coredump, socket_request_spare) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .spare = 1, + .mask = COREDUMP_KERNEL, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_UNSUPPORTED, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* An ack size is a byte count. One that ends inside a field is valid. */ +#define ACK_SIZE_BETWEEN (COREDUMP_ACK_SIZE_VER0 + sizeof(__u32)) + +/* Any size from VER0 up to what the kernel accepts works without memory types. */ +TEST_F(coredump, socket_request_ack_size_between) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .mask = COREDUMP_KERNEL, + .size_ack = ACK_SIZE_BETWEEN, + .shared_dumped = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* The memory types need the whole field, not the part that happens to fit. */ +TEST_F(coredump, socket_request_memory_types_ack_size_between) +{ + struct refused_ack refused = { + .ack = { + .size = ACK_SIZE_BETWEEN, + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + }, + .bytes = ACK_SIZE_BETWEEN, + .mark = COREDUMP_MARK_MINSIZE, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* A server that hangs up without acking gets no marker and no coredump. */ +TEST_F(coredump, socket_request_server_hangs_up) +{ + struct refused_ack refused = { + .bytes = 0, + .no_marker = true, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* A server that hangs up in the middle of its ack looks the same. */ +TEST_F(coredump, socket_request_ack_truncated) +{ + struct refused_ack refused = { + .ack = { + .size = COREDUMP_ACK_SIZE_VER0, + .mask = COREDUMP_KERNEL, + }, + .bytes = COREDUMP_ACK_SIZE_VER0 / 2, + .no_marker = true, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* + * The kernels a server built against this header can't meet here: + * negotiate() against their requests, no coredump involved. + */ + +/* The request of a kernel with the first structs and features. */ +static const struct coredump_req req_ver0 = { + .size = COREDUMP_REQ_SIZE_VER0, + .size_ack = COREDUMP_ACK_SIZE_VER0, + .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | + COREDUMP_REJECT | COREDUMP_WAIT, +}; + +/* The request of this kernel. */ +static const struct coredump_req req_ver1 = { + .size = COREDUMP_REQ_SIZE_VER1, + .size_ack = COREDUMP_ACK_SIZE_VER1, + .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | + COREDUMP_REJECT | COREDUMP_WAIT | + COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .memory_types_mask = TEST_MEMORY_ALL, +}; + +/* A kernel with the first structs gets the first ack and nothing newer. */ +TEST(negotiate_ver0_kernel) +{ + struct coredump_ack ack; + + negotiate(&req_ver0, &server_build_ver1, &ack); + ASSERT_EQ(ack.size, COREDUMP_ACK_SIZE_VER0); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL); + ASSERT_EQ(ack.memory_types, 0); +} + +/* A kernel with records and sparse but the first structs: both, no types. */ +TEST(negotiate_sparse_kernel) +{ + struct coredump_req req = req_ver0; + struct coredump_ack ack; + + req.mask |= COREDUMP_RECORDS | COREDUMP_SPARSE; + negotiate(&req, &server_build_ver1, &ack); + ASSERT_EQ(ack.size, COREDUMP_ACK_SIZE_VER0); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE); + ASSERT_EQ(ack.memory_types, 0); +} + +/* Records without sparse: sparse isn't raised on its own. */ +TEST(negotiate_records_without_sparse) +{ + struct coredump_req req = req_ver0; + struct coredump_ack ack; + + req.mask |= COREDUMP_RECORDS; + negotiate(&req, &server_build_ver1, &ack); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL | COREDUMP_RECORDS); +} + +/* + * A feature whose ack field lies past what the kernel accepts can't be + * raised. No kernel offers the memory types without the room for them, so a + * request that does stands in for a feature newer than this header. + */ +TEST(negotiate_types_need_room) +{ + struct coredump_req req = req_ver0; + struct coredump_ack ack; + + req.mask |= COREDUMP_MEMORY_TYPES; + negotiate(&req, &server_build_ver1, &ack); + ASSERT_EQ(ack.size, COREDUMP_ACK_SIZE_VER0); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL); + ASSERT_EQ(ack.memory_types, 0); +} + +/* This kernel: the policy applied to the task's selection. */ +TEST(negotiate_ver1_kernel) +{ + struct coredump_ack ack; + + negotiate(&req_ver1, &server_build_ver1, &ack); + ASSERT_EQ(ack.size, COREDUMP_ACK_SIZE_VER1); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES); + ASSERT_EQ(ack.memory_types, COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ELF_HEADERS); +} + +/* A kernel that doesn't know a type the policy adds isn't asked for it. */ +TEST(negotiate_unknown_type) +{ + struct coredump_req req = req_ver1; + struct coredump_ack ack; + + req.memory_types_mask &= ~(__u64)COREDUMP_MEMORY_ELF_HEADERS; + negotiate(&req, &server_build_ver1, &ack); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES); + ASSERT_EQ(ack.memory_types, COREDUMP_MEMORY_ANON_PRIVATE); +} + TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index ab94c45cd8be61..4e36e3e4fb78f3 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -1467,6 +1467,29 @@ bool read_marker(int fd, enum coredump_mark mark) return ret == mark; } +/* + * The kernel hung up without sending anything more: end of stream, or a + * reset if it refused the ack on its peeked size and never read it. + */ +bool read_hangup(int fd) +{ + ssize_t ret; + char c; + + ret = recv(fd, &c, sizeof(c), MSG_WAITALL); + if (ret == 0) { + fprintf(stderr, "Kernel closed the connection\n"); + return true; + } + if (ret < 0 && errno == ECONNRESET) { + fprintf(stderr, "Kernel closed the connection with the ack unread\n"); + return true; + } + + fprintf(stderr, "%s: expected a hangup, got %zd: %m\n", __func__, ret); + return false; +} + /* Read the request as a server built with a @user_size byte struct does. */ bool read_coredump_req_sized(int fd, struct coredump_req *req, size_t user_size) { @@ -1593,11 +1616,16 @@ bool send_coredump_ack(int fd, const struct coredump_req *req, bool check_coredump_req(const struct coredump_req *req) { - if (req->size < COREDUMP_REQ_SIZE_VER1) { - fprintf(stderr, "%s: size %u below minimum %d\n", + if (req->size != COREDUMP_REQ_SIZE_VER1) { + fprintf(stderr, "%s: size %u, expected %d\n", __func__, req->size, COREDUMP_REQ_SIZE_VER1); return false; } + if (req->size_ack != COREDUMP_ACK_SIZE_VER1) { + fprintf(stderr, "%s: size_ack %u, expected %d\n", + __func__, req->size_ack, COREDUMP_ACK_SIZE_VER1); + return false; + } if (req->mask != TEST_REQ_MASK_ALL) { fprintf(stderr, "%s: mask 0x%llx, expected 0x%llx\n", __func__, (unsigned long long)req->mask, diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.h b/tools/testing/selftests/coredump/coredump_test_helpers.h index 8e0187645c930d..3f2f87837558e8 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.h +++ b/tools/testing/selftests/coredump/coredump_test_helpers.h @@ -64,6 +64,7 @@ bool get_pidfd_info(int fd_peer_pidfd, struct pidfd_info *info); /* Protocol helper function declarations */ ssize_t recv_marker(int fd); bool read_marker(int fd, enum coredump_mark mark); +bool read_hangup(int fd); bool read_coredump_req(int fd, struct coredump_req *req); bool read_coredump_req_sized(int fd, struct coredump_req *req, size_t user_size); bool send_coredump_ack(int fd, const struct coredump_req *req, From f096faa6bb169293d04194a0a5d84bc782c683b5 Mon Sep 17 00:00:00 2001 From: Namjae Jeon Date: Mon, 31 Aug 2026 16:55:05 +0900 Subject: [PATCH 283/857] exfat: doc: add documentation Add documentation for the Linux exFAT filesystem driver, including supported mount options and exfatprogs. Signed-off-by: Namjae Jeon --- Documentation/filesystems/exfat.rst | 117 ++++++++++++++++++++++++++++ Documentation/filesystems/index.rst | 1 + 2 files changed, 118 insertions(+) create mode 100644 Documentation/filesystems/exfat.rst diff --git a/Documentation/filesystems/exfat.rst b/Documentation/filesystems/exfat.rst new file mode 100644 index 00000000000000..ce5c9344a7b23d --- /dev/null +++ b/Documentation/filesystems/exfat.rst @@ -0,0 +1,117 @@ +.. SPDX-License-Identifier: GPL-2.0 + +================================== +The Linux exFAT filesystem driver +================================== + + +.. Table of contents + + - Overview + - Utilities support + - Supported mount options + + +Overview +======== + +exFAT is a filesystem designed for removable storage and other devices that +need to store large files. The Linux exFAT filesystem driver provides read +and write support for exFAT volumes. + +To mount an exFAT volume, use the ``exfat`` filesystem type:: + + mount -t exfat /dev/sdX1 /mnt + + +Utilities support +================= + +The exfatprogs project provides userspace utilities for creating, checking, +repairing, inspecting, and tuning exFAT filesystems. Use exfatprogs when +creating or checking an exFAT filesystem. For example, use ``mkfs.exfat`` +to create a filesystem and ``fsck.exfat`` to check or repair one. + +The project is available at: + + https://github.com/exfatprogs/exfatprogs + + +Supported mount options +======================= + +The exFAT driver supports the following mount options: + +======================= ==================================================== +uid= +gid= Set the owner and group of all files and + directories. The default is the uid and gid of + the process mounting the filesystem. + +umask= Set the permission mask for files and directories. + The default is the umask of the process mounting + the filesystem. + +dmask= Set the permission mask for directories. + +fmask= Set the permission mask for files. + +allow_utime= Control the permission check for changing file + timestamps. Only permission bits 0022 are used. + Permission bit 0020 allows members of the file's + group to change timestamps, and permission bit 0002 + allows other users to change timestamps. The + default is derived from dmask (``~dmask & 0022``). + +iocharset=name Character set used to convert between user-visible + filenames and the UTF-16 character encoding used by + exFAT. The default is + CONFIG_EXFAT_DEFAULT_IOCHARSET, which is ``utf8`` + unless changed at kernel configuration time. Use + ``iocharset=utf8`` for UTF-8 filename handling. + +errors= Specify exFAT behavior on filesystem errors. The + value must be ``panic``, ``continue``, or + ``remount-ro``. These respectively panic, continue + without changing the filesystem, or remount the + filesystem read-only. The default is + ``remount-ro``. + +discard Issue discard/TRIM requests to the block device + when clusters are freed. This is disabled by + default. ``nodiscard`` disables it explicitly. + +keep_last_dots Keep trailing periods in path components during + lookup. Without this option, trailing periods are + stripped. Existing entries with trailing periods + can be accessed when this option is enabled, but + creating new entries with trailing periods is + rejected. + +sys_tz Use the system timezone as the UTC offset when an + exFAT timestamp does not contain a valid timezone + offset. This takes precedence over time_offset. + +time_offset=minutes Set the UTC offset, in minutes, used when an exFAT + timestamp does not contain a valid timezone offset. + Values from -1440 to 1440 are accepted. The default + is 0. This option is ignored when sys_tz is set. + +zero_size_dir Create directories with zero size and without + allocating a cluster. This is disabled by default; + the default behavior allocates a cluster for a new + directory. ``nozero_size_dir`` disables it + explicitly. +======================= ==================================================== + + +Deprecated mount options +------------------------ + +The following options are accepted for compatibility but should not be used: + +``utf8`` + Deprecated. Use ``iocharset=utf8`` instead. + +``debug``, ``namecase=``, ``codepage=`` + Deprecated and ignored by the exFAT driver. diff --git a/Documentation/filesystems/index.rst b/Documentation/filesystems/index.rst index 734a45e516675c..fbd55915a3183a 100644 --- a/Documentation/filesystems/index.rst +++ b/Documentation/filesystems/index.rst @@ -87,6 +87,7 @@ Documentation for filesystem implementations. ecryptfs efivarfs erofs + exfat ext2 ext3 ext4/index From 22c3e316e050d325a1fff9375ac62e05fb7eb51e Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:49 +1000 Subject: [PATCH 284/857] nfsd: honour client-provided attributes for NFS4_CREATE_EXCLUSIVE4_1 When a file is created with a v4.1 OPEN which requests NFS4_CREATE_EXCLUSIVE4_1, the request can include attributes to be set. However when the mtime/atime are set to hold the verifier, the other ia_valid flags are cleared, so no attributes requested by the client are used. This code was originally written for NFSv3 where NFS3_CREATE_EXCLUSIVE never includes attributes. When it was updated for v4.1, the fact that an exclusive create CAN include attributes was not handled properly. Fixes: ac6721a13e5b ("nfsd41: make sure nfs server process OPEN with EXCLUSIVE4_1 correctly") Cc: stable@vger.kernel.org Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-2-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 50c07561e31f3c..f3f7f14a93683f 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -394,8 +394,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if ((iap->ia_valid & ATTR_SIZE) && (iap->ia_size == 0)) iap->ia_valid &= ~ATTR_SIZE; if (nfsd4_create_is_exclusive(open->op_createmode)) { - iap->ia_valid = ATTR_MTIME | ATTR_ATIME | - ATTR_MTIME_SET|ATTR_ATIME_SET; + iap->ia_valid |= ATTR_MTIME | ATTR_ATIME | + ATTR_MTIME_SET|ATTR_ATIME_SET; iap->ia_mtime.tv_sec = v_mtime; iap->ia_atime.tv_sec = v_atime; iap->ia_mtime.tv_nsec = 0; From 523f4ca2dcf1931c6aca7e43eae7dd162d7cff41 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:50 +1000 Subject: [PATCH 285/857] nfsd: move check_nfsd_access() call into nfsd_cross_mnt() Whenever we cross a mount point, we need to check_nfsd_access() for v4. So move the call into nfsd_cross_mnt() in the place where we actually do cross. This avoids the possibility of calling nfsd_cross_mnt() without the required check_nfsd_access(). Also remove the last arg from check_nfsd_access(), which is always false. nfsd_cross_mnt() now returns an nfserr rather than an errno. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-3-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/export.c | 6 ++--- fs/nfsd/export.h | 3 +-- fs/nfsd/nfs4proc.c | 2 +- fs/nfsd/nfs4xdr.c | 9 +------- fs/nfsd/vfs.c | 55 +++++++++++++++++++++++++--------------------- fs/nfsd/vfs.h | 4 ++-- 6 files changed, 37 insertions(+), 42 deletions(-) diff --git a/fs/nfsd/export.c b/fs/nfsd/export.c index a47c90f40422b7..5aefb388cc27d1 100644 --- a/fs/nfsd/export.c +++ b/fs/nfsd/export.c @@ -1890,21 +1890,19 @@ __be32 check_security_flavor(struct svc_export *exp, struct svc_rqst *rqstp, * check_nfsd_access - check if access to export is allowed. * @exp: svc_export that is being accessed. * @rqstp: svc_rqst attempting to access @exp. - * @may_bypass_gss: reduce strictness of authorization check * * Return values: * %nfs_ok if access is granted, or * %nfserr_wrongsec if access is denied */ -__be32 check_nfsd_access(struct svc_export *exp, struct svc_rqst *rqstp, - bool may_bypass_gss) +__be32 check_nfsd_access(struct svc_export *exp, struct svc_rqst *rqstp) { __be32 status; status = check_xprtsec_policy(exp, rqstp); if (status != nfs_ok) return status; - return check_security_flavor(exp, rqstp, may_bypass_gss); + return check_security_flavor(exp, rqstp, false); } /* diff --git a/fs/nfsd/export.h b/fs/nfsd/export.h index d2b09cd761453d..117fb28db1e024 100644 --- a/fs/nfsd/export.h +++ b/fs/nfsd/export.h @@ -104,8 +104,7 @@ int nfsexp_flags(struct svc_cred *cred, struct svc_export *exp); __be32 check_xprtsec_policy(struct svc_export *exp, struct svc_rqst *rqstp); __be32 check_security_flavor(struct svc_export *exp, struct svc_rqst *rqstp, bool may_bypass_gss); -__be32 check_nfsd_access(struct svc_export *exp, struct svc_rqst *rqstp, - bool may_bypass_gss); +__be32 check_nfsd_access(struct svc_export *exp, struct svc_rqst *rqstp); /* * Function declarations diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index f3f7f14a93683f..93d8e722aae80a 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -3328,7 +3328,7 @@ nfsd4_proc_compound(struct svc_rqst *rqstp) if (current_fh->fh_export && need_wrongsec_check(rqstp)) - op->status = check_nfsd_access(current_fh->fh_export, rqstp, false); + op->status = check_nfsd_access(current_fh->fh_export, rqstp); } encode_op: if (op->status == nfserr_replay_me) { diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 606ddcb085c027..04755c41d87115 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -4574,8 +4574,6 @@ nfsd4_encode_entry4_fattr(struct nfsd4_readdir *cd, const char *name, * directly from the mountpoint dentry. */ if (nfsd_mountpoint(dentry, exp)) { - int err; - if (!(exp->ex_flags & NFSEXP_V4ROOT) && !attributes_need_mount(cd->rd_bmval)) { ignore_crossmnt = 1; @@ -4586,12 +4584,7 @@ nfsd4_encode_entry4_fattr(struct nfsd4_readdir *cd, const char *name, * Different "."/".." handling? Something else? * At least, add a comment here to explain.... */ - err = nfsd_cross_mnt(cd->rd_rqstp, &dentry, &exp); - if (err) { - nfserr = nfserrno(err); - goto out_put; - } - nfserr = check_nfsd_access(exp, cd->rd_rqstp, false); + nfserr = nfsd_cross_mnt(cd->rd_rqstp, &dentry, &exp); if (nfserr) goto out_put; crossed = true; diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index 8923a9910a08f2..7386062ae449aa 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -118,15 +118,15 @@ nfserrno (int errno) return nfserr_io; } -/* - * Called from nfsd_lookup and encode_dirent. Check if we have crossed +/* + * Called from nfsd_lookup and encode_dirent. Check if we have crossed * a mount point. - * Returns -EAGAIN or -ETIMEDOUT leaving *dpp and *expp unchanged, + * Returns an nfs error leaving *dpp and *expp unchanged, * or nfs_ok having possibly changed *dpp and *expp */ -int -nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, - struct svc_export **expp) +__be32 +nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, + struct svc_export **expp) { struct svc_export *exp = *expp, *exp2 = NULL; struct dentry *dentry = *dpp; @@ -134,6 +134,7 @@ nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, .dentry = dget(dentry)}; unsigned int follow_flags = 0; int err = 0; + __be32 nfserr = nfs_ok; if (exp->ex_flags & NFSEXP_CROSSMOUNT) follow_flags = LOOKUP_AUTOMOUNT; @@ -163,23 +164,28 @@ nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, err = 0; } else if (nfsd_v4client(rqstp) || (exp->ex_flags & NFSEXP_CROSSMOUNT) || EX_NOHIDE(exp2)) { - /* successfully crossed mount point */ - /* - * This is subtle: path.dentry is *not* on path.mnt - * at this point. The only reason we are safe is that - * original mnt is pinned down by exp, so we should - * put path *before* putting exp - */ - *dpp = path.dentry; - path.dentry = dentry; - *expp = exp2; - exp2 = exp; + nfserr = check_nfsd_access(exp, rqstp); + if (nfserr == nfs_ok) { + /* successfully crossed mount point */ + /* + * This is subtle: path.dentry is *not* on path.mnt + * at this point. The only reason we are safe is that + * original mnt is pinned down by exp, so we should + * put path *before* putting exp + */ + *dpp = path.dentry; + path.dentry = dentry; + *expp = exp2; + exp2 = exp; + } } out: path_put(&path); if (exp2) exp_put(exp2); - return err; + if (nfserr) + return nfserr; + return nfserrno(err); } static void follow_to_parent(struct path *path) @@ -277,10 +283,12 @@ nfsd_lookup_dentry(struct svc_rqst *rqstp, struct svc_fh *fhp, if (IS_ERR(dentry)) goto out_nfserr; if (nfsd_mountpoint(dentry, exp)) { - host_err = nfsd_cross_mnt(rqstp, &dentry, &exp); - if (host_err) { + __be32 nfserr = nfsd_cross_mnt(rqstp, &dentry, &exp); + + if (nfserr) { dput(dentry); - goto out_nfserr; + exp_put(exp); + return nfserr; } } } @@ -327,9 +335,6 @@ nfsd_lookup(struct svc_rqst *rqstp, struct svc_fh *fhp, const char *name, err = nfsd_lookup_dentry(rqstp, fhp, name, len, &exp, &dentry); if (err) return err; - err = check_nfsd_access(exp, rqstp, false); - if (err) - goto out; /* * Note: we compose the file handle now, but as the * dentry may be negative, it may need to be updated. @@ -337,7 +342,7 @@ nfsd_lookup(struct svc_rqst *rqstp, struct svc_fh *fhp, const char *name, err = fh_compose(resfh, exp, dentry, fhp); if (!err && d_really_is_negative(dentry)) err = nfserr_noent; -out: + dput(dentry); exp_put(exp); return err; diff --git a/fs/nfsd/vfs.h b/fs/nfsd/vfs.h index 4af2ff9e9dfeee..5554878781f464 100644 --- a/fs/nfsd/vfs.h +++ b/fs/nfsd/vfs.h @@ -76,8 +76,8 @@ static inline bool nfsd_attrs_valid(struct nfsd_attrs *attrs) } __be32 nfserrno (int errno); -int nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, - struct svc_export **expp); +__be32 nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, + struct svc_export **expp); __be32 nfsd_lookup(struct svc_rqst *, struct svc_fh *, const char *, unsigned int, struct svc_fh *); __be32 nfsd_lookup_dentry(struct svc_rqst *, struct svc_fh *, From 147f256d430dbd0534649531c489e28b4791430b Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:51 +1000 Subject: [PATCH 286/857] nfsd: correctly handle CREATE of mounted-on files Linux allows a file (non-directory) to be mounted on a file. nfsd mostly supports this if the crossmnt option is in effect. However if CREATE is used on an existing mounted-on file, the filehandle for the underlying file is returns. The client will then continue to use that filehandle. So cat /mnt/file will show the contents of the mounted file as expected, but if the dcache is flushed with "drop_caches" or similar, then >> /mnt/file cat /mnt/file will show the mounted-on file. For exclusive or checked creates this is not a problem as the creation will fail no matter which file is seen. For unchecked creates we need to see if the name is in the dcache, and if it is mounted. If so, we simply provide that filehandle, possibly truncating. This probably has always existed since before the git history. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-4-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs3proc.c | 19 ++++++++++++++++++- fs/nfsd/nfs4proc.c | 28 ++++++++++++++++++++++++++++ fs/nfsd/nfsproc.c | 17 ++++++++++++++++- 3 files changed, 62 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/nfs3proc.c b/fs/nfsd/nfs3proc.c index 0904d953d10e07..1df3c719e0da6c 100644 --- a/fs/nfsd/nfs3proc.c +++ b/fs/nfsd/nfs3proc.c @@ -282,6 +282,7 @@ nfsd3_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, struct nfsd_attrs attrs = { .na_iattr = iap, }; + struct svc_export *exp; __u32 v_mtime, v_atime; struct inode *inode; __be32 status; @@ -320,7 +321,23 @@ nfsd3_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out; } - status = fh_compose(resfhp, fhp->fh_export, child, fhp); + exp = exp_get(fhp->fh_export); + if (argp->createmode == NFS3_CREATE_UNCHECKED) { + /* + * If name is already in dcache we need to check for mountpoints + */ + if (d_is_reg(child) && + unlikely(nfsd_mountpoint(child, exp))) { + status = nfsd_cross_mnt(rqstp, &child, &exp); + if (status != nfs_ok) { + exp_put(exp); + goto out; + } + } + } + + status = fh_compose(resfhp, exp, child, fhp); + exp_put(exp); if (status != nfs_ok) goto out; diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 93d8e722aae80a..424221677fa0ee 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -271,6 +271,34 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, parent = fhp->fh_dentry; inode = d_inode(parent); + if (open->op_createmode == NFS4_CREATE_UNCHECKED) { + /* + * If name is already in dcache we need to check for mountpoints + */ + child = try_lookup_noperm(&QSTR_LEN(open->op_fname, + open->op_fnamelen), + parent); + if (child && !IS_ERR(child) && d_is_reg(child) && + unlikely(nfsd_mountpoint(child, fhp->fh_export))) { + struct svc_export *exp = exp_get(fhp->fh_export); + + status = nfsd_cross_mnt(rqstp, &child, &exp); + if (status == nfs_ok) + status = fh_compose(resfhp, exp, + child, fhp); + if (status == nfs_ok) + status = fh_fill_both_attrs(fhp); + open->op_truncate = + (iap->ia_valid & ATTR_SIZE) && + !iap->ia_size; + dput(child); + exp_put(exp); + return status; + } + if (!IS_ERR(child)) + dput(child); + } + host_err = fh_want_write(fhp); if (host_err) return nfserrno(host_err); diff --git a/fs/nfsd/nfsproc.c b/fs/nfsd/nfsproc.c index e2b5f8a241bea7..2a82fa64e47855 100644 --- a/fs/nfsd/nfsproc.c +++ b/fs/nfsd/nfsproc.c @@ -291,6 +291,7 @@ nfsd_proc_create(struct svc_rqst *rqstp) struct nfsd_attrs attrs = { .na_iattr = attr, }; + struct svc_export *exp; struct inode *inode; struct dentry *dchild; int type, mode; @@ -319,8 +320,22 @@ nfsd_proc_create(struct svc_rqst *rqstp) resp->status = nfserrno(PTR_ERR(dchild)); goto out_write; } + /* + * If name exists we need to check for mountpoints + */ + exp = exp_get(dirfhp->fh_export); + if (d_is_reg(dchild) && + unlikely(nfsd_mountpoint(dchild, exp))) { + resp->status = nfsd_cross_mnt(rqstp, &dchild, &exp); + if (resp->status != nfs_ok) { + exp_put(exp); + goto out_unlock; + } + } + fh_init(newfhp, NFS_FHSIZE); - resp->status = fh_compose(newfhp, dirfhp->fh_export, dchild, dirfhp); + resp->status = fh_compose(newfhp, exp, dchild, dirfhp); + exp_put(exp); if (!resp->status && d_really_is_negative(dchild)) resp->status = nfserr_noent; if (resp->status) { From bfcfb11d814e897fe461d53f54344adf7d0b63c1 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:52 +1000 Subject: [PATCH 287/857] nfsd: replace fh_fill_both_attrs() with fh_fill_post_noop() fh_fill_both_attrs() is only needed for open/create and is used in the case when the target already existed so no creating happens. As part of refactoring this code it is changed to call fh_fill_pre_attrs() once early on (so errors only need to be caught in one place) and then to use a new fh_fill_post_noop() when it is determined that no creation happened. fh_fill_pre_attrs() now stores the attrs (which it had to get all of anyway)_ in ->fh_post_attr. fh_fill_post_noop() simply marks them as valid. fh_fill_post_attrs() replaces them. This change involves moving fh_fill_pre_attrs() out of the inode_lock on the directory. This means that we cannot provide "atomic" wcc data so a new fh_fill_pre_attrs_unlocked() is provided which marks the attrs as non-atomic. This is unfortunate but inevitable if we are ever to allow concurrent updates in a directory (which can significantly improve performance in some cases). To get atomic pre/post attributes we will need to be able to ask the fs to provide them, or to request a lease on the directory for the duration of an operation. Note that we haven't provided pre/post attrs on WRITE requests for a long time for exactly this reason - we cannot lock the file to get them. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-5-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 23 +++++++--------- fs/nfsd/nfsfh.c | 69 +++++++++++++++++++++++----------------------- fs/nfsd/nfsfh.h | 14 +++++++++- 3 files changed, 57 insertions(+), 49 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 424221677fa0ee..8e17e95d2cd058 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -286,8 +286,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (status == nfs_ok) status = fh_compose(resfhp, exp, child, fhp); - if (status == nfs_ok) - status = fh_fill_both_attrs(fhp); + fh_fill_post_noop(fhp); open->op_truncate = (iap->ia_valid & ATTR_SIZE) && !iap->ia_size; @@ -356,9 +355,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, /* NFSv4 protocol requires change attributes even though * no change happened. */ - status = fh_fill_both_attrs(fhp); - if (status != nfs_ok) - goto out; + fh_fill_post_noop(fhp); status = fh_compose(resfhp, fhp->fh_export, child, fhp); if (status != nfs_ok) @@ -405,9 +402,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (!IS_POSIXACL(inode)) iap->ia_mode &= ~current_umask(); - status = fh_fill_pre_attrs(fhp); - if (status != nfs_ok) - goto out; status = nfsd4_vfs_create(fhp, &child, open); if (status != nfs_ok) goto out; @@ -493,6 +487,9 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru fh_init(*resfh, NFS4_FHSIZE); open->op_truncate = false; + status = fh_fill_pre_attrs_unlocked(current_fh); + if (status) + goto out; if (open->op_create) { /* FIXME: check session persistence and pnfs flags. * The nfsv4.1 spec requires the following semantics: @@ -524,11 +521,11 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru } else { status = nfsd_lookup(rqstp, current_fh, open->op_fname, open->op_fnamelen, *resfh); - if (status == nfs_ok) - /* NFSv4 protocol requires change attributes even though - * no change happened. - */ - status = fh_fill_both_attrs(current_fh); + /* + * NFSv4 protocol requires change attributes even though + * no change happened. + */ + fh_fill_post_noop(current_fh); } if (status) goto out; diff --git a/fs/nfsd/nfsfh.c b/fs/nfsd/nfsfh.c index c7c60c35bdfc0e..c0a46784d525ae 100644 --- a/fs/nfsd/nfsfh.c +++ b/fs/nfsd/nfsfh.c @@ -782,34 +782,53 @@ __be32 fh_getattr(const struct svc_fh *fhp, struct kstat *stat) AT_STATX_SYNC_AS_STAT)); } -/** - * fh_fill_pre_attrs - Fill in pre-op attributes - * @fhp: file handle to be updated - * - */ -__be32 __must_check fh_fill_pre_attrs(struct svc_fh *fhp) +static __be32 __must_check __fh_fill_pre_attrs(struct svc_fh *fhp) { bool v4 = (fhp->fh_maxsize == NFS4_FHSIZE); - struct kstat stat; __be32 err; if (fhp->fh_no_wcc || fhp->fh_pre_saved) return nfs_ok; - err = fh_getattr(fhp, &stat); + err = fh_getattr(fhp, &fhp->fh_post_attr); if (err) return err; if (v4) - fhp->fh_pre_change = nfsd4_change_attribute(&stat); + fhp->fh_pre_change = fhp->fh_post_change = + nfsd4_change_attribute(&fhp->fh_post_attr); - fhp->fh_pre_mtime = stat.mtime; - fhp->fh_pre_ctime = stat.ctime; - fhp->fh_pre_size = stat.size; + fhp->fh_pre_mtime = fhp->fh_post_attr.mtime; + fhp->fh_pre_ctime = fhp->fh_post_attr.ctime; + fhp->fh_pre_size = fhp->fh_post_attr.size; fhp->fh_pre_saved = true; return nfs_ok; } +/** + * fh_fill_pre_attrs - Fill in pre-op attributes + * @fhp: file handle to be updated + * + * Post-op attrs are filled and pre-op attrs are copied + * from there. The post-op attrs can later be replaced by + * fh_fill_post_attrs() or activated by fh_fill_post_noop(). + * + * The inode must be locked. + * + * Returns: error from vfs_getattr() which must be checked. + */ +__be32 __must_check fh_fill_pre_attrs(struct svc_fh *fhp) +{ + lockdep_assert_held_write(&fhp->fh_dentry->d_inode->i_rwsem); + return __fh_fill_pre_attrs(fhp); +} + +__be32 __must_check fh_fill_pre_attrs_unlocked(struct svc_fh *fhp) +{ + fhp->fh_no_atomic_attr = true; + return __fh_fill_pre_attrs(fhp); +} + /** * fh_fill_post_attrs - Fill in post-op attributes * @fhp: file handle to be updated @@ -826,6 +845,9 @@ __be32 fh_fill_post_attrs(struct svc_fh *fhp) if (fhp->fh_post_saved) printk("nfsd: inode locked twice during operation.\n"); + if (!fhp->fh_no_atomic_attr) + lockdep_assert_held_write(&fhp->fh_dentry->d_inode->i_rwsem); + err = fh_getattr(fhp, &fhp->fh_post_attr); if (err) return err; @@ -837,29 +859,6 @@ __be32 fh_fill_post_attrs(struct svc_fh *fhp) return nfs_ok; } -/** - * fh_fill_both_attrs - Fill pre-op and post-op attributes - * @fhp: file handle to be updated - * - * This is used when the directory wasn't changed, but wcc attributes - * are needed anyway. - */ -__be32 __must_check fh_fill_both_attrs(struct svc_fh *fhp) -{ - __be32 err; - - err = fh_fill_post_attrs(fhp); - if (err) - return err; - - fhp->fh_pre_change = fhp->fh_post_change; - fhp->fh_pre_mtime = fhp->fh_post_attr.mtime; - fhp->fh_pre_ctime = fhp->fh_post_attr.ctime; - fhp->fh_pre_size = fhp->fh_post_attr.size; - fhp->fh_pre_saved = true; - return nfs_ok; -} - /* * Release a file handle. */ diff --git a/fs/nfsd/nfsfh.h b/fs/nfsd/nfsfh.h index cdeb5eea65a896..ab15b59ac7b3ba 100644 --- a/fs/nfsd/nfsfh.h +++ b/fs/nfsd/nfsfh.h @@ -337,6 +337,18 @@ static inline void fh_clear_pre_post_attrs(struct svc_fh *fhp) u64 nfsd4_change_attribute(const struct kstat *stat); __be32 __must_check fh_fill_pre_attrs(struct svc_fh *fhp); +__be32 __must_check fh_fill_pre_attrs_unlocked(struct svc_fh *fhp); __be32 fh_fill_post_attrs(struct svc_fh *fhp); -__be32 __must_check fh_fill_both_attrs(struct svc_fh *fhp); + +/** + * fh_fill_post_noop - Copy pre attrs to post attrs + * @fhp: file handle to be updated + * + * This is used when the directory wasn't changed, but wcc attributes + * are needed anyway. + */ +static inline void fh_fill_post_noop(struct svc_fh *fhp) +{ + fhp->fh_post_saved = true; +} #endif /* _LINUX_NFSD_NFSFH_H */ From 1f5518dd7df37e63f63d375cf076fab86e8be6ed Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:53 +1000 Subject: [PATCH 288/857] nfsd: move fh_want_write() after preamble in nfsd4_create_file() As part of separating the nfsd-specific code from the VFS interaction code in nfsd4_create_file(), move fh_want_write() to just before we need it. Consequently errors in the "if" statement that this code is moved over can now be returned immediately rather than needing to "goto out". Also restructure that "if" statement to only test is_create_with_attrs() once. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-6-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 31 +++++++++++++++++-------------- 1 file changed, 17 insertions(+), 14 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 8e17e95d2cd058..d3c629491ef10c 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -298,22 +298,18 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, dput(child); } - host_err = fh_want_write(fhp); - if (host_err) - return nfserrno(host_err); - - if (open->op_acl) { + if (!is_create_with_attrs(open)) { + /* No attrs to check */ + } else if (open->op_acl) { if (open->op_dpacl || open->op_pacl) { - status = nfserr_inval; - goto out; + /* Cannot specify both NFSv4 and Posix ACLs */ + return nfserr_inval; } - if (is_create_with_attrs(open)) { - status = nfsd4_acl_to_attr(NF4REG, open->op_acl, + status = nfsd4_acl_to_attr(NF4REG, open->op_acl, &attrs); - if (status) - goto out; - } - } else if (is_create_with_attrs(open)) { + if (status) + return status; + } else { /* The dpacl and pacl will get released by nfsd_attrs_free(). */ attrs.na_dpacl = open->op_dpacl; attrs.na_pacl = open->op_pacl; @@ -321,6 +317,12 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, open->op_pacl = NULL; } + host_err = fh_want_write(fhp); + if (host_err) { + status = nfserrno(host_err); + goto out_free; + } + child = start_creating(&nop_mnt_idmap, parent, &QSTR_LEN(open->op_fname, open->op_fnamelen)); if (IS_ERR(child)) { @@ -437,8 +439,9 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, open->op_bmval[2] &= ~FATTR4_WORD2_POSIX_ACCESS_ACL; out: end_creating(child); - nfsd_attrs_free(&attrs); fh_drop_write(fhp); +out_free: + nfsd_attrs_free(&attrs); return status; } From 75737637c46477b19b0d223cb2b5c12cc5cd4474 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:54 +1000 Subject: [PATCH 289/857] nfsd: move more nfs-specific code into preamble of nfsd4_create_file() Do NFS-specific prep before interacting with the VFS. We now add the verifier to iap early so it applies even when an EXCLUSIVE4_1 replay is detected based on that verifier, so we will set those attributes again. This should be harmless even though it will update ctime and i_version, and so will update the changeid seen by the client. It shouldn't matter because the resend implies that the client hasn't seen the file or its changeid. If some other client happens to have noticed the file, it might see an unnecessary changeid up, but that is of no consequence. Note that ctime would have been updated anyway if the client has included other attributes like an ACL. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-7-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 55 +++++++++++++++++++++++----------------------- 1 file changed, 27 insertions(+), 28 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index d3c629491ef10c..9d17cc41c6af13 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -298,6 +298,9 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, dput(child); } + if (!IS_POSIXACL(inode)) + iap->ia_mode &= ~current_umask(); + if (!is_create_with_attrs(open)) { /* No attrs to check */ } else if (open->op_acl) { @@ -317,6 +320,30 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, open->op_pacl = NULL; } + v_mtime = 0; + v_atime = 0; + if (nfsd4_create_is_exclusive(open->op_createmode)) { + u32 *verifier = (u32 *)open->op_verf.data; + + /* + * Solaris 7 gets confused (bugid 4218508) if these have + * the high bit set, as do xfs filesystems without the + * "bigtime" feature. So just clear the high bits. If this + * is ever changed to use different attrs for storing the + * verifier, then do_open_lookup() will also need to be + * fixed accordingly. + */ + v_mtime = verifier[0] & 0x7fffffff; + v_atime = verifier[1] & 0x7fffffff; + + iap->ia_valid |= ATTR_MTIME | ATTR_ATIME | + ATTR_MTIME_SET|ATTR_ATIME_SET; + iap->ia_mtime.tv_sec = v_mtime; + iap->ia_atime.tv_sec = v_atime; + iap->ia_mtime.tv_nsec = 0; + iap->ia_atime.tv_nsec = 0; + } + host_err = fh_want_write(fhp); if (host_err) { status = nfserrno(host_err); @@ -336,23 +363,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out; } - v_mtime = 0; - v_atime = 0; - if (nfsd4_create_is_exclusive(open->op_createmode)) { - u32 *verifier = (u32 *)open->op_verf.data; - - /* - * Solaris 7 gets confused (bugid 4218508) if these have - * the high bit set, as do xfs filesystems without the - * "bigtime" feature. So just clear the high bits. If this - * is ever changed to use different attrs for storing the - * verifier, then do_open_lookup() will also need to be - * fixed accordingly. - */ - v_mtime = verifier[0] & 0x7fffffff; - v_atime = verifier[1] & 0x7fffffff; - } - if (d_really_is_positive(child)) { /* NFSv4 protocol requires change attributes even though * no change happened. @@ -401,9 +411,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out; } - if (!IS_POSIXACL(inode)) - iap->ia_mode &= ~current_umask(); - status = nfsd4_vfs_create(fhp, &child, open); if (status != nfs_ok) goto out; @@ -417,14 +424,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, /* A newly created file already has a file size of zero. */ if ((iap->ia_valid & ATTR_SIZE) && (iap->ia_size == 0)) iap->ia_valid &= ~ATTR_SIZE; - if (nfsd4_create_is_exclusive(open->op_createmode)) { - iap->ia_valid |= ATTR_MTIME | ATTR_ATIME | - ATTR_MTIME_SET|ATTR_ATIME_SET; - iap->ia_mtime.tv_sec = v_mtime; - iap->ia_atime.tv_sec = v_atime; - iap->ia_mtime.tv_nsec = 0; - iap->ia_atime.tv_nsec = 0; - } set_attr: status = nfsd_create_setattr(rqstp, fhp, resfhp, &attrs); From bc2a408c82cec11c45abb7a0eefac7157bfd8a69 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:55 +1000 Subject: [PATCH 290/857] nfsd: remove subtlety from nfsd4_create_file() nfsd4_create_file() has a switch with cases for NFS4_CREATE_EXCLUSIVE and NFS4_CREATE_EXCLUSIVE4_1 which are identical except for one line which is marked "subtle" in both cases. The difference boils down to a "goto". For the EXCLUSIVE case the target is "out:" which is after a setattr call. For EXCLUSIVE4_1 the target is "set_attr:" which is the start of that setattr call. In the EXCLUSIVE case 'attrs' will only contain the verifier. Setting these again is not harmful as discussed in the previous patch. It will also call commit_metadata(). In performance terms the cost of an extra 'commit' in the rare case of a replaying exclusive create is negligible. So we can safely "goto setattr" in both cases and thus simplify the code. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-8-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 11 ++--------- 1 file changed, 2 insertions(+), 9 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 9d17cc41c6af13..07f5baec14f154 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -391,22 +391,15 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, status = nfserr_exist; break; case NFS4_CREATE_EXCLUSIVE: - if (inode_get_mtime_sec(d_inode(child)) == v_mtime && - inode_get_atime_sec(d_inode(child)) == v_atime && - d_inode(child)->i_size == 0) { - open->op_created = true; - break; /* subtle */ - } - status = nfserr_exist; - break; case NFS4_CREATE_EXCLUSIVE4_1: if (inode_get_mtime_sec(d_inode(child)) == v_mtime && inode_get_atime_sec(d_inode(child)) == v_atime && d_inode(child)->i_size == 0) { open->op_created = true; - goto set_attr; /* subtle */ + goto set_attr; } status = nfserr_exist; + break; } goto out; } From ee8f382f9c7aa0d5ca747d665107e38dd6dbc10f Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:56 +1000 Subject: [PATCH 291/857] nfsd: in nfsd4_create_file() let VFS report if file was created. nfsd4_create_file() currently assumes that if a lookup failed but then a create succeeds, then the "create" operation actually created the file. With atomic_open this may not be the case - some other actor might have created the file between the lookup and the create. So we move the call to nfsd4_vfs_create() earlier and set ->op_created based on the FMODE_CREATED flag that it set. Then use "! ->op_created" to trigger nfserr_exist handling. The switch statement is split up into two if() statements. First we check for the possibility of a successful exclusive create and set ->op_create to true if appropriate. Then we check for NFS4_CREATE_UNCHECKED to decide if a pre-existing file means an error or success. This allows us to combine the two fh_compose() calls to one place. A subtle difference here is that we now must only pass O_EXCL to dentry_create() for NFS4_CREATE_GUARDED. For the EXCLUSIVE create modes we want a successful open even if the file already exists. We then check the verifier after the open succeeded to see if it was exclusive. The above requires changing dentry_create() to reliably set FMODE_CREATED when the file was actually created. Previously it only sets this flag when atomic_open is used. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-9-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/namei.c | 2 ++ fs/nfsd/nfs4proc.c | 69 ++++++++++++++++++++-------------------------- 2 files changed, 32 insertions(+), 39 deletions(-) diff --git a/fs/namei.c b/fs/namei.c index 20a6534ea3efff..d95249dd527c1b 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -5211,6 +5211,8 @@ struct file *dentry_create(struct path *path, int flags, umode_t mode, error = vfs_create(mnt_idmap(path->mnt), path->dentry, mode, NULL); if (!error) error = vfs_open(path, file); + if (!error) + file->f_mode |= FMODE_CREATED; } if (unlikely(error)) return ERR_PTR(error); diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 07f5baec14f154..4e62809d1890a6 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -211,7 +211,11 @@ nfsd4_vfs_create(struct svc_fh *fhp, struct dentry **child, int oflags; oflags = O_CREAT | O_LARGEFILE; - if (nfsd4_create_is_exclusive(open->op_createmode)) + /* + * For the EXCLUSIVE modes we do our own uniqueness tests + * so don't want O_EXCL. + */ + if (open->op_createmode == NFS4_CREATE_GUARDED) oflags |= O_EXCL; switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { @@ -361,22 +365,30 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); if (status != nfs_ok) goto out; - } - if (d_really_is_positive(child)) { - /* NFSv4 protocol requires change attributes even though - * no change happened. - */ - fh_fill_post_noop(fhp); - - status = fh_compose(resfhp, fhp->fh_export, child, fhp); + status = nfsd4_vfs_create(fhp, &child, open); if (status != nfs_ok) goto out; + open->op_created = open->op_filp->f_mode & FMODE_CREATED; + } - switch (open->op_createmode) { - case NFS4_CREATE_UNCHECKED: - if (!d_is_reg(child)) - break; + status = fh_compose(resfhp, fhp->fh_export, child, fhp); + if (status != nfs_ok) + goto out; + + if (!open->op_created && + nfsd4_create_is_exclusive(open->op_createmode) && + inode_get_mtime_sec(d_inode(child)) == v_mtime && + inode_get_atime_sec(d_inode(child)) == v_atime && + d_inode(child)->i_size == 0) + open->op_created = true; + + if (!open->op_created) { + if (open->op_createmode == NFS4_CREATE_UNCHECKED) { + /* NFSv4 protocol requires change attributes + * even though no change happened. + */ + fh_fill_post_noop(fhp); /* * In NFSv4, we don't want to truncate the file @@ -384,41 +396,20 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, * some other reason. Furthermore, if the size is * nonzero, we should ignore it according to spec! */ - open->op_truncate = (iap->ia_valid & ATTR_SIZE) && - !iap->ia_size; - break; - case NFS4_CREATE_GUARDED: - status = nfserr_exist; - break; - case NFS4_CREATE_EXCLUSIVE: - case NFS4_CREATE_EXCLUSIVE4_1: - if (inode_get_mtime_sec(d_inode(child)) == v_mtime && - inode_get_atime_sec(d_inode(child)) == v_atime && - d_inode(child)->i_size == 0) { - open->op_created = true; - goto set_attr; - } + open->op_truncate = (d_is_reg(child) && + (iap->ia_valid & ATTR_SIZE) && + !iap->ia_size); + } else status = nfserr_exist; - break; - } goto out; } - - status = nfsd4_vfs_create(fhp, &child, open); - if (status != nfs_ok) - goto out; - open->op_created = true; + /* file was created */ fh_fill_post_attrs(fhp); - status = fh_compose(resfhp, fhp->fh_export, child, fhp); - if (status != nfs_ok) - goto out; - /* A newly created file already has a file size of zero. */ if ((iap->ia_valid & ATTR_SIZE) && (iap->ia_size == 0)) iap->ia_valid &= ~ATTR_SIZE; -set_attr: status = nfsd_create_setattr(rqstp, fhp, resfhp, &attrs); if (attrs.na_labelerr) From ce2ea586d196d87f895f37fd6d0a5d5c307a4a11 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:57 +1000 Subject: [PATCH 292/857] nfsd: nfsd4_create_file(): Move NFSD_MAY_CREATE check earlier We only need NFS_MAY_CREATE check if the file doesn't exist, but it is nfsd-specific code as it needs to check NFSEXP_READONLY and I want that to be separate from vfs-specific code, which eventually all be provided by the VFS. So move that check earlier, but hold the error status until needed. The if/else chain here looks a bit clumsy, but it will make a later patch cleaner. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-10-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 21 ++++++++++++--------- 1 file changed, 12 insertions(+), 9 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 4e62809d1890a6..5f43a4a26f3dc2 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -261,7 +261,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, struct dentry *parent, *child = ERR_PTR(-EINVAL); __u32 v_mtime, v_atime; struct inode *inode; - __be32 status; + __be32 status, create_status; int host_err; if (name_is_dot_dotdot(open->op_fname, open->op_fnamelen)) @@ -348,6 +348,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, iap->ia_atime.tv_nsec = 0; } + create_status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); + host_err = fh_want_write(fhp); if (host_err) { status = nfserrno(host_err); @@ -361,16 +363,17 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out; } - if (d_really_is_negative(child)) { - status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); - if (status != nfs_ok) - goto out; - + if (d_really_is_positive(child)) { + /* No creation needed */ + } else if (create_status) { + status = create_status; + } else { status = nfsd4_vfs_create(fhp, &child, open); - if (status != nfs_ok) - goto out; - open->op_created = open->op_filp->f_mode & FMODE_CREATED; + if (status == nfs_ok) + open->op_created = open->op_filp->f_mode & FMODE_CREATED; } + if (status != nfs_ok) + goto out; status = fh_compose(resfhp, fhp->fh_export, child, fhp); if (status != nfs_ok) From d917eeb82c33db2b114dc266f7914e9979fe985f Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:58 +1000 Subject: [PATCH 293/857] nfsd: fh_want_write) failure need not be immediately fatal for nfsd4_create_file() If nfsd4_create_file() is asked to create a file, then failure to get write access to the mount need not be fatal if the file already exists. So we can delay handling the error until it is known if creation was needed, just like with the error from testing for write permission in parent. This is similar to want_write error handling in lookup_open() in fs/namei.c. Note that getting mnt write access to support O_RDWR is handled separately in do_dentry_open(), and op_truncate is handled in do_open_permission(), so nfsd doesn't need to be concerned with these. It only needs to be concerned with creation, and setattr. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-11-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 15 +++++++-------- 1 file changed, 7 insertions(+), 8 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 5f43a4a26f3dc2..527602698d380c 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -262,7 +262,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, __u32 v_mtime, v_atime; struct inode *inode; __be32 status, create_status; - int host_err; + int want_write_err; if (name_is_dot_dotdot(open->op_fname, open->op_fnamelen)) return nfserr_exist; @@ -350,11 +350,10 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, create_status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); - host_err = fh_want_write(fhp); - if (host_err) { - status = nfserrno(host_err); - goto out_free; - } + want_write_err = fh_want_write(fhp); + if (want_write_err) + /* Might still succeed if no create is needed */ + create_status = nfserrno(want_write_err); child = start_creating(&nop_mnt_idmap, parent, &QSTR_LEN(open->op_fname, open->op_fnamelen)); @@ -425,8 +424,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, open->op_bmval[2] &= ~FATTR4_WORD2_POSIX_ACCESS_ACL; out: end_creating(child); - fh_drop_write(fhp); -out_free: + if (!want_write_err) + fh_drop_write(fhp); nfsd_attrs_free(&attrs); return status; } From db8b567ffa984f423d34b8c58ef1d33a27a30e01 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:59 +1000 Subject: [PATCH 294/857] nfsd: (almost) always open file in nfsd4_create_file() If the file is found to already exist, open it anyway. This will normally be needed eventually anyway, and providing a consistently valid op_filp will simplify future changes. To simplify this, change nfsd_check_obj_isreg() to take a dentry. This doesn't apply in the case where the file was found in the dcache to be mounted-on. That takes a different path and doesn't require an early open. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-12-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 39 +++++++++++++++++++++++++++++++++++---- 1 file changed, 35 insertions(+), 4 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 527602698d380c..226993ca761aa5 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -169,9 +169,9 @@ do_open_permission(struct svc_rqst *rqstp, struct svc_fh *current_fh, struct nfs return fh_verify(rqstp, current_fh, S_IFREG, accmode); } -static __be32 nfsd_check_obj_isreg(struct svc_fh *fh, u32 minor_version) +static __be32 nfsd_check_obj_isreg(struct dentry *child, u32 minor_version) { - umode_t mode = d_inode(fh->fh_dentry)->i_mode; + umode_t mode = d_inode(child)->i_mode; if (S_ISREG(mode)) return nfs_ok; @@ -253,6 +253,8 @@ static __be32 nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, struct svc_fh *resfhp, struct nfsd4_open *open) { + struct nfsd4_compoundres *resp = rqstp->rq_resp; + struct nfsd4_compound_state *cstate = &resp->cstate; struct iattr *iap = &open->op_iattr; struct nfsd_attrs attrs = { .na_iattr = iap, @@ -363,7 +365,35 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } if (d_really_is_positive(child)) { - /* No creation needed */ + /* + * open the file so that we consistently have a valid + * op_filp. + */ + struct path path = {.mnt = fhp->fh_export->ex_path.mnt, + .dentry = child, + }; + unsigned int oflags = O_LARGEFILE; + + switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { + case NFS4_SHARE_ACCESS_WRITE: + oflags |= O_WRONLY; + break; + case NFS4_SHARE_ACCESS_BOTH: + oflags |= O_RDWR; + break; + default: + oflags |= O_RDONLY; + } + + status = nfsd_check_obj_isreg(child, cstate->minorversion); + if (status == nfs_ok) { + open->op_filp = dentry_open(&path, oflags, + current_cred()); + if (IS_ERR(open->op_filp)) { + status = nfserrno(PTR_ERR(open->op_filp)); + open->op_filp = NULL; + } + } } else if (create_status) { status = create_status; } else { @@ -517,7 +547,8 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru } if (status) goto out; - status = nfsd_check_obj_isreg(*resfh, cstate->minorversion); + status = nfsd_check_obj_isreg((*resfh)->fh_dentry, + cstate->minorversion); if (status) goto out; From b5a1d3e7730f0f89c524b3f98fd6647a0dd671f6 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:00 +1000 Subject: [PATCH 295/857] nfsd: reduce range of directory lock in nfsd4_create_file() We only need to hold the lock taken by start_creating() until the create has been attempted. Holding for longer can serve no purpose. The lock is currently held across the setattr call. This might be the intent but it serves no purpose. Holding the lock prevents the name from being removed or renamed, but it doesn't prevent a GETATTR or a racing SETATTR or an OPEN. Calling end_creating() puts the reference to 'child', but we can still use the reference that was stored in open->op_filp. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-13-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 226993ca761aa5..2ef67dd951be30 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -367,7 +367,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (d_really_is_positive(child)) { /* * open the file so that we consistently have a valid - * op_filp. + * op_filp and consequently a valid ->f_path.dentry. */ struct path path = {.mnt = fhp->fh_export->ex_path.mnt, .dentry = child, @@ -401,9 +401,12 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (status == nfs_ok) open->op_created = open->op_filp->f_mode & FMODE_CREATED; } + end_creating(child); if (status != nfs_ok) goto out; + child = open->op_filp->f_path.dentry; + status = fh_compose(resfhp, fhp->fh_export, child, fhp); if (status != nfs_ok) goto out; @@ -453,7 +456,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (attrs.na_paclerr) open->op_bmval[2] &= ~FATTR4_WORD2_POSIX_ACCESS_ACL; out: - end_creating(child); if (!want_write_err) fh_drop_write(fhp); nfsd_attrs_free(&attrs); From 9f21e40d88694e008cf3187a6a1907cfdd26d657 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:01 +1000 Subject: [PATCH 296/857] nfsd: open-code nfsd4_vfs_create() into nfsd4_create_file() Having this sub function separate doesn't really add clarity, and merging allows for some refactoring and ultimately using a different VFS interface. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-14-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 76 +++++++++++++++++++++------------------------- 1 file changed, 34 insertions(+), 42 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 2ef67dd951be30..ee8616c7918575 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -202,46 +202,6 @@ static inline bool nfsd4_create_is_exclusive(int createmode) createmode == NFS4_CREATE_EXCLUSIVE4_1; } -static __be32 -nfsd4_vfs_create(struct svc_fh *fhp, struct dentry **child, - struct nfsd4_open *open) -{ - struct file *filp; - struct path path; - int oflags; - - oflags = O_CREAT | O_LARGEFILE; - /* - * For the EXCLUSIVE modes we do our own uniqueness tests - * so don't want O_EXCL. - */ - if (open->op_createmode == NFS4_CREATE_GUARDED) - oflags |= O_EXCL; - - switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { - case NFS4_SHARE_ACCESS_WRITE: - oflags |= O_WRONLY; - break; - case NFS4_SHARE_ACCESS_BOTH: - oflags |= O_RDWR; - break; - default: - oflags |= O_RDONLY; - } - - path.mnt = fhp->fh_export->ex_path.mnt; - path.dentry = *child; - filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, - current_cred()); - *child = path.dentry; - - if (IS_ERR(filp)) - return nfserrno(PTR_ERR(filp)); - - open->op_filp = filp; - return nfs_ok; -} - /* * Implement NFSv4's unchecked, guarded, and exclusive create * semantics for regular files. Open state for this new file is @@ -397,9 +357,41 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } else if (create_status) { status = create_status; } else { - status = nfsd4_vfs_create(fhp, &child, open); - if (status == nfs_ok) + struct file *filp; + struct path path; + int oflags; + + oflags = O_CREAT | O_LARGEFILE; + /* + * For the EXCLUSIVE modes we do our own uniqueness tests + * so don't want O_EXCL. + */ + if (open->op_createmode == NFS4_CREATE_GUARDED) + oflags |= O_EXCL; + + switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { + case NFS4_SHARE_ACCESS_WRITE: + oflags |= O_WRONLY; + break; + case NFS4_SHARE_ACCESS_BOTH: + oflags |= O_RDWR; + break; + default: + oflags |= O_RDONLY; + } + + path.mnt = fhp->fh_export->ex_path.mnt; + path.dentry = child; + filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, + current_cred()); + child = path.dentry; + + if (IS_ERR(filp)) { + status = nfserrno(PTR_ERR(filp)); + } else { + open->op_filp = filp; open->op_created = open->op_filp->f_mode & FMODE_CREATED; + } } end_creating(child); if (status != nfs_ok) From 527adad8b0f67e4117dd8755e20a59e03dd7d549 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:02 +1000 Subject: [PATCH 297/857] nfsd: move some code out of the d_really_is_negative() branch in nfsd4_create_file() The benefit of this code movement isn't immediately obvious, but it will make it easier to switch to using vfs_lookup_open(). One immediate benefit is that common code in the d_is_positive() branch can be discarded. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-15-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 73 ++++++++++++++++++---------------------------- 1 file changed, 28 insertions(+), 45 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index ee8616c7918575..6dff6013a068b4 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -220,7 +220,11 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, .na_iattr = iap, .na_seclabel = &open->op_label, }; + int oflags = O_CREAT | O_LARGEFILE; struct dentry *parent, *child = ERR_PTR(-EINVAL); + struct path path = { + .mnt = fhp->fh_export->ex_path.mnt, + }; __u32 v_mtime, v_atime; struct inode *inode; __be32 status, create_status; @@ -267,6 +271,24 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (!IS_POSIXACL(inode)) iap->ia_mode &= ~current_umask(); + /* + * For the EXCLUSIVE modes we do our own uniqueness tests + * so don't want O_EXCL. + */ + if (open->op_createmode == NFS4_CREATE_GUARDED) + oflags |= O_EXCL; + + switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { + case NFS4_SHARE_ACCESS_WRITE: + oflags |= O_WRONLY; + break; + case NFS4_SHARE_ACCESS_BOTH: + oflags |= O_RDWR; + break; + default: + oflags |= O_RDONLY; + } + if (!is_create_with_attrs(open)) { /* No attrs to check */ } else if (open->op_acl) { @@ -323,27 +345,13 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, status = nfserrno(PTR_ERR(child)); goto out; } + path.dentry = child; if (d_really_is_positive(child)) { /* * open the file so that we consistently have a valid * op_filp and consequently a valid ->f_path.dentry. */ - struct path path = {.mnt = fhp->fh_export->ex_path.mnt, - .dentry = child, - }; - unsigned int oflags = O_LARGEFILE; - - switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { - case NFS4_SHARE_ACCESS_WRITE: - oflags |= O_WRONLY; - break; - case NFS4_SHARE_ACCESS_BOTH: - oflags |= O_RDWR; - break; - default: - oflags |= O_RDONLY; - } status = nfsd_check_obj_isreg(child, cstate->minorversion); if (status == nfs_ok) { @@ -357,39 +365,14 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } else if (create_status) { status = create_status; } else { - struct file *filp; - struct path path; - int oflags; - - oflags = O_CREAT | O_LARGEFILE; - /* - * For the EXCLUSIVE modes we do our own uniqueness tests - * so don't want O_EXCL. - */ - if (open->op_createmode == NFS4_CREATE_GUARDED) - oflags |= O_EXCL; - - switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { - case NFS4_SHARE_ACCESS_WRITE: - oflags |= O_WRONLY; - break; - case NFS4_SHARE_ACCESS_BOTH: - oflags |= O_RDWR; - break; - default: - oflags |= O_RDONLY; - } - - path.mnt = fhp->fh_export->ex_path.mnt; - path.dentry = child; - filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, - current_cred()); + open->op_filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, + current_cred()); child = path.dentry; - if (IS_ERR(filp)) { - status = nfserrno(PTR_ERR(filp)); + if (IS_ERR(open->op_filp)) { + status = nfserrno(PTR_ERR(open->op_filp)); + open->op_filp = NULL; } else { - open->op_filp = filp; open->op_created = open->op_filp->f_mode & FMODE_CREATED; } } From 0d06c3fc14b2343dd2444929873255c0d8522490 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:03 +1000 Subject: [PATCH 298/857] nfsd: reduce want-write range in nfsd4_create_file() nfsd4_create_file() needs write access to the mount for two purposes: 1/ to create/open the file. 2/ to set attributes on the newly created (or pre-existing) file. Currently this is all handled by holding the write access across the open and the setattr. A subsequent patch will necessarily change how write access is gained for the open. So we reduce the range for the first want_write, and add another one to cover setattr. If we failed to get write access, it is only fatal if there were attrs to set. We call nfsd_create_setattr() if at all possible, even when no attrs, as it also calls commit_metadata and we need to be certain that the file creation has been synced. If the mount became read-only since the creation happened, we can safely assume that the sync happened as part of that. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-16-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 6dff6013a068b4..5e047469ba78e9 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -343,6 +343,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, &QSTR_LEN(open->op_fname, open->op_fnamelen)); if (IS_ERR(child)) { status = nfserrno(PTR_ERR(child)); + if (!want_write_err) + fh_drop_write(fhp); goto out; } path.dentry = child; @@ -377,6 +379,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } } end_creating(child); + if (!want_write_err) + fh_drop_write(fhp); if (status != nfs_ok) goto out; @@ -420,7 +424,16 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if ((iap->ia_valid & ATTR_SIZE) && (iap->ia_size == 0)) iap->ia_valid &= ~ATTR_SIZE; - status = nfsd_create_setattr(rqstp, fhp, resfhp, &attrs); + /* We will need write access to set the attrs */ + want_write_err = fh_want_write(fhp); + if (!want_write_err) { + status = nfsd_create_setattr(rqstp, fhp, + resfhp, &attrs); + fh_drop_write(fhp); + } else if (nfsd_attrs_valid(&attrs)) { + /* Needed write access */ + status = nfserrno(want_write_err); + } if (attrs.na_labelerr) open->op_bmval[2] &= ~FATTR4_WORD2_SECURITY_LABEL; @@ -431,8 +444,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (attrs.na_paclerr) open->op_bmval[2] &= ~FATTR4_WORD2_POSIX_ACCESS_ACL; out: - if (!want_write_err) - fh_drop_write(fhp); nfsd_attrs_free(&attrs); return status; } From 73e9961379511a0e713061814e2fc746ce63d24d Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:04 +1000 Subject: [PATCH 299/857] nfsd: move v0 checking out of nfsd_check_obj_isreg() A future patch will use nfsd_check_obj_isreg() in a context where the protocol version is not easily available. So move the version check out and put it at the end of do_open_lookup(). Also change to return errno error code and use nfserrno() to convert to nfs error codes. Use -ELOOP for nfserr_symlink, which is an error indication a problem with symlinks. -EFTYPE is a good match for nfserr_wrong_type. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-17-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 29 ++++++++++++----------------- fs/nfsd/vfs.c | 4 +++- 2 files changed, 15 insertions(+), 18 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 5e047469ba78e9..7853bc379b9f11 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -169,23 +169,17 @@ do_open_permission(struct svc_rqst *rqstp, struct svc_fh *current_fh, struct nfs return fh_verify(rqstp, current_fh, S_IFREG, accmode); } -static __be32 nfsd_check_obj_isreg(struct dentry *child, u32 minor_version) +static __be32 nfsd_check_obj_isreg(struct dentry *child) { umode_t mode = d_inode(child)->i_mode; if (S_ISREG(mode)) - return nfs_ok; + return 0; if (S_ISDIR(mode)) - return nfserr_isdir; + return -EISDIR; if (S_ISLNK(mode)) - return nfserr_symlink; - - /* RFC 7530 - 16.16.6 */ - if (minor_version == 0) - return nfserr_symlink; - else - return nfserr_wrong_type; - + return -ELOOP; + return -EFTYPE; } static void nfsd4_set_open_owner_reply_cache(struct nfsd4_compound_state *cstate, struct nfsd4_open *open, struct svc_fh *resfh) @@ -213,8 +207,6 @@ static __be32 nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, struct svc_fh *resfhp, struct nfsd4_open *open) { - struct nfsd4_compoundres *resp = rqstp->rq_resp; - struct nfsd4_compound_state *cstate = &resp->cstate; struct iattr *iap = &open->op_iattr; struct nfsd_attrs attrs = { .na_iattr = iap, @@ -355,8 +347,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, * op_filp and consequently a valid ->f_path.dentry. */ - status = nfsd_check_obj_isreg(child, cstate->minorversion); - if (status == nfs_ok) { + status = nfserrno(nfsd_check_obj_isreg(child)); + if (!status) { open->op_filp = dentry_open(&path, oflags, current_cred()); if (IS_ERR(open->op_filp)) { @@ -535,8 +527,7 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru } if (status) goto out; - status = nfsd_check_obj_isreg((*resfh)->fh_dentry, - cstate->minorversion); + status = nfserrno(nfsd_check_obj_isreg((*resfh)->fh_dentry)); if (status) goto out; @@ -548,6 +539,10 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru status = do_open_permission(rqstp, *resfh, open, accmode); set_change_info(&open->op_cinfo, current_fh); out: + if (status == nfserr_wrong_type && cstate->minorversion == 0) + /* RFC 7530 - 16.16.6 */ + return nfserr_symlink; + return status; } diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index 7386062ae449aa..f65dad403ee08f 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -63,7 +63,7 @@ u64 nfsd_io_cache_write __read_mostly = NFSD_IO_BUFFERED; * it's an error we don't expect, log it once and return nfserr_io. */ __be32 -nfserrno (int errno) +nfserrno(int errno) { static struct { __be32 nfserr; @@ -107,6 +107,8 @@ nfserrno (int errno) { nfserr_perm, -ENOKEY }, { nfserr_no_grace, -ENOGRACE}, { nfserr_io, -EBADMSG }, + { nfserr_symlink, -ELOOP }, + { nfserr_wrong_type, -EFTYPE }, }; int i; From 79132f8af91b9dcef28b2b93d3a493be83d37cfe Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:05 +1000 Subject: [PATCH 300/857] nfsd: separate out VFS-specific code from nfsd4_create_file() All the code in nfsd4_create_file() that is VFS manipulation, with now NFS-specific knowledge, has been localised. Now we split that out into a separate function: do_lookup_open(). It is planned to provide a vfs_lookup_open() in vfs code which provides this functionality. This will share more code with the syscall open path, and make it easier to modify locking at the VFS level. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-18-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 121 ++++++++++++++++++++++++--------------------- 1 file changed, 66 insertions(+), 55 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 7853bc379b9f11..2d43ff327b8708 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -169,7 +169,7 @@ do_open_permission(struct svc_rqst *rqstp, struct svc_fh *current_fh, struct nfs return fh_verify(rqstp, current_fh, S_IFREG, accmode); } -static __be32 nfsd_check_obj_isreg(struct dentry *child) +static int nfsd_check_obj_isreg(struct dentry *child) { umode_t mode = d_inode(child)->i_mode; @@ -196,6 +196,52 @@ static inline bool nfsd4_create_is_exclusive(int createmode) createmode == NFS4_CREATE_EXCLUSIVE4_1; } +static struct file *do_lookup_open(struct path *parent, + struct qstr *name, + unsigned int oflags, + umode_t mode) +{ + struct file *filp = NULL; + struct path path; + struct dentry *child; + int want_write_err = 0; + + want_write_err = mnt_want_write(parent->mnt); + + child = start_creating(&nop_mnt_idmap, parent->dentry, name); + if (IS_ERR(child)) { + filp = ERR_CAST(child); + goto out; + } + path.mnt = parent->mnt; + path.dentry = child; + + if (d_really_is_positive(child)) { + /* + * open the file so that we consistently have a valid + * op_filp and consequently a valid ->f_path.dentry. + */ + int err = nfsd_check_obj_isreg(child); + + if (err) + filp = ERR_PTR(err); + else + filp = dentry_open(&path, oflags, current_cred()); + } else if (!(oflags & O_CREAT)) { + filp = ERR_PTR(-ENOENT); + } else if (want_write_err) { + filp = ERR_PTR(want_write_err); + } else { + filp = dentry_create(&path, oflags, mode, current_cred()); + child = path.dentry; + } + end_creating(child); +out: + if (!want_write_err) + mnt_drop_write(parent->mnt); + return filp; +} + /* * Implement NFSv4's unchecked, guarded, and exclusive create * semantics for regular files. Open state for this new file is @@ -213,12 +259,12 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, .na_seclabel = &open->op_label, }; int oflags = O_CREAT | O_LARGEFILE; - struct dentry *parent, *child = ERR_PTR(-EINVAL); - struct path path = { + struct dentry *child = ERR_PTR(-EINVAL); + struct path parent = { .mnt = fhp->fh_export->ex_path.mnt, + .dentry = fhp->fh_dentry, }; __u32 v_mtime, v_atime; - struct inode *inode; __be32 status, create_status; int want_write_err; @@ -230,8 +276,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_EXEC); if (status != nfs_ok) return status; - parent = fhp->fh_dentry; - inode = d_inode(parent); if (open->op_createmode == NFS4_CREATE_UNCHECKED) { /* @@ -239,7 +283,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, */ child = try_lookup_noperm(&QSTR_LEN(open->op_fname, open->op_fnamelen), - parent); + parent.dentry); if (child && !IS_ERR(child) && d_is_reg(child) && unlikely(nfsd_mountpoint(child, fhp->fh_export))) { struct svc_export *exp = exp_get(fhp->fh_export); @@ -260,7 +304,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, dput(child); } - if (!IS_POSIXACL(inode)) + if (!IS_POSIXACL(d_inode(parent.dentry))) iap->ia_mode &= ~current_umask(); /* @@ -325,58 +369,25 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } create_status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); - - want_write_err = fh_want_write(fhp); - if (want_write_err) + if (create_status) /* Might still succeed if no create is needed */ - create_status = nfserrno(want_write_err); - - child = start_creating(&nop_mnt_idmap, parent, - &QSTR_LEN(open->op_fname, open->op_fnamelen)); - if (IS_ERR(child)) { - status = nfserrno(PTR_ERR(child)); - if (!want_write_err) - fh_drop_write(fhp); + oflags &= ~O_CREAT; + + open->op_filp = do_lookup_open(&parent, + &QSTR_LEN(open->op_fname, + open->op_fnamelen), + oflags, + open->op_iattr.ia_mode); + if (IS_ERR(open->op_filp)) { + status = nfserrno(PTR_ERR(open->op_filp)); + open->op_filp = NULL; + if (status == nfserr_noent && create_status) + status = create_status; goto out; } - path.dentry = child; - - if (d_really_is_positive(child)) { - /* - * open the file so that we consistently have a valid - * op_filp and consequently a valid ->f_path.dentry. - */ - - status = nfserrno(nfsd_check_obj_isreg(child)); - if (!status) { - open->op_filp = dentry_open(&path, oflags, - current_cred()); - if (IS_ERR(open->op_filp)) { - status = nfserrno(PTR_ERR(open->op_filp)); - open->op_filp = NULL; - } - } - } else if (create_status) { - status = create_status; - } else { - open->op_filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, - current_cred()); - child = path.dentry; - - if (IS_ERR(open->op_filp)) { - status = nfserrno(PTR_ERR(open->op_filp)); - open->op_filp = NULL; - } else { - open->op_created = open->op_filp->f_mode & FMODE_CREATED; - } - } - end_creating(child); - if (!want_write_err) - fh_drop_write(fhp); - if (status != nfs_ok) - goto out; child = open->op_filp->f_path.dentry; + open->op_created = open->op_filp->f_mode & FMODE_CREATED; status = fh_compose(resfhp, fhp->fh_export, child, fhp); if (status != nfs_ok) From d8ef3871b6188daaf72858de76b3a97614b2b380 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:07 -0400 Subject: [PATCH 301/857] NFSD: Move XDR encoding helpers out of xdr4.h These static inline helpers use the nfserr_resource macro, which pulls in the whole rack of NFS status codes. Move those helpers into the only file that uses them, to get rid of the nfserr macro dependency globally. These helper were originally placed in xdr4.h because I thought they would be utilized in the rest of the NFSv4 XDR code, but XDR translation is eventually to be subsumed by xdrgen instead. I'm not converting them now because that is much more churn than this patch is. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-2-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/blocklayoutxdr.c | 10 +++ fs/nfsd/nfs4xdr.c | 92 ++++++++++++++++++++++++++ fs/nfsd/xdr4.h | 139 --------------------------------------- 3 files changed, 102 insertions(+), 139 deletions(-) diff --git a/fs/nfsd/blocklayoutxdr.c b/fs/nfsd/blocklayoutxdr.c index f80dbc41fd5f01..51cc2a07d7b384 100644 --- a/fs/nfsd/blocklayoutxdr.c +++ b/fs/nfsd/blocklayoutxdr.c @@ -13,6 +13,16 @@ #define NFSDDBG_FACILITY NFSDDBG_PNFS +static __be32 +nfsd4_decode_deviceid4(struct xdr_stream *xdr, struct nfsd4_deviceid *devid) +{ + __be32 *p = xdr_inline_decode(xdr, NFS4_DEVICEID4_SIZE); + + if (unlikely(!p)) + return nfserr_bad_xdr; + svcxdr_decode_deviceid4(p, devid); + return nfs_ok; +} /** * nfsd4_block_encode_layoutget - encode block/scsi layout extent array diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 04755c41d87115..05d64733094b2a 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -1919,6 +1919,17 @@ nfsd4_decode_get_dir_delegation(struct nfsd4_compoundargs *argp, } #ifdef CONFIG_NFSD_PNFS +static __be32 +nfsd4_decode_deviceid4(struct xdr_stream *xdr, struct nfsd4_deviceid *devid) +{ + __be32 *p = xdr_inline_decode(xdr, NFS4_DEVICEID4_SIZE); + + if (unlikely(!p)) + return nfserr_bad_xdr; + svcxdr_decode_deviceid4(p, devid); + return nfs_ok; +} + static __be32 nfsd4_decode_getdeviceinfo(struct nfsd4_compoundargs *argp, union nfsd4_op_u *u) @@ -2733,6 +2744,87 @@ nfsd4_decode_compound(struct nfsd4_compoundargs *argp) return true; } +static __always_inline __be32 +nfsd4_encode_bool(struct xdr_stream *xdr, bool val) +{ + __be32 *p = xdr_reserve_space(xdr, XDR_UNIT); + + if (unlikely(p == NULL)) + return nfserr_resource; + *p = val ? xdr_one : xdr_zero; + return nfs_ok; +} + +static __always_inline __be32 +nfsd4_encode_uint32_t(struct xdr_stream *xdr, u32 val) +{ + __be32 *p = xdr_reserve_space(xdr, XDR_UNIT); + + if (unlikely(p == NULL)) + return nfserr_resource; + *p = cpu_to_be32(val); + return nfs_ok; +} + +#define nfsd4_encode_aceflag4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_acemask4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_acetype4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_count4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_mode4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_nfs_lease4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_qop4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_sequenceid4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_slotid4(x, v) nfsd4_encode_uint32_t(x, v) + +static __always_inline __be32 +nfsd4_encode_uint64_t(struct xdr_stream *xdr, u64 val) +{ + __be32 *p = xdr_reserve_space(xdr, XDR_UNIT * 2); + + if (unlikely(p == NULL)) + return nfserr_resource; + put_unaligned_be64(val, p); + return nfs_ok; +} + +#define nfsd4_encode_changeid4(x, v) nfsd4_encode_uint64_t(x, v) +#define nfsd4_encode_nfs_cookie4(x, v) nfsd4_encode_uint64_t(x, v) +#define nfsd4_encode_length4(x, v) nfsd4_encode_uint64_t(x, v) +#define nfsd4_encode_offset4(x, v) nfsd4_encode_uint64_t(x, v) + +static __always_inline __be32 +nfsd4_encode_opaque_fixed(struct xdr_stream *xdr, const void *data, + size_t size) +{ + __be32 *p = xdr_reserve_space(xdr, xdr_align_size(size)); + size_t pad = xdr_pad_size(size); + + if (unlikely(p == NULL)) + return nfserr_resource; + memcpy(p, data, size); + if (pad) + memset((char *)p + size, 0, pad); + return nfs_ok; +} + +static __always_inline __be32 +nfsd4_encode_opaque(struct xdr_stream *xdr, const void *data, size_t size) +{ + size_t pad = xdr_pad_size(size); + __be32 *p; + + p = xdr_reserve_space(xdr, XDR_UNIT + xdr_align_size(size)); + if (unlikely(p == NULL)) + return nfserr_resource; + *p++ = cpu_to_be32(size); + memcpy(p, data, size); + if (pad) + memset((char *)p + size, 0, pad); + return nfs_ok; +} + +#define nfsd4_encode_component4(x, d, s) nfsd4_encode_opaque(x, d, s) + static __be32 nfsd4_encode_nfs_fh4(struct xdr_stream *xdr, const struct knfsd_fh *fh_handle) { diff --git a/fs/nfsd/xdr4.h b/fs/nfsd/xdr4.h index c7eda5bc833b1a..e833407859c8d8 100644 --- a/fs/nfsd/xdr4.h +++ b/fs/nfsd/xdr4.h @@ -50,134 +50,6 @@ #define HAS_CSTATE_FLAG(c, f) ((c)->sid_flags & (f)) #define CLEAR_CSTATE_FLAG(c, f) ((c)->sid_flags &= ~(f)) -/** - * nfsd4_encode_bool - Encode an XDR bool type result - * @xdr: target XDR stream - * @val: boolean value to encode - * - * Return values: - * %nfs_ok: @val encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_bool(struct xdr_stream *xdr, bool val) -{ - __be32 *p = xdr_reserve_space(xdr, XDR_UNIT); - - if (unlikely(p == NULL)) - return nfserr_resource; - *p = val ? xdr_one : xdr_zero; - return nfs_ok; -} - -/** - * nfsd4_encode_uint32_t - Encode an XDR uint32_t type result - * @xdr: target XDR stream - * @val: integer value to encode - * - * Return values: - * %nfs_ok: @val encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_uint32_t(struct xdr_stream *xdr, u32 val) -{ - __be32 *p = xdr_reserve_space(xdr, XDR_UNIT); - - if (unlikely(p == NULL)) - return nfserr_resource; - *p = cpu_to_be32(val); - return nfs_ok; -} - -#define nfsd4_encode_aceflag4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_acemask4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_acetype4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_count4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_mode4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_nfs_lease4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_qop4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_sequenceid4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_slotid4(x, v) nfsd4_encode_uint32_t(x, v) - -/** - * nfsd4_encode_uint64_t - Encode an XDR uint64_t type result - * @xdr: target XDR stream - * @val: integer value to encode - * - * Return values: - * %nfs_ok: @val encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_uint64_t(struct xdr_stream *xdr, u64 val) -{ - __be32 *p = xdr_reserve_space(xdr, XDR_UNIT * 2); - - if (unlikely(p == NULL)) - return nfserr_resource; - put_unaligned_be64(val, p); - return nfs_ok; -} - -#define nfsd4_encode_changeid4(x, v) nfsd4_encode_uint64_t(x, v) -#define nfsd4_encode_nfs_cookie4(x, v) nfsd4_encode_uint64_t(x, v) -#define nfsd4_encode_length4(x, v) nfsd4_encode_uint64_t(x, v) -#define nfsd4_encode_offset4(x, v) nfsd4_encode_uint64_t(x, v) - -/** - * nfsd4_encode_opaque_fixed - Encode a fixed-length XDR opaque type result - * @xdr: target XDR stream - * @data: pointer to data - * @size: length of data in bytes - * - * Return values: - * %nfs_ok: @data encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_opaque_fixed(struct xdr_stream *xdr, const void *data, - size_t size) -{ - __be32 *p = xdr_reserve_space(xdr, xdr_align_size(size)); - size_t pad = xdr_pad_size(size); - - if (unlikely(p == NULL)) - return nfserr_resource; - memcpy(p, data, size); - if (pad) - memset((char *)p + size, 0, pad); - return nfs_ok; -} - -/** - * nfsd4_encode_opaque - Encode a variable-length XDR opaque type result - * @xdr: target XDR stream - * @data: pointer to data - * @size: length of data in bytes - * - * Return values: - * %nfs_ok: @data encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_opaque(struct xdr_stream *xdr, const void *data, size_t size) -{ - size_t pad = xdr_pad_size(size); - __be32 *p; - - p = xdr_reserve_space(xdr, XDR_UNIT + xdr_align_size(size)); - if (unlikely(p == NULL)) - return nfserr_resource; - *p++ = cpu_to_be32(size); - memcpy(p, data, size); - if (pad) - memset((char *)p + size, 0, pad); - return nfs_ok; -} - -#define nfsd4_encode_component4(x, d, s) nfsd4_encode_opaque(x, d, s) - struct nfsd4_compound_state { struct svc_fh current_fh; struct svc_fh save_fh; @@ -642,17 +514,6 @@ svcxdr_decode_deviceid4(__be32 *p, struct nfsd4_deviceid *devid) return p; } -static inline __be32 -nfsd4_decode_deviceid4(struct xdr_stream *xdr, struct nfsd4_deviceid *devid) -{ - __be32 *p = xdr_inline_decode(xdr, NFS4_DEVICEID4_SIZE); - - if (unlikely(!p)) - return nfserr_bad_xdr; - svcxdr_decode_deviceid4(p, devid); - return nfs_ok; -} - struct nfsd4_layout_seg { u32 iomode; u64 offset; From 7332bdd618ef15c99f467559a9eaacab051f30ba Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:08 -0400 Subject: [PATCH 302/857] NFSD: Move pre-xdr'ed status codes out of nfsd.h nfsd.h is included by nearly every NFSD translation unit, so its include of reaches all of them, whether or not they touch NFSv4. That include existed solely for the block of pre-xdr'ed nfserr_* values at the end of the file: several of those values, such as nfserr_delay and nfserr_admin_revoked, are built from NFS4ERR_* constants defined in nfs4.h. The NFSD-internal error enum that follows the block (NFSERR_EOF and friends) is anchored at an impossible nfsstat4 value, thus it also needs nothing from nfs4.h. But these codes are used by all NFS versions, so their new home must be version-neutral. Move the pre-xdr'ed value block and the internal error enum into a new fs/nfsd/nfserr.h, which includes nfs4.h itself, and drop the nfs4.h include from nfsd.h. Include nfserr.h directly from each translation unit that references the pre-xdr'ed values or the internal error codes, rather than carrying it in a widely-included header. A translation unit that includes nfsd.h without using the error block no longer pulls in nfs4.h. The ones that reference the block can still reach it through nfserr.h. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/blocklayout.c | 1 + fs/nfsd/blocklayoutxdr.c | 1 + fs/nfsd/export.c | 1 + fs/nfsd/filecache.c | 1 + fs/nfsd/flexfilelayout.c | 1 + fs/nfsd/flexfilelayoutxdr.c | 1 + fs/nfsd/lockd.c | 1 + fs/nfsd/nfs2acl.c | 1 + fs/nfsd/nfs3acl.c | 1 + fs/nfsd/nfs3proc.c | 1 + fs/nfsd/nfs3xdr.c | 1 + fs/nfsd/nfs4acl.c | 1 + fs/nfsd/nfs4callback.c | 1 + fs/nfsd/nfs4idmap.c | 1 + fs/nfsd/nfs4layouts.c | 1 + fs/nfsd/nfs4proc.c | 1 + fs/nfsd/nfs4state.c | 1 + fs/nfsd/nfs4xdr.c | 1 + fs/nfsd/nfscache.c | 1 + fs/nfsd/nfsctl.c | 1 + fs/nfsd/nfsd.h | 143 -------------------------------- fs/nfsd/nfserr.h | 158 ++++++++++++++++++++++++++++++++++++ fs/nfsd/nfsfh.c | 1 + fs/nfsd/nfsproc.c | 1 + fs/nfsd/nfssvc.c | 2 + fs/nfsd/nfsxdr.c | 1 + fs/nfsd/vfs.c | 1 + 27 files changed, 184 insertions(+), 143 deletions(-) create mode 100644 fs/nfsd/nfserr.h diff --git a/fs/nfsd/blocklayout.c b/fs/nfsd/blocklayout.c index 5be7721c22c235..df02cf7464799c 100644 --- a/fs/nfsd/blocklayout.c +++ b/fs/nfsd/blocklayout.c @@ -9,6 +9,7 @@ #include +#include "nfserr.h" #include "blocklayoutxdr.h" #include "pnfs.h" #include "filecache.h" diff --git a/fs/nfsd/blocklayoutxdr.c b/fs/nfsd/blocklayoutxdr.c index 51cc2a07d7b384..a6589f5c878aca 100644 --- a/fs/nfsd/blocklayoutxdr.c +++ b/fs/nfsd/blocklayoutxdr.c @@ -8,6 +8,7 @@ #include #include "nfsd.h" +#include "nfserr.h" #include "blocklayoutxdr.h" #include "vfs.h" diff --git a/fs/nfsd/export.c b/fs/nfsd/export.c index 5aefb388cc27d1..f18de5f66ce667 100644 --- a/fs/nfsd/export.c +++ b/fs/nfsd/export.c @@ -21,6 +21,7 @@ #include #include "nfsd.h" +#include "nfserr.h" #include "nfsfh.h" #include "netns.h" #include "pnfs.h" diff --git a/fs/nfsd/filecache.c b/fs/nfsd/filecache.c index b9548eb17c77de..3539149cc75f77 100644 --- a/fs/nfsd/filecache.c +++ b/fs/nfsd/filecache.c @@ -43,6 +43,7 @@ #include "vfs.h" #include "nfsd.h" +#include "nfserr.h" #include "nfsfh.h" #include "netns.h" #include "filecache.h" diff --git a/fs/nfsd/flexfilelayout.c b/fs/nfsd/flexfilelayout.c index 6d531285ab439e..9f532418cac8ae 100644 --- a/fs/nfsd/flexfilelayout.c +++ b/fs/nfsd/flexfilelayout.c @@ -13,6 +13,7 @@ #include +#include "nfserr.h" #include "flexfilelayoutxdr.h" #include "pnfs.h" #include "vfs.h" diff --git a/fs/nfsd/flexfilelayoutxdr.c b/fs/nfsd/flexfilelayoutxdr.c index 374e52d3064a65..97d8a28dd3a04f 100644 --- a/fs/nfsd/flexfilelayoutxdr.c +++ b/fs/nfsd/flexfilelayoutxdr.c @@ -6,6 +6,7 @@ #include #include "nfsd.h" +#include "nfserr.h" #include "flexfilelayoutxdr.h" #define NFSDDBG_FACILITY NFSDDBG_PNFS diff --git a/fs/nfsd/lockd.c b/fs/nfsd/lockd.c index 72a5b499839d81..f5a4f352f8abf9 100644 --- a/fs/nfsd/lockd.c +++ b/fs/nfsd/lockd.c @@ -10,6 +10,7 @@ #include #include #include "nfsd.h" +#include "nfserr.h" #include "vfs.h" #define NFSDDBG_FACILITY NFSDDBG_LOCKD diff --git a/fs/nfsd/nfs2acl.c b/fs/nfsd/nfs2acl.c index 190f5a00190091..aba69dd278a136 100644 --- a/fs/nfsd/nfs2acl.c +++ b/fs/nfsd/nfs2acl.c @@ -6,6 +6,7 @@ */ #include "nfsd.h" +#include "nfserr.h" /* FIXME: nfsacl.h is a broken header */ #include #include diff --git a/fs/nfsd/nfs3acl.c b/fs/nfsd/nfs3acl.c index 6b6b289db63613..7183995182ab48 100644 --- a/fs/nfsd/nfs3acl.c +++ b/fs/nfsd/nfs3acl.c @@ -6,6 +6,7 @@ */ #include "nfsd.h" +#include "nfserr.h" /* FIXME: nfsacl.h is a broken header */ #include #include diff --git a/fs/nfsd/nfs3proc.c b/fs/nfsd/nfs3proc.c index 1df3c719e0da6c..4b3075c05b9793 100644 --- a/fs/nfsd/nfs3proc.c +++ b/fs/nfsd/nfs3proc.c @@ -13,6 +13,7 @@ #include "cache.h" #include "xdr3.h" #include "vfs.h" +#include "nfserr.h" #include "filecache.h" #include "trace.h" diff --git a/fs/nfsd/nfs3xdr.c b/fs/nfsd/nfs3xdr.c index e481804bb120c0..196bcc6edebb98 100644 --- a/fs/nfsd/nfs3xdr.c +++ b/fs/nfsd/nfs3xdr.c @@ -13,6 +13,7 @@ #include "auth.h" #include "netns.h" #include "vfs.h" +#include "nfserr.h" /* * Force construction of an empty post-op attr diff --git a/fs/nfsd/nfs4acl.c b/fs/nfsd/nfs4acl.c index 2c2f2fd89e8795..94f6ad381ebe5a 100644 --- a/fs/nfsd/nfs4acl.c +++ b/fs/nfsd/nfs4acl.c @@ -40,6 +40,7 @@ #include "nfsfh.h" #include "nfsd.h" +#include "nfserr.h" #include "acl.h" #include "vfs.h" diff --git a/fs/nfsd/nfs4callback.c b/fs/nfsd/nfs4callback.c index a901bbe67e0345..509195d488c9a3 100644 --- a/fs/nfsd/nfs4callback.c +++ b/fs/nfsd/nfs4callback.c @@ -37,6 +37,7 @@ #include #include #include "nfsd.h" +#include "nfserr.h" #include "state.h" #include "netns.h" #include "stats.h" diff --git a/fs/nfsd/nfs4idmap.c b/fs/nfsd/nfs4idmap.c index e9faf8b78f74a4..4e529759396342 100644 --- a/fs/nfsd/nfs4idmap.c +++ b/fs/nfsd/nfs4idmap.c @@ -41,6 +41,7 @@ #include "auth.h" #include "idmap.h" #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "vfs.h" diff --git a/fs/nfsd/nfs4layouts.c b/fs/nfsd/nfs4layouts.c index 22bcb6d09f7037..4187202f9acca5 100644 --- a/fs/nfsd/nfs4layouts.c +++ b/fs/nfsd/nfs4layouts.c @@ -9,6 +9,7 @@ #include #include +#include "nfserr.h" #include "pnfs.h" #include "netns.h" #include "trace.h" diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 2d43ff327b8708..0bbf781d4ac591 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -51,6 +51,7 @@ #include "netns.h" #include "acl.h" #include "pnfs.h" +#include "nfserr.h" #include "trace.h" static bool inter_copy_offload_enable; diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 18e17232cf945f..a3bfcefa7bc9aa 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -52,6 +52,7 @@ #include "vfs.h" #include "current_stateid.h" #include "stats.h" +#include "nfserr.h" #include "netns.h" #include "pnfs.h" diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 05d64733094b2a..7ccc7c897b00cb 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -54,6 +54,7 @@ #include "xdr4.h" #include "vfs.h" #include "state.h" +#include "nfserr.h" #include "cache.h" #include "netns.h" #include "pnfs.h" diff --git a/fs/nfsd/nfscache.c b/fs/nfsd/nfscache.c index c7db532c852376..80364b91331a2a 100644 --- a/fs/nfsd/nfscache.c +++ b/fs/nfsd/nfscache.c @@ -19,6 +19,7 @@ #include #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "stats.h" #include "cache.h" diff --git a/fs/nfsd/nfsctl.c b/fs/nfsd/nfsctl.c index adb032b7311a31..8521b648a970e1 100644 --- a/fs/nfsd/nfsctl.c +++ b/fs/nfsd/nfsctl.c @@ -23,6 +23,7 @@ #include "idmap.h" #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "stats.h" #include "cache.h" diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 76a69d9a4e7332..384a2498b6a295 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -15,7 +15,6 @@ #include #include #include -#include #include #include @@ -198,148 +197,6 @@ void nfsd_lockd_init(void); void nfsd_lockd_shutdown(void); -/* - * These macros provide pre-xdr'ed values for faster operation. - */ -#define nfs_ok cpu_to_be32(NFS_OK) -#define nfserr_perm cpu_to_be32(NFSERR_PERM) -#define nfserr_noent cpu_to_be32(NFSERR_NOENT) -#define nfserr_io cpu_to_be32(NFSERR_IO) -#define nfserr_nxio cpu_to_be32(NFSERR_NXIO) -#define nfserr_acces cpu_to_be32(NFSERR_ACCES) -#define nfserr_exist cpu_to_be32(NFSERR_EXIST) -#define nfserr_xdev cpu_to_be32(NFSERR_XDEV) -#define nfserr_nodev cpu_to_be32(NFSERR_NODEV) -#define nfserr_notdir cpu_to_be32(NFSERR_NOTDIR) -#define nfserr_isdir cpu_to_be32(NFSERR_ISDIR) -#define nfserr_inval cpu_to_be32(NFSERR_INVAL) -#define nfserr_fbig cpu_to_be32(NFSERR_FBIG) -#define nfserr_nospc cpu_to_be32(NFSERR_NOSPC) -#define nfserr_rofs cpu_to_be32(NFSERR_ROFS) -#define nfserr_mlink cpu_to_be32(NFSERR_MLINK) -#define nfserr_nametoolong cpu_to_be32(NFSERR_NAMETOOLONG) -#define nfserr_notempty cpu_to_be32(NFSERR_NOTEMPTY) -#define nfserr_dquot cpu_to_be32(NFSERR_DQUOT) -#define nfserr_stale cpu_to_be32(NFSERR_STALE) -#define nfserr_remote cpu_to_be32(NFSERR_REMOTE) -#define nfserr_wflush cpu_to_be32(NFSERR_WFLUSH) -#define nfserr_badhandle cpu_to_be32(NFSERR_BADHANDLE) -#define nfserr_notsync cpu_to_be32(NFSERR_NOT_SYNC) -#define nfserr_badcookie cpu_to_be32(NFSERR_BAD_COOKIE) -#define nfserr_notsupp cpu_to_be32(NFSERR_NOTSUPP) -#define nfserr_toosmall cpu_to_be32(NFSERR_TOOSMALL) -#define nfserr_serverfault cpu_to_be32(NFSERR_SERVERFAULT) -#define nfserr_badtype cpu_to_be32(NFSERR_BADTYPE) -#define nfserr_jukebox cpu_to_be32(NFSERR_JUKEBOX) -#define nfserr_denied cpu_to_be32(NFSERR_DENIED) -#define nfserr_deadlock cpu_to_be32(NFSERR_DEADLOCK) -#define nfserr_expired cpu_to_be32(NFSERR_EXPIRED) -#define nfserr_bad_cookie cpu_to_be32(NFSERR_BAD_COOKIE) -#define nfserr_same cpu_to_be32(NFSERR_SAME) -#define nfserr_clid_inuse cpu_to_be32(NFSERR_CLID_INUSE) -#define nfserr_stale_clientid cpu_to_be32(NFSERR_STALE_CLIENTID) -#define nfserr_resource cpu_to_be32(NFSERR_RESOURCE) -#define nfserr_moved cpu_to_be32(NFSERR_MOVED) -#define nfserr_nofilehandle cpu_to_be32(NFSERR_NOFILEHANDLE) -#define nfserr_minor_vers_mismatch cpu_to_be32(NFSERR_MINOR_VERS_MISMATCH) -#define nfserr_share_denied cpu_to_be32(NFSERR_SHARE_DENIED) -#define nfserr_stale_stateid cpu_to_be32(NFSERR_STALE_STATEID) -#define nfserr_old_stateid cpu_to_be32(NFSERR_OLD_STATEID) -#define nfserr_bad_stateid cpu_to_be32(NFSERR_BAD_STATEID) -#define nfserr_bad_seqid cpu_to_be32(NFSERR_BAD_SEQID) -#define nfserr_symlink cpu_to_be32(NFSERR_SYMLINK) -#define nfserr_not_same cpu_to_be32(NFSERR_NOT_SAME) -#define nfserr_lock_range cpu_to_be32(NFSERR_LOCK_RANGE) -#define nfserr_restorefh cpu_to_be32(NFSERR_RESTOREFH) -#define nfserr_attrnotsupp cpu_to_be32(NFSERR_ATTRNOTSUPP) -#define nfserr_bad_xdr cpu_to_be32(NFSERR_BAD_XDR) -#define nfserr_openmode cpu_to_be32(NFSERR_OPENMODE) -#define nfserr_badowner cpu_to_be32(NFSERR_BADOWNER) -#define nfserr_locks_held cpu_to_be32(NFSERR_LOCKS_HELD) -#define nfserr_op_illegal cpu_to_be32(NFSERR_OP_ILLEGAL) -#define nfserr_grace cpu_to_be32(NFSERR_GRACE) -#define nfserr_no_grace cpu_to_be32(NFSERR_NO_GRACE) -#define nfserr_reclaim_bad cpu_to_be32(NFSERR_RECLAIM_BAD) -#define nfserr_badname cpu_to_be32(NFSERR_BADNAME) -#define nfserr_admin_revoked cpu_to_be32(NFS4ERR_ADMIN_REVOKED) -#define nfserr_cb_path_down cpu_to_be32(NFSERR_CB_PATH_DOWN) -#define nfserr_locked cpu_to_be32(NFSERR_LOCKED) -#define nfserr_wrongsec cpu_to_be32(NFSERR_WRONGSEC) -#define nfserr_delay cpu_to_be32(NFS4ERR_DELAY) -#define nfserr_badiomode cpu_to_be32(NFS4ERR_BADIOMODE) -#define nfserr_badlayout cpu_to_be32(NFS4ERR_BADLAYOUT) -#define nfserr_bad_session_digest cpu_to_be32(NFS4ERR_BAD_SESSION_DIGEST) -#define nfserr_badsession cpu_to_be32(NFS4ERR_BADSESSION) -#define nfserr_badslot cpu_to_be32(NFS4ERR_BADSLOT) -#define nfserr_complete_already cpu_to_be32(NFS4ERR_COMPLETE_ALREADY) -#define nfserr_conn_not_bound_to_session cpu_to_be32(NFS4ERR_CONN_NOT_BOUND_TO_SESSION) -#define nfserr_deleg_already_wanted cpu_to_be32(NFS4ERR_DELEG_ALREADY_WANTED) -#define nfserr_back_chan_busy cpu_to_be32(NFS4ERR_BACK_CHAN_BUSY) -#define nfserr_layouttrylater cpu_to_be32(NFS4ERR_LAYOUTTRYLATER) -#define nfserr_layoutunavailable cpu_to_be32(NFS4ERR_LAYOUTUNAVAILABLE) -#define nfserr_nomatching_layout cpu_to_be32(NFS4ERR_NOMATCHING_LAYOUT) -#define nfserr_recallconflict cpu_to_be32(NFS4ERR_RECALLCONFLICT) -#define nfserr_unknown_layouttype cpu_to_be32(NFS4ERR_UNKNOWN_LAYOUTTYPE) -#define nfserr_seq_misordered cpu_to_be32(NFS4ERR_SEQ_MISORDERED) -#define nfserr_sequence_pos cpu_to_be32(NFS4ERR_SEQUENCE_POS) -#define nfserr_req_too_big cpu_to_be32(NFS4ERR_REQ_TOO_BIG) -#define nfserr_rep_too_big cpu_to_be32(NFS4ERR_REP_TOO_BIG) -#define nfserr_rep_too_big_to_cache cpu_to_be32(NFS4ERR_REP_TOO_BIG_TO_CACHE) -#define nfserr_retry_uncached_rep cpu_to_be32(NFS4ERR_RETRY_UNCACHED_REP) -#define nfserr_unsafe_compound cpu_to_be32(NFS4ERR_UNSAFE_COMPOUND) -#define nfserr_too_many_ops cpu_to_be32(NFS4ERR_TOO_MANY_OPS) -#define nfserr_op_not_in_session cpu_to_be32(NFS4ERR_OP_NOT_IN_SESSION) -#define nfserr_hash_alg_unsupp cpu_to_be32(NFS4ERR_HASH_ALG_UNSUPP) -#define nfserr_clientid_busy cpu_to_be32(NFS4ERR_CLIENTID_BUSY) -#define nfserr_pnfs_io_hole cpu_to_be32(NFS4ERR_PNFS_IO_HOLE) -#define nfserr_seq_false_retry cpu_to_be32(NFS4ERR_SEQ_FALSE_RETRY) -#define nfserr_bad_high_slot cpu_to_be32(NFS4ERR_BAD_HIGH_SLOT) -#define nfserr_deadsession cpu_to_be32(NFS4ERR_DEADSESSION) -#define nfserr_encr_alg_unsupp cpu_to_be32(NFS4ERR_ENCR_ALG_UNSUPP) -#define nfserr_pnfs_no_layout cpu_to_be32(NFS4ERR_PNFS_NO_LAYOUT) -#define nfserr_not_only_op cpu_to_be32(NFS4ERR_NOT_ONLY_OP) -#define nfserr_wrong_cred cpu_to_be32(NFS4ERR_WRONG_CRED) -#define nfserr_wrong_type cpu_to_be32(NFS4ERR_WRONG_TYPE) -#define nfserr_dirdeleg_unavail cpu_to_be32(NFS4ERR_DIRDELEG_UNAVAIL) -#define nfserr_reject_deleg cpu_to_be32(NFS4ERR_REJECT_DELEG) -#define nfserr_returnconflict cpu_to_be32(NFS4ERR_RETURNCONFLICT) -#define nfserr_deleg_revoked cpu_to_be32(NFS4ERR_DELEG_REVOKED) -#define nfserr_partner_notsupp cpu_to_be32(NFS4ERR_PARTNER_NOTSUPP) -#define nfserr_partner_no_auth cpu_to_be32(NFS4ERR_PARTNER_NO_AUTH) -#define nfserr_union_notsupp cpu_to_be32(NFS4ERR_UNION_NOTSUPP) -#define nfserr_offload_denied cpu_to_be32(NFS4ERR_OFFLOAD_DENIED) -#define nfserr_wrong_lfs cpu_to_be32(NFS4ERR_WRONG_LFS) -#define nfserr_badlabel cpu_to_be32(NFS4ERR_BADLABEL) -#define nfserr_file_open cpu_to_be32(NFS4ERR_FILE_OPEN) -#define nfserr_xattr2big cpu_to_be32(NFS4ERR_XATTR2BIG) -#define nfserr_noxattr cpu_to_be32(NFS4ERR_NOXATTR) - -/* - * Error codes for internal use. These are based at an impossible - * nfsstat4 value so that, once converted to be32, they cannot conflict - * with any value defined by the protocol (compare the nlm__int__* codes - * in fs/lockd/lockd.h). - */ -enum { -/* end-of-file indicator in readdir */ - NFSERR_EOF = 30000, -#define nfserr_eof cpu_to_be32(NFSERR_EOF) - -/* replay detected */ - NFSERR_REPLAY_ME, -#define nfserr_replay_me cpu_to_be32(NFSERR_REPLAY_ME) - -/* nfs41 replay detected */ - NFSERR_REPLAY_CACHE, -#define nfserr_replay_cache cpu_to_be32(NFSERR_REPLAY_CACHE) - -/* symlink found where dir expected - handled differently to - * other symlink found errors by NFSv3. - */ - NFSERR_SYMLINK_NOT_DIR, -#define nfserr_symlink_not_dir cpu_to_be32(NFSERR_SYMLINK_NOT_DIR) -}; - #ifdef CONFIG_NFSD_V4 /* before processing a COMPOUND operation, we have to check that there diff --git a/fs/nfsd/nfserr.h b/fs/nfsd/nfserr.h new file mode 100644 index 00000000000000..9b9df7aab220d7 --- /dev/null +++ b/fs/nfsd/nfserr.h @@ -0,0 +1,158 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * Pre-xdr'ed nfsd error values and nfsd-internal error codes. + * + * Separated from nfsd.h so that nfsd.h itself does not have to pull + * in : the NFS4ERR_* values used below are the only + * reason that include was needed. + */ + +#ifndef LINUX_NFSD_NFSERR_H +#define LINUX_NFSD_NFSERR_H + +#include +#include + +/* + * These macros provide pre-xdr'ed values for faster operation. + */ +#define nfs_ok cpu_to_be32(NFS_OK) +#define nfserr_perm cpu_to_be32(NFSERR_PERM) +#define nfserr_noent cpu_to_be32(NFSERR_NOENT) +#define nfserr_io cpu_to_be32(NFSERR_IO) +#define nfserr_nxio cpu_to_be32(NFSERR_NXIO) +#define nfserr_acces cpu_to_be32(NFSERR_ACCES) +#define nfserr_exist cpu_to_be32(NFSERR_EXIST) +#define nfserr_xdev cpu_to_be32(NFSERR_XDEV) +#define nfserr_nodev cpu_to_be32(NFSERR_NODEV) +#define nfserr_notdir cpu_to_be32(NFSERR_NOTDIR) +#define nfserr_isdir cpu_to_be32(NFSERR_ISDIR) +#define nfserr_inval cpu_to_be32(NFSERR_INVAL) +#define nfserr_fbig cpu_to_be32(NFSERR_FBIG) +#define nfserr_nospc cpu_to_be32(NFSERR_NOSPC) +#define nfserr_rofs cpu_to_be32(NFSERR_ROFS) +#define nfserr_mlink cpu_to_be32(NFSERR_MLINK) +#define nfserr_nametoolong cpu_to_be32(NFSERR_NAMETOOLONG) +#define nfserr_notempty cpu_to_be32(NFSERR_NOTEMPTY) +#define nfserr_dquot cpu_to_be32(NFSERR_DQUOT) +#define nfserr_stale cpu_to_be32(NFSERR_STALE) +#define nfserr_remote cpu_to_be32(NFSERR_REMOTE) +#define nfserr_wflush cpu_to_be32(NFSERR_WFLUSH) +#define nfserr_badhandle cpu_to_be32(NFSERR_BADHANDLE) +#define nfserr_notsync cpu_to_be32(NFSERR_NOT_SYNC) +#define nfserr_badcookie cpu_to_be32(NFSERR_BAD_COOKIE) +#define nfserr_notsupp cpu_to_be32(NFSERR_NOTSUPP) +#define nfserr_toosmall cpu_to_be32(NFSERR_TOOSMALL) +#define nfserr_serverfault cpu_to_be32(NFSERR_SERVERFAULT) +#define nfserr_badtype cpu_to_be32(NFSERR_BADTYPE) +#define nfserr_jukebox cpu_to_be32(NFSERR_JUKEBOX) +#define nfserr_denied cpu_to_be32(NFSERR_DENIED) +#define nfserr_deadlock cpu_to_be32(NFSERR_DEADLOCK) +#define nfserr_expired cpu_to_be32(NFSERR_EXPIRED) +#define nfserr_bad_cookie cpu_to_be32(NFSERR_BAD_COOKIE) +#define nfserr_same cpu_to_be32(NFSERR_SAME) +#define nfserr_clid_inuse cpu_to_be32(NFSERR_CLID_INUSE) +#define nfserr_stale_clientid cpu_to_be32(NFSERR_STALE_CLIENTID) +#define nfserr_resource cpu_to_be32(NFSERR_RESOURCE) +#define nfserr_moved cpu_to_be32(NFSERR_MOVED) +#define nfserr_nofilehandle cpu_to_be32(NFSERR_NOFILEHANDLE) +#define nfserr_minor_vers_mismatch cpu_to_be32(NFSERR_MINOR_VERS_MISMATCH) +#define nfserr_share_denied cpu_to_be32(NFSERR_SHARE_DENIED) +#define nfserr_stale_stateid cpu_to_be32(NFSERR_STALE_STATEID) +#define nfserr_old_stateid cpu_to_be32(NFSERR_OLD_STATEID) +#define nfserr_bad_stateid cpu_to_be32(NFSERR_BAD_STATEID) +#define nfserr_bad_seqid cpu_to_be32(NFSERR_BAD_SEQID) +#define nfserr_symlink cpu_to_be32(NFSERR_SYMLINK) +#define nfserr_not_same cpu_to_be32(NFSERR_NOT_SAME) +#define nfserr_lock_range cpu_to_be32(NFSERR_LOCK_RANGE) +#define nfserr_restorefh cpu_to_be32(NFSERR_RESTOREFH) +#define nfserr_attrnotsupp cpu_to_be32(NFSERR_ATTRNOTSUPP) +#define nfserr_bad_xdr cpu_to_be32(NFSERR_BAD_XDR) +#define nfserr_openmode cpu_to_be32(NFSERR_OPENMODE) +#define nfserr_badowner cpu_to_be32(NFSERR_BADOWNER) +#define nfserr_locks_held cpu_to_be32(NFSERR_LOCKS_HELD) +#define nfserr_op_illegal cpu_to_be32(NFSERR_OP_ILLEGAL) +#define nfserr_grace cpu_to_be32(NFSERR_GRACE) +#define nfserr_no_grace cpu_to_be32(NFSERR_NO_GRACE) +#define nfserr_reclaim_bad cpu_to_be32(NFSERR_RECLAIM_BAD) +#define nfserr_badname cpu_to_be32(NFSERR_BADNAME) +#define nfserr_admin_revoked cpu_to_be32(NFS4ERR_ADMIN_REVOKED) +#define nfserr_cb_path_down cpu_to_be32(NFSERR_CB_PATH_DOWN) +#define nfserr_locked cpu_to_be32(NFSERR_LOCKED) +#define nfserr_wrongsec cpu_to_be32(NFSERR_WRONGSEC) +#define nfserr_delay cpu_to_be32(NFS4ERR_DELAY) +#define nfserr_badiomode cpu_to_be32(NFS4ERR_BADIOMODE) +#define nfserr_badlayout cpu_to_be32(NFS4ERR_BADLAYOUT) +#define nfserr_bad_session_digest cpu_to_be32(NFS4ERR_BAD_SESSION_DIGEST) +#define nfserr_badsession cpu_to_be32(NFS4ERR_BADSESSION) +#define nfserr_badslot cpu_to_be32(NFS4ERR_BADSLOT) +#define nfserr_complete_already cpu_to_be32(NFS4ERR_COMPLETE_ALREADY) +#define nfserr_conn_not_bound_to_session cpu_to_be32(NFS4ERR_CONN_NOT_BOUND_TO_SESSION) +#define nfserr_deleg_already_wanted cpu_to_be32(NFS4ERR_DELEG_ALREADY_WANTED) +#define nfserr_back_chan_busy cpu_to_be32(NFS4ERR_BACK_CHAN_BUSY) +#define nfserr_layouttrylater cpu_to_be32(NFS4ERR_LAYOUTTRYLATER) +#define nfserr_layoutunavailable cpu_to_be32(NFS4ERR_LAYOUTUNAVAILABLE) +#define nfserr_nomatching_layout cpu_to_be32(NFS4ERR_NOMATCHING_LAYOUT) +#define nfserr_recallconflict cpu_to_be32(NFS4ERR_RECALLCONFLICT) +#define nfserr_unknown_layouttype cpu_to_be32(NFS4ERR_UNKNOWN_LAYOUTTYPE) +#define nfserr_seq_misordered cpu_to_be32(NFS4ERR_SEQ_MISORDERED) +#define nfserr_sequence_pos cpu_to_be32(NFS4ERR_SEQUENCE_POS) +#define nfserr_req_too_big cpu_to_be32(NFS4ERR_REQ_TOO_BIG) +#define nfserr_rep_too_big cpu_to_be32(NFS4ERR_REP_TOO_BIG) +#define nfserr_rep_too_big_to_cache cpu_to_be32(NFS4ERR_REP_TOO_BIG_TO_CACHE) +#define nfserr_retry_uncached_rep cpu_to_be32(NFS4ERR_RETRY_UNCACHED_REP) +#define nfserr_unsafe_compound cpu_to_be32(NFS4ERR_UNSAFE_COMPOUND) +#define nfserr_too_many_ops cpu_to_be32(NFS4ERR_TOO_MANY_OPS) +#define nfserr_op_not_in_session cpu_to_be32(NFS4ERR_OP_NOT_IN_SESSION) +#define nfserr_hash_alg_unsupp cpu_to_be32(NFS4ERR_HASH_ALG_UNSUPP) +#define nfserr_clientid_busy cpu_to_be32(NFS4ERR_CLIENTID_BUSY) +#define nfserr_pnfs_io_hole cpu_to_be32(NFS4ERR_PNFS_IO_HOLE) +#define nfserr_seq_false_retry cpu_to_be32(NFS4ERR_SEQ_FALSE_RETRY) +#define nfserr_bad_high_slot cpu_to_be32(NFS4ERR_BAD_HIGH_SLOT) +#define nfserr_deadsession cpu_to_be32(NFS4ERR_DEADSESSION) +#define nfserr_encr_alg_unsupp cpu_to_be32(NFS4ERR_ENCR_ALG_UNSUPP) +#define nfserr_pnfs_no_layout cpu_to_be32(NFS4ERR_PNFS_NO_LAYOUT) +#define nfserr_not_only_op cpu_to_be32(NFS4ERR_NOT_ONLY_OP) +#define nfserr_wrong_cred cpu_to_be32(NFS4ERR_WRONG_CRED) +#define nfserr_wrong_type cpu_to_be32(NFS4ERR_WRONG_TYPE) +#define nfserr_dirdeleg_unavail cpu_to_be32(NFS4ERR_DIRDELEG_UNAVAIL) +#define nfserr_reject_deleg cpu_to_be32(NFS4ERR_REJECT_DELEG) +#define nfserr_returnconflict cpu_to_be32(NFS4ERR_RETURNCONFLICT) +#define nfserr_deleg_revoked cpu_to_be32(NFS4ERR_DELEG_REVOKED) +#define nfserr_partner_notsupp cpu_to_be32(NFS4ERR_PARTNER_NOTSUPP) +#define nfserr_partner_no_auth cpu_to_be32(NFS4ERR_PARTNER_NO_AUTH) +#define nfserr_union_notsupp cpu_to_be32(NFS4ERR_UNION_NOTSUPP) +#define nfserr_offload_denied cpu_to_be32(NFS4ERR_OFFLOAD_DENIED) +#define nfserr_wrong_lfs cpu_to_be32(NFS4ERR_WRONG_LFS) +#define nfserr_badlabel cpu_to_be32(NFS4ERR_BADLABEL) +#define nfserr_file_open cpu_to_be32(NFS4ERR_FILE_OPEN) +#define nfserr_xattr2big cpu_to_be32(NFS4ERR_XATTR2BIG) +#define nfserr_noxattr cpu_to_be32(NFS4ERR_NOXATTR) + +/* + * Error codes for internal use. These are based at an impossible + * nfsstat4 value so that, once converted to be32, they cannot conflict + * with any value defined by the protocol (compare the nlm__int__* codes + * in fs/lockd/lockd.h). + */ +enum { +/* end-of-file indicator in readdir */ + NFSERR_EOF = 30000, +#define nfserr_eof cpu_to_be32(NFSERR_EOF) + +/* replay detected */ + NFSERR_REPLAY_ME, +#define nfserr_replay_me cpu_to_be32(NFSERR_REPLAY_ME) + +/* nfs41 replay detected */ + NFSERR_REPLAY_CACHE, +#define nfserr_replay_cache cpu_to_be32(NFSERR_REPLAY_CACHE) + +/* symlink found where dir expected - handled differently to + * other symlink found errors by NFSv3. + */ + NFSERR_SYMLINK_NOT_DIR, +#define nfserr_symlink_not_dir cpu_to_be32(NFSERR_SYMLINK_NOT_DIR) +}; + +#endif /* LINUX_NFSD_NFSERR_H */ diff --git a/fs/nfsd/nfsfh.c b/fs/nfsd/nfsfh.c index c0a46784d525ae..fd721a5a6b37bd 100644 --- a/fs/nfsd/nfsfh.c +++ b/fs/nfsd/nfsfh.c @@ -13,6 +13,7 @@ #include #include #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "stats.h" #include "vfs.h" diff --git a/fs/nfsd/nfsproc.c b/fs/nfsd/nfsproc.c index 2a82fa64e47855..919acfba356a4f 100644 --- a/fs/nfsd/nfsproc.c +++ b/fs/nfsd/nfsproc.c @@ -10,6 +10,7 @@ #include "cache.h" #include "xdr.h" #include "vfs.h" +#include "nfserr.h" #include "trace.h" #define NFSDDBG_FACILITY NFSDDBG_PROC diff --git a/fs/nfsd/nfssvc.c b/fs/nfsd/nfssvc.c index 2edf716ea022ff..7f6ffbe7be289b 100644 --- a/fs/nfsd/nfssvc.c +++ b/fs/nfsd/nfssvc.c @@ -25,7 +25,9 @@ #include #include #include + #include "nfsd.h" +#include "nfserr.h" #include "cache.h" #include "vfs.h" #include "netns.h" diff --git a/fs/nfsd/nfsxdr.c b/fs/nfsd/nfsxdr.c index 019f0cc971a7aa..0961c13d6ab1d7 100644 --- a/fs/nfsd/nfsxdr.c +++ b/fs/nfsd/nfsxdr.c @@ -8,6 +8,7 @@ #include #include "vfs.h" +#include "nfserr.h" #include "xdr.h" #include "auth.h" diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index f65dad403ee08f..68fc2a45b64cc2 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -43,6 +43,7 @@ #endif /* CONFIG_NFSD_V4 */ #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "stats.h" #include "vfs.h" From a3c704a75678a0051474ba161dfaab1d8c900da7 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:09 -0400 Subject: [PATCH 303/857] NFSD: Remove two unused NFSv4 constants Neither COMPOUND_SLACK_SPACE nor NFSD_COURTESY_CLIENT_TIMEOUT has a remaining user. COMPOUND_SLACK_SPACE lost its last reference in commit ea8d7720b274 ("nfsd4: remove redundant encode buffer size checking"), which deleted the encode buffer-space check the macro fed; the comment above it still describes that departed check. NFSD_COURTESY_CLIENT_TIMEOUT is likewise unreferenced: the courteous-server code expires clients through the laundromat's reaper and conflict paths, never a fixed 24-hour timer. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-4-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfsd.h | 7 ------- 1 file changed, 7 deletions(-) diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 384a2498b6a295..27e5384cd84902 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -204,21 +204,14 @@ void nfsd_lockd_shutdown(void); * we might process an operation with side effects, and be unable to * tell the client that the operation succeeded. * - * COMPOUND_SLACK_SPACE - this is the minimum bytes of buffer space - * needed to encode an "ordinary" _successful_ operation. (GETATTR, - * READ, READDIR, and READLINK have their own buffer checks.) if we - * fall below this level, we fail the next operation with NFS4ERR_RESOURCE. - * * COMPOUND_ERR_SLACK_SPACE - this is the minimum bytes of buffer space * needed to encode an operation which has failed with NFS4ERR_RESOURCE. * care is taken to ensure that we never fall below this level for any * reason. */ -#define COMPOUND_SLACK_SPACE 140 /* OP_GETFH */ #define COMPOUND_ERR_SLACK_SPACE 16 /* OP_SETATTR */ #define NFSD_LAUNDROMAT_MINTIMEOUT 1 /* seconds */ -#define NFSD_COURTESY_CLIENT_TIMEOUT (24 * 60 * 60) /* seconds */ #define NFSD_CLIENT_MAX_TRIM_PER_RUN 128 #define NFS4_CLIENTS_PER_GB 1024 #define NFSD_DELEGRETURN_TIMEOUT (HZ / 34) /* 30ms */ From b450e8349f794f951eb3cef9b678f7728cdf5db8 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:10 -0400 Subject: [PATCH 304/857] NFSD: Relocate NFSv4-internal constants to state.h The COMPOUND encode-slack sizes and the state-management timeouts at the tail of nfsd.h are evaluated only by NFSv4 code (nfs4state.c, nfs4proc.c, and nfs4xdr.c). They nonetheless sit in nfsd.h, where every NFSD translation unit, including the NFSv2 and NFSv3 paths that have no use for them, has to parse them. All three consumers already reach state.h through xdr4.h, so move the block there. nfsd.h keeps the NFSv4 prototypes for now. Only the pure constants move. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-5-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfsd.h | 18 ------------------ fs/nfsd/state.h | 19 +++++++++++++++++++ 2 files changed, 19 insertions(+), 18 deletions(-) diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 27e5384cd84902..1886f6d4292910 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -199,24 +199,6 @@ void nfsd_lockd_shutdown(void); #ifdef CONFIG_NFSD_V4 -/* before processing a COMPOUND operation, we have to check that there - * is enough space in the buffer for XDR encode to succeed. otherwise, - * we might process an operation with side effects, and be unable to - * tell the client that the operation succeeded. - * - * COMPOUND_ERR_SLACK_SPACE - this is the minimum bytes of buffer space - * needed to encode an operation which has failed with NFS4ERR_RESOURCE. - * care is taken to ensure that we never fall below this level for any - * reason. - */ -#define COMPOUND_ERR_SLACK_SPACE 16 /* OP_SETATTR */ - -#define NFSD_LAUNDROMAT_MINTIMEOUT 1 /* seconds */ -#define NFSD_CLIENT_MAX_TRIM_PER_RUN 128 -#define NFS4_CLIENTS_PER_GB 1024 -#define NFSD_DELEGRETURN_TIMEOUT (HZ / 34) /* 30ms */ -#define NFSD_CB_GETATTR_TIMEOUT NFSD_DELEGRETURN_TIMEOUT - extern int nfsd4_is_junction(struct dentry *dentry); extern int register_cld_notifier(void); extern void unregister_cld_notifier(void); diff --git a/fs/nfsd/state.h b/fs/nfsd/state.h index 2d00a411c6634e..c4627dc91e2067 100644 --- a/fs/nfsd/state.h +++ b/fs/nfsd/state.h @@ -45,6 +45,25 @@ #include "nfsfh.h" #include "nfsd.h" +/* + * Before processing a COMPOUND operation, we have to check that there + * is enough space in the buffer for XDR encode to succeed. otherwise, + * we might process an operation with side effects, and be unable to + * tell the client that the operation succeeded. + * + * COMPOUND_ERR_SLACK_SPACE - this is the minimum bytes of buffer space + * needed to encode an operation which has failed with NFS4ERR_RESOURCE. + * care is taken to ensure that we never fall below this level for any + * reason. + */ +#define COMPOUND_ERR_SLACK_SPACE 16 /* OP_SETATTR */ + +#define NFSD_LAUNDROMAT_MINTIMEOUT 1 /* seconds */ +#define NFSD_CLIENT_MAX_TRIM_PER_RUN 128 +#define NFS4_CLIENTS_PER_GB 1024 +#define NFSD_DELEGRETURN_TIMEOUT (HZ / 34) /* 30ms */ +#define NFSD_CB_GETATTR_TIMEOUT NFSD_DELEGRETURN_TIMEOUT + typedef struct { u32 cl_boot; u32 cl_id; From 983e0fdccc8ce23a678fcbf4e952d426f58f5838 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:11 -0400 Subject: [PATCH 305/857] NFSD: Evacuate NFSv4 entry-point prototypes from nfsd.h nfsd.h is included by nearly every NFSD translation unit, yet the two blocks of NFSv4 lifecycle and control prototypes it carries are referenced by only seven of them (out of over two dozen). The remaining consumers, including the NFSv2 and NFSv3 ACL and XDR paths, parse these declarations for no benefit. The declarations cannot simply move into an NFSv4-only header such as state.h: nfssvc.c, nfsctl.c, vfs.c, and export.c call the routines unconditionally and rely on the CONFIG_NFSD_V4=n stubs, and pulling the heavy state.h types into those lean translation units to obtain a handful of prototypes would trade one form of coupling for a worse one. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-6-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/export.c | 1 + fs/nfsd/nfs4ctl.h | 83 +++++++++++++++++++++++++++++++++++++++++++ fs/nfsd/nfs4proc.c | 1 + fs/nfsd/nfs4recover.c | 1 + fs/nfsd/nfs4state.c | 1 + fs/nfsd/nfsctl.c | 1 + fs/nfsd/nfsd.h | 63 -------------------------------- fs/nfsd/nfssvc.c | 1 + fs/nfsd/vfs.c | 1 + 9 files changed, 90 insertions(+), 63 deletions(-) create mode 100644 fs/nfsd/nfs4ctl.h diff --git a/fs/nfsd/export.c b/fs/nfsd/export.c index f18de5f66ce667..961d660e2c627f 100644 --- a/fs/nfsd/export.c +++ b/fs/nfsd/export.c @@ -22,6 +22,7 @@ #include "nfsd.h" #include "nfserr.h" +#include "nfs4ctl.h" #include "nfsfh.h" #include "netns.h" #include "pnfs.h" diff --git a/fs/nfsd/nfs4ctl.h b/fs/nfsd/nfs4ctl.h new file mode 100644 index 00000000000000..bcec4c4ef1d536 --- /dev/null +++ b/fs/nfsd/nfs4ctl.h @@ -0,0 +1,83 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * Entry points by which the knfsd core drives the optional NFSv4 + * subsystem: state lifecycle, the laundromat workqueue, the recovery + * directory, junctions, the CLD notifier, and leases-net setup. + * + * Separated from nfsd.h so that the many translation units that + * include nfsd.h but call none of these -- among them the NFSv2 and + * NFSv3 paths -- do not have to parse them. The CONFIG_NFSD_V4=n + * stubs let the version-agnostic callers invoke the routines + * unconditionally. + */ + +#ifndef LINUX_NFSD_NFS4CTL_H +#define LINUX_NFSD_NFS4CTL_H + +#include +#include + +struct net; +struct inode; +struct dentry; +struct svc_rqst; +struct nfsd_net; + +#ifdef CONFIG_NFSD_V4 +extern unsigned long max_delegations; +int nfsd4_init_slabs(void); +void nfsd4_free_slabs(void); +int nfs4_state_start(void); +int nfs4_state_start_net(struct net *net); +void nfs4_state_shutdown(void); +void nfs4_state_shutdown_net(struct net *net); +int nfs4_reset_recoverydir(char *recdir); +char * nfs4_recoverydir(void); +bool nfsd4_spo_must_allow(struct svc_rqst *rqstp); +int nfsd4_create_laundry_wq(void); +void nfsd4_destroy_laundry_wq(void); +bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, struct inode *inode); + +extern int nfsd4_is_junction(struct dentry *dentry); +extern int register_cld_notifier(void); +extern void unregister_cld_notifier(void); +#ifdef CONFIG_NFSD_V4_2_INTER_SSC +extern void nfsd4_ssc_init_umount_work(struct nfsd_net *nn); +#endif + +extern void nfsd4_init_leases_net(struct nfsd_net *nn); + +#else /* CONFIG_NFSD_V4 */ +static inline int nfsd4_init_slabs(void) { return 0; } +static inline void nfsd4_free_slabs(void) { } +static inline int nfs4_state_start(void) { return 0; } +static inline int nfs4_state_start_net(struct net *net) { return 0; } +static inline void nfs4_state_shutdown(void) { } +static inline void nfs4_state_shutdown_net(struct net *net) { } +static inline int nfs4_reset_recoverydir(char *recdir) { return 0; } +static inline char * nfs4_recoverydir(void) {return NULL; } +static inline bool nfsd4_spo_must_allow(struct svc_rqst *rqstp) +{ + return false; +} +static inline int nfsd4_create_laundry_wq(void) { return 0; }; +static inline void nfsd4_destroy_laundry_wq(void) {}; +static inline bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, + struct inode *inode) +{ + return false; +} + +static inline int nfsd4_is_junction(struct dentry *dentry) +{ + return 0; +} + +static inline void nfsd4_init_leases_net(struct nfsd_net *nn) { }; + +#define register_cld_notifier() 0 +#define unregister_cld_notifier() do { } while(0) + +#endif /* CONFIG_NFSD_V4 */ + +#endif /* LINUX_NFSD_NFS4CTL_H */ diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 0bbf781d4ac591..d4265ee3c73dee 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -46,6 +46,7 @@ #include "idmap.h" #include "cache.h" #include "xdr4.h" +#include "nfs4ctl.h" #include "vfs.h" #include "current_stateid.h" #include "netns.h" diff --git a/fs/nfsd/nfs4recover.c b/fs/nfsd/nfs4recover.c index d513971fb119d5..aee3a0b22d1cb0 100644 --- a/fs/nfsd/nfs4recover.c +++ b/fs/nfsd/nfs4recover.c @@ -47,6 +47,7 @@ #include #include "nfsd.h" +#include "nfs4ctl.h" #include "state.h" #include "vfs.h" #include "netns.h" diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index a3bfcefa7bc9aa..e78d4fe5fbfb59 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -48,6 +48,7 @@ #include #include "xdr4.h" +#include "nfs4ctl.h" #include "xdr4cb.h" #include "vfs.h" #include "current_stateid.h" diff --git a/fs/nfsd/nfsctl.c b/fs/nfsd/nfsctl.c index 8521b648a970e1..4e5e083d847758 100644 --- a/fs/nfsd/nfsctl.c +++ b/fs/nfsd/nfsctl.c @@ -24,6 +24,7 @@ #include "idmap.h" #include "nfsd.h" #include "nfserr.h" +#include "nfs4ctl.h" #include "netns.h" #include "stats.h" #include "cache.h" diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 1886f6d4292910..1cccf5a3c034bd 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -151,45 +151,6 @@ static inline int nfsd_v4client(struct svc_rqst *rq) return rq && rq->rq_prog == NFS_PROGRAM && rq->rq_vers == 4; } -/* - * NFSv4 State - */ -#ifdef CONFIG_NFSD_V4 -extern unsigned long max_delegations; -int nfsd4_init_slabs(void); -void nfsd4_free_slabs(void); -int nfs4_state_start(void); -int nfs4_state_start_net(struct net *net); -void nfs4_state_shutdown(void); -void nfs4_state_shutdown_net(struct net *net); -int nfs4_reset_recoverydir(char *recdir); -char * nfs4_recoverydir(void); -bool nfsd4_spo_must_allow(struct svc_rqst *rqstp); -int nfsd4_create_laundry_wq(void); -void nfsd4_destroy_laundry_wq(void); -bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, struct inode *inode); -#else -static inline int nfsd4_init_slabs(void) { return 0; } -static inline void nfsd4_free_slabs(void) { } -static inline int nfs4_state_start(void) { return 0; } -static inline int nfs4_state_start_net(struct net *net) { return 0; } -static inline void nfs4_state_shutdown(void) { } -static inline void nfs4_state_shutdown_net(struct net *net) { } -static inline int nfs4_reset_recoverydir(char *recdir) { return 0; } -static inline char * nfs4_recoverydir(void) {return NULL; } -static inline bool nfsd4_spo_must_allow(struct svc_rqst *rqstp) -{ - return false; -} -static inline int nfsd4_create_laundry_wq(void) { return 0; }; -static inline void nfsd4_destroy_laundry_wq(void) {}; -static inline bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, - struct inode *inode) -{ - return false; -} -#endif - /* * lockd binding */ @@ -197,28 +158,4 @@ void nfsd_lockd_init(void); void nfsd_lockd_shutdown(void); -#ifdef CONFIG_NFSD_V4 - -extern int nfsd4_is_junction(struct dentry *dentry); -extern int register_cld_notifier(void); -extern void unregister_cld_notifier(void); -#ifdef CONFIG_NFSD_V4_2_INTER_SSC -extern void nfsd4_ssc_init_umount_work(struct nfsd_net *nn); -#endif - -extern void nfsd4_init_leases_net(struct nfsd_net *nn); - -#else /* CONFIG_NFSD_V4 */ -static inline int nfsd4_is_junction(struct dentry *dentry) -{ - return 0; -} - -static inline void nfsd4_init_leases_net(struct nfsd_net *nn) { }; - -#define register_cld_notifier() 0 -#define unregister_cld_notifier() do { } while(0) - -#endif /* CONFIG_NFSD_V4 */ - #endif /* LINUX_NFSD_NFSD_H */ diff --git a/fs/nfsd/nfssvc.c b/fs/nfsd/nfssvc.c index 7f6ffbe7be289b..9cc8489978a367 100644 --- a/fs/nfsd/nfssvc.c +++ b/fs/nfsd/nfssvc.c @@ -28,6 +28,7 @@ #include "nfsd.h" #include "nfserr.h" +#include "nfs4ctl.h" #include "cache.h" #include "vfs.h" #include "netns.h" diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index 68fc2a45b64cc2..cbd1e34a548a83 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -44,6 +44,7 @@ #include "nfsd.h" #include "nfserr.h" +#include "nfs4ctl.h" #include "netns.h" #include "stats.h" #include "vfs.h" From f45d05e4c3ee08cfaf18709c9b9529f76389b471 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:12 -0400 Subject: [PATCH 306/857] NFSD: Move nfsd_v4client() out of nfsd.h nfsd_v4client() is the last user in nfsd.h of XDR-defined item references. Once this helper moves out, and the other XDR-specific headers nfsd.h includes have nothing left to provide, and can be dropped. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-7-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfsd.h | 5 +---- fs/nfsd/nfssvc.c | 5 +++++ 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 1cccf5a3c034bd..64315890eef589 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -146,10 +146,7 @@ extern u64 nfsd_io_cache_write __read_mostly; extern int nfsd_max_blksize; -static inline int nfsd_v4client(struct svc_rqst *rq) -{ - return rq && rq->rq_prog == NFS_PROGRAM && rq->rq_vers == 4; -} +bool nfsd_v4client(struct svc_rqst *rqstp); /* * lockd binding diff --git a/fs/nfsd/nfssvc.c b/fs/nfsd/nfssvc.c index 9cc8489978a367..c04ef9d180ceab 100644 --- a/fs/nfsd/nfssvc.c +++ b/fs/nfsd/nfssvc.c @@ -207,6 +207,11 @@ int nfsd_minorversion(struct nfsd_net *nn, u32 minorversion, enum vers_op change return 0; } +bool nfsd_v4client(struct svc_rqst *rqstp) +{ + return rqstp && rqstp->rq_prog == NFS_PROGRAM && rqstp->rq_vers == 4; +} + bool nfsd_net_try_get(struct net *net) __must_hold(rcu) { struct nfsd_net *nn = net_generic(net, nfsd_net_id); From 63d60f12103c2fb50c315ee4758ce8e4c8bb9cda Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 20 Jul 2026 12:37:24 -0400 Subject: [PATCH 307/857] NFSD: Map flex file layout IDs through the request's user namespace nfsd4_ff_proc_layoutget() and nfsd4_ff_encode_layoutget() translate the file's owner and group with init_user_ns, but every other identity nfsd places on the wire goes through nfsd_user_namespace(). When the transport carries a credential from a non-initial user namespace, the flex file layout reports host-global IDs. The client copies those IDs into the AUTH_SYS credential it presents to the data server, and svcauth_unix_accept() resolves that credential in the transport's namespace, so data server I/O runs under an identity unrelated to the file's owner. Switching to the request's namespace introduces a second hazard. from_kuid() returns (uid_t)-1 when the target namespace has no mapping for the owner, and the IOMODE_READ arm adds one to that result to derive an identity for which the data server denies writes. The addition would wrap to zero, handing the client uid 0 instead of an identity distinct from the owner. Translate both IDs in nfsd4_ff_proc_layoutget(), which has the svc_rqst, and carry the wire values in struct pnfs_ff_layout. from_kuid_munged() substitutes overflowuid for an unmapped owner and thus never returns (uid_t)-1, so the increment cannot wrap to zero. Fixes: 9b9960a0ca47 ("nfsd: Add a super simple flex file server") Cc: stable@vger.kernel.org Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260720141442.783935-1-cel@kernel.org?part=3 Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260720163724.810227-1-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/flexfilelayout.c | 20 ++++++++++++-------- fs/nfsd/flexfilelayoutxdr.c | 4 ++-- fs/nfsd/flexfilelayoutxdr.h | 5 +++-- 3 files changed, 17 insertions(+), 12 deletions(-) diff --git a/fs/nfsd/flexfilelayout.c b/fs/nfsd/flexfilelayout.c index 9f532418cac8ae..c6e6b9106a8393 100644 --- a/fs/nfsd/flexfilelayout.c +++ b/fs/nfsd/flexfilelayout.c @@ -15,6 +15,7 @@ #include "nfserr.h" #include "flexfilelayoutxdr.h" +#include "auth.h" #include "pnfs.h" #include "vfs.h" @@ -24,10 +25,10 @@ static __be32 nfsd4_ff_proc_layoutget(struct svc_rqst *rqstp, struct inode *inode, const struct svc_fh *fhp, struct nfsd4_layoutget *args) { + struct user_namespace *userns = nfsd_user_namespace(rqstp); struct nfsd4_layout_seg *seg = &args->lg_seg; u32 device_generation = 0; int error; - uid_t u; struct pnfs_ff_layout *fl; @@ -50,13 +51,16 @@ nfsd4_ff_proc_layoutget(struct svc_rqst *rqstp, struct inode *inode, fl->flags = FF_FLAGS_NO_LAYOUTCOMMIT | FF_FLAGS_NO_IO_THRU_MDS | FF_FLAGS_NO_READ_IO; - /* Do not allow a IOMODE_READ segment to have write pemissions */ - if (seg->iomode == IOMODE_READ) { - u = from_kuid(&init_user_ns, inode->i_uid) + 1; - fl->uid = make_kuid(&init_user_ns, u); - } else - fl->uid = inode->i_uid; - fl->gid = inode->i_gid; + fl->uid = from_kuid_munged(userns, inode->i_uid); + fl->gid = from_kgid_munged(userns, inode->i_gid); + + /* + * Do not allow an IOMODE_READ segment to have write permissions. + * The group is left intact so group-readable files stay readable; + * nfsd_setuser() squashes an unmapped uid to the export's anon ID. + */ + if (seg->iomode == IOMODE_READ) + fl->uid++; error = nfsd4_set_deviceid(&fl->deviceid, fhp, device_generation); if (error) diff --git a/fs/nfsd/flexfilelayoutxdr.c b/fs/nfsd/flexfilelayoutxdr.c index 97d8a28dd3a04f..c12bb7c371b0e5 100644 --- a/fs/nfsd/flexfilelayoutxdr.c +++ b/fs/nfsd/flexfilelayoutxdr.c @@ -33,8 +33,8 @@ nfsd4_ff_encode_layoutget(struct xdr_stream *xdr, fh_len = 4 + xdr_align_size(fl->fh.size); - uid.len = sprintf(uid.buf, "%u", from_kuid(&init_user_ns, fl->uid)); - gid.len = sprintf(gid.buf, "%u", from_kgid(&init_user_ns, fl->gid)); + uid.len = sprintf(uid.buf, "%u", fl->uid); + gid.len = sprintf(gid.buf, "%u", fl->gid); /* data server entry: deviceid + efficiency + stateid + fh list + * user + group + flags + stats_collect_hint diff --git a/fs/nfsd/flexfilelayoutxdr.h b/fs/nfsd/flexfilelayoutxdr.h index 6d5a1066a903c1..3e1876d49db234 100644 --- a/fs/nfsd/flexfilelayoutxdr.h +++ b/fs/nfsd/flexfilelayoutxdr.h @@ -35,8 +35,9 @@ struct pnfs_ff_device_addr { struct pnfs_ff_layout { u32 flags; u32 stats_collect_hint; - kuid_t uid; - kgid_t gid; + /* Values to encode; nfsd4_ff_proc_layoutget() has mapped these */ + u32 uid; + u32 gid; struct nfsd4_deviceid deviceid; stateid_t stateid; struct nfs_fh fh; From 6defa44b4232ed2f4f168f338bf6c1a252d90e32 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 20 Jul 2026 10:14:40 -0400 Subject: [PATCH 308/857] NFS: Add linux/nfs_fh.h Plenty of spots around the kernel need the full definition of struct nfs_fh but not the cred, sunrpc, and uapi dependencies that linux/nfs.h pulls in along with it. Relocate struct nfs_fh to its own header, and include that header in linux/nfs.h so existing consumers keep building. Over time, consumers can then replace #include with #include While relocating the code, add kernel-doc comments for the FH operations and convert nfs_compare_fh() to return bool. Link: https://patch.msgid.link/20260720141442.783935-2-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/nfs.h | 39 ++------------------------ include/linux/nfs_fh.h | 63 ++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 65 insertions(+), 37 deletions(-) create mode 100644 include/linux/nfs_fh.h diff --git a/include/linux/nfs.h b/include/linux/nfs.h index 0906a0b40c6aa5..0e2a0b1e306156 100644 --- a/include/linux/nfs.h +++ b/include/linux/nfs.h @@ -11,8 +11,8 @@ #include #include #include -#include -#include +#include + #include /* The LOCALIO program is entirely private to Linux and is @@ -22,30 +22,6 @@ #define LOCALIOPROC_NULL 0 #define LOCALIOPROC_UUID_IS_LOCAL 1 -/* - * This is the kernel NFS client file handle representation - */ -#define NFS_MAXFHSIZE 128 -struct nfs_fh { - unsigned short size; - unsigned char data[NFS_MAXFHSIZE]; -}; - -/* - * Returns a zero iff the size and data fields match. - * Checks only "size" bytes in the data field. - */ -static inline int nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b) -{ - return a->size != b->size || memcmp(a->data, b->data, a->size) != 0; -} - -static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source) -{ - target->size = source->size; - memcpy(target->data, source->data, source->size); -} - enum nfs3_stable_how { NFS_UNSTABLE = 0, NFS_DATA_SYNC = 1, @@ -55,15 +31,4 @@ enum nfs3_stable_how { NFS_INVALID_STABLE_HOW = -1 }; -/** - * nfs_fhandle_hash - calculate the crc32 hash for the filehandle - * @fh - pointer to filehandle - * - * returns a crc32 hash for the filehandle that is compatible with - * the one displayed by "wireshark". - */ -static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh) -{ - return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size); -} #endif /* _LINUX_NFS_H */ diff --git a/include/linux/nfs_fh.h b/include/linux/nfs_fh.h new file mode 100644 index 00000000000000..49dfc5ec60fec8 --- /dev/null +++ b/include/linux/nfs_fh.h @@ -0,0 +1,63 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * struct nfs_fh is an NFS version-agnostic data structure that + * stores an NFS file handle. It is also commonly used in NFS + * related APIs. + */ +#ifndef _LINUX_NFS_FH_H +#define _LINUX_NFS_FH_H + +#include +#include +#include + +/* + * The largest file handle size today is an NFSv4 file handle, + * which can be up to 128 octets long. + */ +#define NFS_MAXFHSIZE 128 +struct nfs_fh { + unsigned short size; + unsigned char data[NFS_MAXFHSIZE]; +}; + +/** + * nfs_compare_fh - Compare two NFS file handles + * @a: An NFS file handle to be compared + * @b: An NFS file handle to be compared + * + * Checks only "size" bytes in each data field. + * + * Return: %false if the two file handles are equal, otherwise %true + */ +static inline bool nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b) +{ + return a->size != b->size || memcmp(a->data, b->data, a->size) != 0; +} + +/** + * nfs_copy_fh - Copy an NFS file handle + * @target: Destination file handle + * @source: Source file handle + * + * Copies source->size bytes of file handle data into target. + */ +static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source) +{ + target->size = source->size; + memcpy(target->data, source->data, source->size); +} + +/** + * nfs_fhandle_hash - Calculate the crc32 hash for the filehandle + * @fh: An NFS file handle to hash + * + * Return: a crc32 hash for the filehandle that is compatible with + * the one displayed by "wireshark" + */ +static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh) +{ + return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size); +} + +#endif /* _LINUX_NFS_FH_H */ From bb954b12e4ea799d91a63da4e8c186f8a67e9c4d Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 20 Jul 2026 10:14:41 -0400 Subject: [PATCH 309/857] lockd: Switch linux/nfs.h to linux/nfs_fh.h lockd references "struct nfs_fh" but none of the other definitions in linux/nfs.h. That header also pulls in cred.h, several sunrpc headers, and uapi/linux/nfs.h, none of which lockd needs. A new linux/nfs_fh.h provides "struct nfs_fh" and its helpers without the rest of that surface. Switch lockd's xdr.h to linux/nfs_fh.h, and drop the now-redundant linux/nfs.h includes from svc.c and trace.h. Link: https://patch.msgid.link/20260720141442.783935-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/lockd/svc.c | 1 - fs/lockd/trace.h | 1 - fs/lockd/xdr.h | 2 +- 3 files changed, 1 insertion(+), 3 deletions(-) diff --git a/fs/lockd/svc.c b/fs/lockd/svc.c index ee90e743064afb..f0e1a58c910666 100644 --- a/fs/lockd/svc.c +++ b/fs/lockd/svc.c @@ -36,7 +36,6 @@ #include #include #include -#include #include "lockd.h" #include "netns.h" diff --git a/fs/lockd/trace.h b/fs/lockd/trace.h index a11d04e8c835eb..1f79955ea0f5f6 100644 --- a/fs/lockd/trace.h +++ b/fs/lockd/trace.h @@ -7,7 +7,6 @@ #include #include -#include #include "lockd.h" diff --git a/fs/lockd/xdr.h b/fs/lockd/xdr.h index a1126cca98c6ce..56b9796aa39d96 100644 --- a/fs/lockd/xdr.h +++ b/fs/lockd/xdr.h @@ -10,7 +10,7 @@ #include #include -#include +#include #include #define SM_MAXSTRLEN 1024 From 4c4fb5d3868b3d5a30886e2a206700de0c845b80 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 20 Jul 2026 10:14:42 -0400 Subject: [PATCH 310/857] NFSD: Use struct knfsd_fh in struct pnfs_ff_layout The file handle held in struct pnfs_ff_layout is copied directly out of a struct svc_fh, whose fh_handle member is a struct knfsd_fh. Storing the layout's copy as struct nfs_fh instead forced an open-coded field-by-field copy between two unrelated structures. Hold the layout's file handle in struct knfsd_fh so the copy uses the canonical fh_copy_shallow() helper and server code no longer reaches into a separate file handle representation. Cc: Thomas Haynes Link: https://patch.msgid.link/20260720141442.783935-4-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/flexfilelayout.c | 3 +-- fs/nfsd/flexfilelayoutxdr.c | 4 ++-- fs/nfsd/flexfilelayoutxdr.h | 3 ++- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/fs/nfsd/flexfilelayout.c b/fs/nfsd/flexfilelayout.c index c6e6b9106a8393..0deb913493a398 100644 --- a/fs/nfsd/flexfilelayout.c +++ b/fs/nfsd/flexfilelayout.c @@ -66,8 +66,7 @@ nfsd4_ff_proc_layoutget(struct svc_rqst *rqstp, struct inode *inode, if (error) goto out_error; - fl->fh.size = fhp->fh_handle.fh_size; - memcpy(fl->fh.data, &fhp->fh_handle.fh_raw, fl->fh.size); + fh_copy_shallow(&fl->fh, &fhp->fh_handle); /* Give whole file layout segments */ seg->offset = 0; diff --git a/fs/nfsd/flexfilelayoutxdr.c b/fs/nfsd/flexfilelayoutxdr.c index c12bb7c371b0e5..e297100a2ac3ce 100644 --- a/fs/nfsd/flexfilelayoutxdr.c +++ b/fs/nfsd/flexfilelayoutxdr.c @@ -31,7 +31,7 @@ nfsd4_ff_encode_layoutget(struct xdr_stream *xdr, struct ff_idmap uid; struct ff_idmap gid; - fh_len = 4 + xdr_align_size(fl->fh.size); + fh_len = 4 + xdr_align_size(fl->fh.fh_size); uid.len = sprintf(uid.buf, "%u", fl->uid); gid.len = sprintf(gid.buf, "%u", fl->gid); @@ -69,7 +69,7 @@ nfsd4_ff_encode_layoutget(struct xdr_stream *xdr, sizeof(stateid_opaque_t)); *p++ = cpu_to_be32(1); /* single file handle */ - p = xdr_encode_opaque(p, fl->fh.data, fl->fh.size); + p = xdr_encode_opaque(p, fl->fh.fh_raw, fl->fh.fh_size); p = xdr_encode_opaque(p, uid.buf, uid.len); p = xdr_encode_opaque(p, gid.buf, gid.len); diff --git a/fs/nfsd/flexfilelayoutxdr.h b/fs/nfsd/flexfilelayoutxdr.h index 3e1876d49db234..f7d1dd0708ec46 100644 --- a/fs/nfsd/flexfilelayoutxdr.h +++ b/fs/nfsd/flexfilelayoutxdr.h @@ -6,6 +6,7 @@ #define _NFSD_FLEXFILELAYOUTXDR_H 1 #include +#include "nfsfh.h" #include "xdr4.h" #define FF_FLAGS_NO_LAYOUTCOMMIT 1 @@ -40,7 +41,7 @@ struct pnfs_ff_layout { u32 gid; struct nfsd4_deviceid deviceid; stateid_t stateid; - struct nfs_fh fh; + struct knfsd_fh fh; }; __be32 nfsd4_ff_encode_getdeviceinfo(struct xdr_stream *xdr, From 3c5ea5e2b5e87b631ab47b63c788af8880f51329 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 21 Jul 2026 12:23:03 -0400 Subject: [PATCH 311/857] nfs_common: Remove unused nfs_ssc_client_ops infrastructure Clean up: Commit 75333d48f922 ("NFSD: fix use-after-free in __nfs42_ssc_open()") addressed a use-after-free bug by removing the nfsd4_interssc_disconnect() function. Post-copy clean-up was then delegated to NFSD's laundromat. Since that commit, the nfs_do_sb_deactive() wrapper function and the entire nfs_ssc_client_ops infrastructure no longer have any consumers. This includes nfs_do_sb_deactive(), struct nfs_ssc_client_ops, nfs_ssc_register(), nfs_ssc_unregister(), and related registrations in the NFS client. Cc: Olga Kornievskaia Cc: Dai Ngo Link: https://patch.msgid.link/20260721162306.894558-2-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfs/super.c | 25 ------------------------ fs/nfs_common/nfs_ssc.c | 42 ----------------------------------------- fs/nfsd/nfs4proc.c | 2 -- include/linux/nfs_ssc.h | 20 -------------------- 4 files changed, 89 deletions(-) diff --git a/fs/nfs/super.c b/fs/nfs/super.c index cb19f1540d9841..23292680adbd5d 100644 --- a/fs/nfs/super.c +++ b/fs/nfs/super.c @@ -58,7 +58,6 @@ #include #include -#include #include @@ -92,12 +91,6 @@ const struct super_operations nfs_sops = { }; EXPORT_SYMBOL_GPL(nfs_sops); -#ifdef CONFIG_NFS_V4_2 -static const struct nfs_ssc_client_ops nfs_ssc_clnt_ops_tbl = { - .sco_sb_deactive = nfs_sb_deactive, -}; -#endif - #if IS_ENABLED(CONFIG_NFS_V4) static int __init register_nfs4_fs(void) { @@ -119,18 +112,6 @@ static void unregister_nfs4_fs(void) } #endif -#ifdef CONFIG_NFS_V4_2 -static void nfs_ssc_register_ops(void) -{ - nfs_ssc_register(&nfs_ssc_clnt_ops_tbl); -} - -static void nfs_ssc_unregister_ops(void) -{ - nfs_ssc_unregister(&nfs_ssc_clnt_ops_tbl); -} -#endif /* CONFIG_NFS_V4_2 */ - static struct shrinker *acl_shrinker; /* @@ -163,9 +144,6 @@ int __init register_nfs_fs(void) shrinker_register(acl_shrinker); -#ifdef CONFIG_NFS_V4_2 - nfs_ssc_register_ops(); -#endif return 0; error_3: nfs_unregister_sysctl(); @@ -185,9 +163,6 @@ void __exit unregister_nfs_fs(void) shrinker_free(acl_shrinker); nfs_unregister_sysctl(); unregister_nfs4_fs(); -#ifdef CONFIG_NFS_V4_2 - nfs_ssc_unregister_ops(); -#endif unregister_filesystem(&nfs_fs_type); } diff --git a/fs/nfs_common/nfs_ssc.c b/fs/nfs_common/nfs_ssc.c index 832246b22c5175..8d8b7344ab4fe5 100644 --- a/fs/nfs_common/nfs_ssc.c +++ b/fs/nfs_common/nfs_ssc.c @@ -47,45 +47,3 @@ void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops) } EXPORT_SYMBOL_GPL(nfs42_ssc_unregister); #endif /* CONFIG_NFS_V4_2 */ - -#ifdef CONFIG_NFS_V4_2 -/** - * nfs_ssc_register - install the NFS_FS client ops in the nfs_ssc_client_tbl - * @ops: NFS_FS ops to be installed - * - * Return values: - * None - */ -void nfs_ssc_register(const struct nfs_ssc_client_ops *ops) -{ - nfs_ssc_client_tbl.ssc_nfs_ops = ops; -} -EXPORT_SYMBOL_GPL(nfs_ssc_register); - -/** - * nfs_ssc_unregister - uninstall the NFS_FS client ops from - * the nfs_ssc_client_tbl - * @ops: ops to be uninstalled - * - * Return values: - * None - */ -void nfs_ssc_unregister(const struct nfs_ssc_client_ops *ops) -{ - if (nfs_ssc_client_tbl.ssc_nfs_ops != ops) - return; - nfs_ssc_client_tbl.ssc_nfs_ops = NULL; -} -EXPORT_SYMBOL_GPL(nfs_ssc_unregister); - -#else -void nfs_ssc_register(const struct nfs_ssc_client_ops *ops) -{ -} -EXPORT_SYMBOL_GPL(nfs_ssc_register); - -void nfs_ssc_unregister(const struct nfs_ssc_client_ops *ops) -{ -} -EXPORT_SYMBOL_GPL(nfs_ssc_unregister); -#endif /* CONFIG_NFS_V4_2 */ diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index d4265ee3c73dee..8a2742d248c00f 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -1698,8 +1698,6 @@ extern struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, nfs4_stateid *stateid); extern void nfs42_ssc_close(struct file *filep); -extern void nfs_sb_deactive(struct super_block *sb); - #define NFSD42_INTERSSC_MOUNTOPS "vers=4.2,addr=%s,sec=sys" /* diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index 22265b1ff08005..ba236dba8975cb 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -21,16 +21,8 @@ struct nfs4_ssc_client_ops { void (*sco_close)(struct file *filep); }; -/* - * NFS_FS - */ -struct nfs_ssc_client_ops { - void (*sco_sb_deactive)(struct super_block *sb); -}; - struct nfs_ssc_client_ops_tbl { const struct nfs4_ssc_client_ops *ssc_nfs4_ops; - const struct nfs_ssc_client_ops *ssc_nfs_ops; }; extern void nfs42_ssc_register_ops(void); @@ -67,15 +59,3 @@ struct nfsd4_ssc_umount_item { struct vfsmount *nsui_vfsmount; char nsui_ipaddr[RPC_MAX_ADDRBUFLEN + 1]; }; - -/* - * NFS_FS - */ -extern void nfs_ssc_register(const struct nfs_ssc_client_ops *ops); -extern void nfs_ssc_unregister(const struct nfs_ssc_client_ops *ops); - -static inline void nfs_do_sb_deactive(struct super_block *sb) -{ - if (nfs_ssc_client_tbl.ssc_nfs_ops) - (*nfs_ssc_client_tbl.ssc_nfs_ops->sco_sb_deactive)(sb); -} From 4813f79bf091af02e6b68fbe25d2266fb549a3f7 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 21 Jul 2026 12:23:04 -0400 Subject: [PATCH 312/857] NFSD: Hoist nfs42_ssc_open() into fs/nfs_common/nfs_ssc.c Refactor: The infrastructure and details for calling the client's ssc_open method can be hidden in nfs_ssc.c. This reduces the SSC footprint in fs/nfsd/nfs4proc.c, a step toward removing that file's dependency on , which indirectly includes . The open and close functions are named "nfsd42_" since they are meant to be invoked only by NFSD. Cc: Olga Kornievskaia Cc: Dai Ngo Link: https://patch.msgid.link/20260721162306.894558-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfs_common/nfs_ssc.c | 56 +++++++++++++++++++++++++++++++++++++++-- fs/nfsd/nfs4proc.c | 17 +++---------- include/linux/nfs_ssc.h | 23 +++++++---------- 3 files changed, 66 insertions(+), 30 deletions(-) diff --git a/fs/nfs_common/nfs_ssc.c b/fs/nfs_common/nfs_ssc.c index 8d8b7344ab4fe5..a8e79ec687018a 100644 --- a/fs/nfs_common/nfs_ssc.c +++ b/fs/nfs_common/nfs_ssc.c @@ -12,9 +12,61 @@ #include #include "../nfs/nfs4_fs.h" +struct nfs_ssc_client_ops_tbl { + const struct nfs4_ssc_client_ops *ssc_nfs4_ops; +}; -struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl; -EXPORT_SYMBOL_GPL(nfs_ssc_client_tbl); +static struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl __read_mostly; + +/** + * nfsd42_ssc_open - Open a file to be used for server-to-server copy + * @ss_mnt: active mount point on which the source file resides + * @src_fh: file handle of the source file to be copied + * @stateid: stateid to use for COPY operation + * + * Caller must close the returned file using nfsd42_ssc_close(). + * + * Return: an open file, or an ERR_PTR on error + */ +struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, + nfs4_stateid *stateid) +{ + /* + * Built under CONFIG_NFS_V4_2_SSC_HELPER, which the NFS client + * enables on its own. The dispatch below is live only when the + * server also sets CONFIG_NFSD_V4_2_INTER_SSC; without it the + * source file cannot be opened, so callers get -EIO. + */ +#if IS_ENABLED(CONFIG_NFSD_V4_2_INTER_SSC) + const struct nfs4_ssc_client_ops *ops = nfs_ssc_client_tbl.ssc_nfs4_ops; + + if (ops) + return ops->sco_open(ss_mnt, src_fh, stateid); +#endif + + return ERR_PTR(-EIO); +} +EXPORT_SYMBOL_GPL(nfsd42_ssc_open); + +/** + * nfsd42_ssc_close - Close a file opened with nfsd42_ssc_open() + * @filp: struct file to be closed + * + * The real cleanup happens unconditionally in nfsd4_cleanup_inter_ssc(). + * The vfsmount is pinned until this function is called, preventing + * the client from unregistering its SSC ops. + */ +void nfsd42_ssc_close(struct file *filp) +{ + /* Live only under CONFIG_NFSD_V4_2_INTER_SSC; see nfsd42_ssc_open(). */ +#if IS_ENABLED(CONFIG_NFSD_V4_2_INTER_SSC) + const struct nfs4_ssc_client_ops *ops = nfs_ssc_client_tbl.ssc_nfs4_ops; + + if (ops) + ops->sco_close(filp); +#endif +} +EXPORT_SYMBOL_GPL(nfsd42_ssc_close); #ifdef CONFIG_NFS_V4_2 /** diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 8a2742d248c00f..aacfa80bddd864 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -1693,11 +1693,6 @@ void nfsd4_cancel_copy_by_sb(struct net *net, struct super_block *sb) #ifdef CONFIG_NFSD_V4_2_INTER_SSC -extern struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, - nfs4_stateid *stateid); -extern void nfs42_ssc_close(struct file *filep); - #define NFSD42_INTERSSC_MOUNTOPS "vers=4.2,addr=%s,sec=sys" /* @@ -1920,7 +1915,7 @@ nfsd4_cleanup_inter_ssc(struct nfsd4_ssc_umount_item *nsui, struct file *filp, struct nfsd_net *nn = net_generic(dst->nf_net, nfsd_net_id); long timeout = msecs_to_jiffies(nfsd4_ssc_umount_timeout); - nfs42_ssc_close(filp); + nfsd42_ssc_close(filp); fput(filp); spin_lock(&nn->nfsd_ssc_lock); @@ -1952,12 +1947,6 @@ nfsd4_cleanup_inter_ssc(struct nfsd4_ssc_umount_item *nsui, struct file *filp, { } -static struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, - nfs4_stateid *stateid) -{ - return NULL; -} #endif /* CONFIG_NFSD_V4_2_INTER_SSC */ static __be32 @@ -2180,8 +2169,8 @@ static int nfsd4_do_async_copy(void *data) if (nfsd4_ssc_is_inter(copy)) { struct file *filp; - filp = nfs42_ssc_open(copy->ss_nsui->nsui_vfsmount, - ©->c_fh, ©->stateid); + filp = nfsd42_ssc_open(copy->ss_nsui->nsui_vfsmount, + ©->c_fh, ©->stateid); if (IS_ERR(filp)) { switch (PTR_ERR(filp)) { case -EBADF: diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index ba236dba8975cb..fc0d5d48dec2a7 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -10,8 +10,6 @@ #include #include -extern struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl; - /* * NFS_V4 */ @@ -21,29 +19,26 @@ struct nfs4_ssc_client_ops { void (*sco_close)(struct file *filep); }; -struct nfs_ssc_client_ops_tbl { - const struct nfs4_ssc_client_ops *ssc_nfs4_ops; -}; - extern void nfs42_ssc_register_ops(void); extern void nfs42_ssc_unregister_ops(void); extern void nfs42_ssc_register(const struct nfs4_ssc_client_ops *ops); extern void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops); -#ifdef CONFIG_NFSD_V4_2_INTER_SSC -static inline struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, nfs4_stateid *stateid) +#if IS_ENABLED(CONFIG_NFS_V4_2_SSC_HELPER) +struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, + nfs4_stateid *stateid); +void nfsd42_ssc_close(struct file *filp); +#else +static inline struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, + struct nfs_fh *src_fh, + nfs4_stateid *stateid) { - if (nfs_ssc_client_tbl.ssc_nfs4_ops) - return (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_open)(ss_mnt, src_fh, stateid); return ERR_PTR(-EIO); } -static inline void nfs42_ssc_close(struct file *filep) +static inline void nfsd42_ssc_close(struct file *filp) { - if (nfs_ssc_client_tbl.ssc_nfs4_ops) - (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_close)(filep); } #endif From 171b43532b86d09817a9100a26ce540f7c91d4ab Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 21 Jul 2026 12:23:05 -0400 Subject: [PATCH 313/857] nfs_common: Synchronize access to the SSC client ops table nfsd42_ssc_open() and nfsd42_ssc_close() load ssc_nfs4_ops without synchronization while nfs42_ssc_register() and nfs42_ssc_unregister() store to it. Those reads are safe today only through a non-obvious invariant: an inter-server copy holds an active vers=4.2 mount of the source across both calls, the mount pins the nfsv4 module through the nfs_client's cl_nfs_mod reference, and unregister runs only at nfsv4 module exit, so it cannot run while a call is in flight. Replace that implicit contract with synchronization local to the broker, so its safety no longer rests on a caller in another subsystem. Read the pointer under RCU so a reader observes it atomically as a valid table or NULL. nfs42_ssc_unregister() stores NULL and then calls synchronize_rcu(), so it cannot return while a reader still holds the pointer. The two readers need different handling because one sleeps and the other does not. sco_close() does not sleep, so nfsd42_ssc_close() runs it to completion inside the RCU read-side section and the synchronize_rcu() in unregister waits for it. __nfs42_ssc_open() does sleep -- it issues a GETATTR RPC to the source server and allocates with GFP_KERNEL -- so it must not run inside an RCU read-side section. Pin the provider module with try_module_get() while still under rcu_read_lock(), drop the lock, invoke the open, then release the module. The reference keeps the provider mapped across the sleep without relying on the caller's mount. If the table has already been torn down the copy gets -EIO. Cc: Olga Kornievskaia Cc: Dai Ngo Link: https://patch.msgid.link/20260721162306.894558-4-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfs/nfs4file.c | 1 + fs/nfs_common/nfs_ssc.c | 40 ++++++++++++++++++++++++++++++---------- include/linux/nfs_ssc.h | 1 + 3 files changed, 32 insertions(+), 10 deletions(-) diff --git a/fs/nfs/nfs4file.c b/fs/nfs/nfs4file.c index 6401f6363f7534..9a434f5dda8d36 100644 --- a/fs/nfs/nfs4file.c +++ b/fs/nfs/nfs4file.c @@ -402,6 +402,7 @@ static void __nfs42_ssc_close(struct file *filep) } static const struct nfs4_ssc_client_ops nfs4_ssc_clnt_ops_tbl = { + .owner = THIS_MODULE, .sco_open = __nfs42_ssc_open, .sco_close = __nfs42_ssc_close, }; diff --git a/fs/nfs_common/nfs_ssc.c b/fs/nfs_common/nfs_ssc.c index a8e79ec687018a..ef158008b80330 100644 --- a/fs/nfs_common/nfs_ssc.c +++ b/fs/nfs_common/nfs_ssc.c @@ -13,7 +13,7 @@ #include "../nfs/nfs4_fs.h" struct nfs_ssc_client_ops_tbl { - const struct nfs4_ssc_client_ops *ssc_nfs4_ops; + const struct nfs4_ssc_client_ops __rcu *ssc_nfs4_ops; }; static struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl __read_mostly; @@ -38,10 +38,24 @@ struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, * source file cannot be opened, so callers get -EIO. */ #if IS_ENABLED(CONFIG_NFSD_V4_2_INTER_SSC) - const struct nfs4_ssc_client_ops *ops = nfs_ssc_client_tbl.ssc_nfs4_ops; + const struct nfs4_ssc_client_ops *ops; + struct file *res; - if (ops) - return ops->sco_open(ss_mnt, src_fh, stateid); + /* + * sco_open() sleeps and must not run inside an RCU read-side + * section. Pin the provider module so the open runs with the + * module held; try_module_get() fails once unregister begins, + * and the copy then gets -EIO. + */ + rcu_read_lock(); + ops = rcu_dereference(nfs_ssc_client_tbl.ssc_nfs4_ops); + if (ops && try_module_get(ops->owner)) { + rcu_read_unlock(); + res = ops->sco_open(ss_mnt, src_fh, stateid); + module_put(ops->owner); + return res; + } + rcu_read_unlock(); #endif return ERR_PTR(-EIO); @@ -53,17 +67,21 @@ EXPORT_SYMBOL_GPL(nfsd42_ssc_open); * @filp: struct file to be closed * * The real cleanup happens unconditionally in nfsd4_cleanup_inter_ssc(). - * The vfsmount is pinned until this function is called, preventing - * the client from unregistering its SSC ops. + * The client ops table is read under RCU; nfs42_ssc_unregister() calls + * synchronize_rcu() so unregistration cannot complete while a close is + * in flight. */ void nfsd42_ssc_close(struct file *filp) { /* Live only under CONFIG_NFSD_V4_2_INTER_SSC; see nfsd42_ssc_open(). */ #if IS_ENABLED(CONFIG_NFSD_V4_2_INTER_SSC) - const struct nfs4_ssc_client_ops *ops = nfs_ssc_client_tbl.ssc_nfs4_ops; + const struct nfs4_ssc_client_ops *ops; + rcu_read_lock(); + ops = rcu_dereference(nfs_ssc_client_tbl.ssc_nfs4_ops); if (ops) ops->sco_close(filp); + rcu_read_unlock(); #endif } EXPORT_SYMBOL_GPL(nfsd42_ssc_close); @@ -78,7 +96,7 @@ EXPORT_SYMBOL_GPL(nfsd42_ssc_close); */ void nfs42_ssc_register(const struct nfs4_ssc_client_ops *ops) { - nfs_ssc_client_tbl.ssc_nfs4_ops = ops; + rcu_assign_pointer(nfs_ssc_client_tbl.ssc_nfs4_ops, ops); } EXPORT_SYMBOL_GPL(nfs42_ssc_register); @@ -92,10 +110,12 @@ EXPORT_SYMBOL_GPL(nfs42_ssc_register); */ void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops) { - if (nfs_ssc_client_tbl.ssc_nfs4_ops != ops) + if (rcu_dereference_protected(nfs_ssc_client_tbl.ssc_nfs4_ops, + true) != ops) return; - nfs_ssc_client_tbl.ssc_nfs4_ops = NULL; + rcu_assign_pointer(nfs_ssc_client_tbl.ssc_nfs4_ops, NULL); + synchronize_rcu(); } EXPORT_SYMBOL_GPL(nfs42_ssc_unregister); #endif /* CONFIG_NFS_V4_2 */ diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index fc0d5d48dec2a7..b392d56a4dc252 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -14,6 +14,7 @@ * NFS_V4 */ struct nfs4_ssc_client_ops { + struct module *owner; struct file *(*sco_open)(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, nfs4_stateid *stateid); void (*sco_close)(struct file *filep); From 30373444a8131cc139b935ce5eb9b0b941ed481d Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 21 Jul 2026 12:23:06 -0400 Subject: [PATCH 314/857] NFSD: Split linux/nfs_ssc.h The nfs_ssc.h header contains both client- and server-side data structures, which means each of those implementations has to pull in headers from the other. Create a linux/nfsd_ssc.h for the server side APIs which no longer includes uapi/linux/nfs.h either directly or indirectly. Because nfsd_ssc.h drops the transitive include of the NFS client headers, fs/nfsd/nfs4proc.c now includes directly for filemap_check_wb_err(). struct nfsd4_ssc_umount_item is private to nfsd. Move it into fs/nfsd/xdr4.h alongside its only consumers rather than into the exported nfsd_ssc.h. As an added clean-up, add missing header guard macros and the struct file and struct vfsmount forward declarations the server prototypes need. Cc: Olga Kornievskaia Cc: Dai Ngo Link: https://patch.msgid.link/20260721162306.894558-5-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfs_common/nfs_ssc.c | 2 +- fs/nfsd/nfs4proc.c | 3 ++- fs/nfsd/nfs4state.c | 2 +- fs/nfsd/xdr4.h | 13 ++++++++++++ include/linux/nfs_ssc.h | 45 ++++++++++------------------------------ include/linux/nfsd_ssc.h | 38 +++++++++++++++++++++++++++++++++ 6 files changed, 66 insertions(+), 37 deletions(-) create mode 100644 include/linux/nfsd_ssc.h diff --git a/fs/nfs_common/nfs_ssc.c b/fs/nfs_common/nfs_ssc.c index ef158008b80330..e521e3c836fe7b 100644 --- a/fs/nfs_common/nfs_ssc.c +++ b/fs/nfs_common/nfs_ssc.c @@ -10,7 +10,7 @@ #include #include #include -#include "../nfs/nfs4_fs.h" +#include struct nfs_ssc_client_ops_tbl { const struct nfs4_ssc_client_ops __rcu *ssc_nfs4_ops; diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index aacfa80bddd864..8ffabe7a480dd7 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -38,9 +38,10 @@ #include #include #include +#include #include -#include +#include #include "attr4.h" #include "idmap.h" diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index e78d4fe5fbfb59..d7731bb4959c7a 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -45,7 +45,7 @@ #include #include #include -#include +#include #include "xdr4.h" #include "nfs4ctl.h" diff --git a/fs/nfsd/xdr4.h b/fs/nfsd/xdr4.h index e833407859c8d8..7bbb375874ef80 100644 --- a/fs/nfsd/xdr4.h +++ b/fs/nfsd/xdr4.h @@ -597,6 +597,19 @@ struct nfsd4_cb_offload { u32 co_referring_seqno; }; +struct nfsd4_ssc_umount_item { + struct list_head nsui_list; + bool nsui_busy; + /* + * nsui_refcnt inited to 2, 1 on list and 1 for consumer. Entry + * is removed when refcnt drops to 1 and nsui_expire expires. + */ + refcount_t nsui_refcnt; + unsigned long nsui_expire; + struct vfsmount *nsui_vfsmount; + char nsui_ipaddr[RPC_MAX_ADDRBUFLEN + 1]; +}; + struct nfsd4_copy { /* request */ stateid_t cp_src_stateid; diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index b392d56a4dc252..c199ea23e7eb77 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -2,17 +2,22 @@ /* * include/linux/nfs_ssc.h * + * NFSv4.2 server-to-server copy, NFS client side APIs + * * Author: Dai Ngo * * Copyright (c) 2020, Oracle and/or its affiliates. */ -#include -#include +#ifndef _LINUX_NFS_SSC_H +#define _LINUX_NFS_SSC_H + +#include +#include + +struct file; +struct vfsmount; -/* - * NFS_V4 - */ struct nfs4_ssc_client_ops { struct module *owner; struct file *(*sco_open)(struct vfsmount *ss_mnt, @@ -26,32 +31,4 @@ extern void nfs42_ssc_unregister_ops(void); extern void nfs42_ssc_register(const struct nfs4_ssc_client_ops *ops); extern void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops); -#if IS_ENABLED(CONFIG_NFS_V4_2_SSC_HELPER) -struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, - nfs4_stateid *stateid); -void nfsd42_ssc_close(struct file *filp); -#else -static inline struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, - nfs4_stateid *stateid) -{ - return ERR_PTR(-EIO); -} - -static inline void nfsd42_ssc_close(struct file *filp) -{ -} -#endif - -struct nfsd4_ssc_umount_item { - struct list_head nsui_list; - bool nsui_busy; - /* - * nsui_refcnt inited to 2, 1 on list and 1 for consumer. Entry - * is removed when refcnt drops to 1 and nsui_expire expires. - */ - refcount_t nsui_refcnt; - unsigned long nsui_expire; - struct vfsmount *nsui_vfsmount; - char nsui_ipaddr[RPC_MAX_ADDRBUFLEN + 1]; -}; +#endif /* _LINUX_NFS_SSC_H */ diff --git a/include/linux/nfsd_ssc.h b/include/linux/nfsd_ssc.h new file mode 100644 index 00000000000000..7001410f01c299 --- /dev/null +++ b/include/linux/nfsd_ssc.h @@ -0,0 +1,38 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * include/linux/nfsd_ssc.h + * + * NFSv4.2 server-to-server copy, NFS server side APIs + * + * Author: Dai Ngo + * + * Copyright (c) 2020, Oracle and/or its affiliates. + */ + +#ifndef _LINUX_NFSD_SSC_H +#define _LINUX_NFSD_SSC_H + +#include +#include + +struct file; +struct vfsmount; + +#if IS_ENABLED(CONFIG_NFS_V4_2_SSC_HELPER) +struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, + nfs4_stateid *stateid); +void nfsd42_ssc_close(struct file *filp); +#else +static inline struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, + struct nfs_fh *src_fh, + nfs4_stateid *stateid) +{ + return ERR_PTR(-EIO); +} + +static inline void nfsd42_ssc_close(struct file *filp) +{ +} +#endif + +#endif /* _LINUX_NFSD_SSC_H */ From 68a22862402e56d60d7b9afa327e940690af82dc Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Thu, 23 Jul 2026 14:20:42 -0400 Subject: [PATCH 315/857] NFS: Move definition of enum nfs3_stable_how Clean up: enum nfs3_stable_how was introduced in NFSv3. NFSv2 has no stable_how on the wire; its write path passes NFS_FILE_SYNC only as a placeholder that the protocol ignores. The stable_how constants describe an NFSv3 wire value, so they belong in linux/nfs3.h. Link: https://patch.msgid.link/20260723182043.990391-2-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/nfs.h | 9 --------- include/linux/nfs3.h | 8 ++++++++ include/trace/misc/nfs.h | 1 + 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/include/linux/nfs.h b/include/linux/nfs.h index 0e2a0b1e306156..0e2b210c103b69 100644 --- a/include/linux/nfs.h +++ b/include/linux/nfs.h @@ -22,13 +22,4 @@ #define LOCALIOPROC_NULL 0 #define LOCALIOPROC_UUID_IS_LOCAL 1 -enum nfs3_stable_how { - NFS_UNSTABLE = 0, - NFS_DATA_SYNC = 1, - NFS_FILE_SYNC = 2, - - /* used by direct.c to mark verf as invalid */ - NFS_INVALID_STABLE_HOW = -1 -}; - #endif /* _LINUX_NFS_H */ diff --git a/include/linux/nfs3.h b/include/linux/nfs3.h index 404b8f724fc956..1d18da0860d52b 100644 --- a/include/linux/nfs3.h +++ b/include/linux/nfs3.h @@ -7,6 +7,14 @@ #include +enum nfs3_stable_how { + NFS_UNSTABLE = 0, + NFS_DATA_SYNC = 1, + NFS_FILE_SYNC = 2, + + /* used to mark verf as invalid */ + NFS_INVALID_STABLE_HOW = -1 +}; /* Number of 32bit words in post_op_attr */ #define NFS3_POST_OP_ATTR_WORDS 22 diff --git a/include/trace/misc/nfs.h b/include/trace/misc/nfs.h index a394b4d38e18fa..b5fb77d7954b35 100644 --- a/include/trace/misc/nfs.h +++ b/include/trace/misc/nfs.h @@ -8,6 +8,7 @@ */ #include +#include #include #include From 0a3cb78ff4ad059709e020e7b1092969850936a6 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Thu, 23 Jul 2026 14:20:43 -0400 Subject: [PATCH 316/857] NFSD: Replace nfsd_write()'s "stable" argument with "iocb_flags" The current nfsd_write() API is not NFS version-agnostic, as it relies on callers to pass an NFSv3 stable_how value to determine the persistence of the requested WRITE. NFSv2 does not use a stable-how value on the wire, and NFSv4 has its own stable_how4 (though stable_how and stable_how4 happen to share the same numeric values). To remove the dependence on NFSv3-specific XDR values from NFSD's generic VFS APIs, replace nfsd_write()'s stable argument with an argument that passes a set of IOCB flags instead of an XDR-defined value. The NFSv4 WRITE and COPY paths had been borrowing the NFSv3 stable_how constants for their own on-the-wire stable values, relying on the numeric coincidence noted above. Convert those sites to the stable_how4 enumerators so the v4 code expresses its own protocol's values directly, with no change in behavior. While here, bound-check the decoded NFSv3 WRITE stable value, as the NFSv4 WRITE decoder already does, and make the nfsd3_writeargs stable field unsigned to suit. The larger benefit is one less NFSv4 dependency on nfs3.h. Link: https://patch.msgid.link/20260723182043.990391-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs3proc.c | 17 ++++++++++++++++- fs/nfsd/nfs3xdr.c | 2 ++ fs/nfsd/nfs4proc.c | 18 ++++++++++++++++-- fs/nfsd/nfs4xdr.c | 2 +- fs/nfsd/nfsproc.c | 2 +- fs/nfsd/vfs.c | 30 ++++++++++-------------------- fs/nfsd/vfs.h | 6 ++++-- fs/nfsd/xdr3.h | 2 +- include/linux/nfs4.h | 6 ++++++ 9 files changed, 57 insertions(+), 28 deletions(-) diff --git a/fs/nfsd/nfs3proc.c b/fs/nfsd/nfs3proc.c index 4b3075c05b9793..19ab0a713d8212 100644 --- a/fs/nfsd/nfs3proc.c +++ b/fs/nfsd/nfs3proc.c @@ -49,6 +49,20 @@ static bool nfsd3_time_in_range(const struct iattr *iap) return true; } +static int nfsd3_iocb_flags(enum nfs3_stable_how how) +{ + switch (how) { + case NFS_FILE_SYNC: + /* persist data and timestamps */ + return IOCB_DSYNC | IOCB_SYNC; + case NFS_DATA_SYNC: + /* persist data only */ + return IOCB_DSYNC; + default: + return 0; + } +} + static __be32 nfsd3_map_status(__be32 status) { switch (status) { @@ -261,7 +275,8 @@ nfsd3_proc_write(struct svc_rqst *rqstp) resp->committed = argp->stable; resp->status = nfsd_write(rqstp, &resp->fh, argp->offset, &argp->payload, &cnt, - resp->committed, resp->verf); + nfsd3_iocb_flags(resp->committed), + resp->verf); resp->count = cnt; resp->status = nfsd3_map_status(resp->status); return rpc_success; diff --git a/fs/nfsd/nfs3xdr.c b/fs/nfsd/nfs3xdr.c index 196bcc6edebb98..090cea8e545dc6 100644 --- a/fs/nfsd/nfs3xdr.c +++ b/fs/nfsd/nfs3xdr.c @@ -557,6 +557,8 @@ nfs3svc_decode_writeargs(struct svc_rqst *rqstp, struct xdr_stream *xdr) return false; if (xdr_stream_decode_u32(xdr, &args->stable) < 0) return false; + if (args->stable > NFS_FILE_SYNC) + return false; /* opaque data */ if (xdr_stream_decode_u32(xdr, &args->len) < 0) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 8ffabe7a480dd7..33dc92d48ce0f7 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -72,6 +72,20 @@ MODULE_PARM_DESC(nfsd4_ssc_umount_timeout, #define NFSDDBG_FACILITY NFSDDBG_PROC +static int nfsd4_iocb_flags(enum stable_how4 how) +{ + switch (how) { + case FILE_SYNC4: + /* persist data and timestamps */ + return IOCB_DSYNC | IOCB_SYNC; + case DATA_SYNC4: + /* persist data only */ + return IOCB_DSYNC; + default: + return 0; + } +} + static u32 nfsd_attrmask[] = { NFSD_WRITEABLE_ATTRS_WORD0, NFSD_WRITEABLE_ATTRS_WORD1, @@ -1418,7 +1432,7 @@ nfsd4_write(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, write->wr_how_written = write->wr_stable_how; status = nfsd_vfs_write(rqstp, &cstate->current_fh, nf, write->wr_offset, &write->wr_payload, - &cnt, write->wr_how_written, + &cnt, nfsd4_iocb_flags(write->wr_how_written), (__be32 *)write->wr_verifier.data); nfsd_file_put(nf); @@ -2002,7 +2016,7 @@ static void nfsd4_init_copy_res(struct nfsd4_copy *copy, bool sync) { copy->cp_res.wr_stable_how = test_bit(NFSD4_COPY_F_COMMITTED, ©->cp_flags) ? - NFS_FILE_SYNC : NFS_UNSTABLE; + FILE_SYNC4 : UNSTABLE4; nfsd4_copy_set_sync(copy, sync); } diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 7ccc7c897b00cb..a47eb544b99f6f 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -1607,7 +1607,7 @@ nfsd4_decode_write(struct nfsd4_compoundargs *argp, union nfsd4_op_u *u) return nfserr_bad_xdr; if (xdr_stream_decode_u32(argp->xdr, &write->wr_stable_how) < 0) return nfserr_bad_xdr; - if (write->wr_stable_how > NFS_FILE_SYNC) + if (write->wr_stable_how > FILE_SYNC4) return nfserr_bad_xdr; if (xdr_stream_decode_u32(argp->xdr, &write->wr_buflen) < 0) return nfserr_bad_xdr; diff --git a/fs/nfsd/nfsproc.c b/fs/nfsd/nfsproc.c index 919acfba356a4f..48541ef7644f4d 100644 --- a/fs/nfsd/nfsproc.c +++ b/fs/nfsd/nfsproc.c @@ -266,7 +266,7 @@ nfsd_proc_write(struct svc_rqst *rqstp) fh_copy(&resp->fh, &argp->fh); resp->status = nfsd_write(rqstp, &resp->fh, argp->offset, - &argp->payload, &cnt, NFS_DATA_SYNC, NULL); + &argp->payload, &cnt, IOCB_DSYNC, NULL); if (resp->status == nfs_ok) resp->status = fh_getattr(&resp->fh, &resp->stat); else if (resp->status == nfserr_jukebox) diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index cbd1e34a548a83..807e09521e0c3e 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -1433,7 +1433,7 @@ nfsd_direct_write(struct svc_rqst *rqstp, struct svc_fh *fhp, * @offset: Byte offset of start * @payload: xdr_buf containing the write payload * @cnt: IN: number of bytes to write, OUT: number of bytes actually written - * @stable: An NFS stable_how value + * @iocb_flags: VFS IOCB_* flags expressing the requested write stability * @verf: NFS WRITE verifier * * Upon return, caller must invoke fh_put on @fhp. @@ -1445,7 +1445,7 @@ __be32 nfsd_vfs_write(struct svc_rqst *rqstp, struct svc_fh *fhp, struct nfsd_file *nf, loff_t offset, const struct xdr_buf *payload, unsigned long *cnt, - int stable, __be32 *verf) + int iocb_flags, __be32 *verf) { struct nfsd_net *nn = net_generic(SVC_NET(rqstp), nfsd_net_id); struct file *file = nf->nf_file; @@ -1482,21 +1482,11 @@ nfsd_vfs_write(struct svc_rqst *rqstp, struct svc_fh *fhp, exp = fhp->fh_export; if (!EX_ISSYNC(exp)) - stable = NFS_UNSTABLE; + iocb_flags = 0; init_sync_kiocb(&kiocb, file); kiocb.ki_pos = offset; - if (likely(!fhp->fh_use_wgather)) { - switch (stable) { - case NFS_FILE_SYNC: - /* persist data and timestamps */ - kiocb.ki_flags |= IOCB_DSYNC | IOCB_SYNC; - break; - case NFS_DATA_SYNC: - /* persist data only */ - kiocb.ki_flags |= IOCB_DSYNC; - break; - } - } + if (likely(!fhp->fh_use_wgather)) + kiocb.ki_flags |= iocb_flags; nvecs = xdr_buf_to_bvec(rqstp->rq_bvec, rqstp->rq_maxpages, payload); if (nvecs < 0) { @@ -1537,7 +1527,7 @@ nfsd_vfs_write(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out_nfserr; } - if (stable && fhp->fh_use_wgather) { + if (iocb_flags && fhp->fh_use_wgather) { host_err = wait_for_concurrent_writes(file); if (host_err < 0) commit_reset_write_verifier(nn, rqstp, host_err); @@ -1628,7 +1618,7 @@ __be32 nfsd_read(struct svc_rqst *rqstp, struct svc_fh *fhp, * @offset: Byte offset of start * @payload: xdr_buf containing the write payload * @cnt: IN: number of bytes to write, OUT: number of bytes actually written - * @stable: An NFS stable_how value + * @iocb_flags: VFS IOCB_* flags expressing the requested write stability * @verf: NFS WRITE verifier * * Upon return, caller must invoke fh_put on @fhp. @@ -1638,8 +1628,8 @@ __be32 nfsd_read(struct svc_rqst *rqstp, struct svc_fh *fhp, */ __be32 nfsd_write(struct svc_rqst *rqstp, struct svc_fh *fhp, loff_t offset, - const struct xdr_buf *payload, unsigned long *cnt, int stable, - __be32 *verf) + const struct xdr_buf *payload, unsigned long *cnt, + int iocb_flags, __be32 *verf) { struct nfsd_file *nf; __be32 err; @@ -1651,7 +1641,7 @@ nfsd_write(struct svc_rqst *rqstp, struct svc_fh *fhp, loff_t offset, goto out; err = nfsd_vfs_write(rqstp, fhp, nf, offset, payload, cnt, - stable, verf); + iocb_flags, verf); nfsd_file_put(nf); out: trace_nfsd_write_done(rqstp, fhp, offset, *cnt); diff --git a/fs/nfsd/vfs.h b/fs/nfsd/vfs.h index 5554878781f464..aa7679d4c54adf 100644 --- a/fs/nfsd/vfs.h +++ b/fs/nfsd/vfs.h @@ -135,11 +135,13 @@ __be32 nfsd_read(struct svc_rqst *rqstp, struct svc_fh *fhp, u32 *eof); __be32 nfsd_write(struct svc_rqst *rqstp, struct svc_fh *fhp, loff_t offset, const struct xdr_buf *payload, - unsigned long *cnt, int stable, __be32 *verf); + unsigned long *cnt, int iocb_flags, + __be32 *verf); __be32 nfsd_vfs_write(struct svc_rqst *rqstp, struct svc_fh *fhp, struct nfsd_file *nf, loff_t offset, const struct xdr_buf *payload, - unsigned long *cnt, int stable, __be32 *verf); + unsigned long *cnt, int iocb_flags, + __be32 *verf); __be32 nfsd_readlink(struct svc_rqst *, struct svc_fh *, char *, int *); __be32 nfsd_symlink(struct svc_rqst *, struct svc_fh *, diff --git a/fs/nfsd/xdr3.h b/fs/nfsd/xdr3.h index 344203874b4c47..cad875d1423136 100644 --- a/fs/nfsd/xdr3.h +++ b/fs/nfsd/xdr3.h @@ -39,7 +39,7 @@ struct nfsd3_writeargs { svc_fh fh; __u64 offset; __u32 count; - int stable; + __u32 stable; __u32 len; struct xdr_buf payload; }; diff --git a/include/linux/nfs4.h b/include/linux/nfs4.h index 1a3981c26b23bf..41b7cdcc674f12 100644 --- a/include/linux/nfs4.h +++ b/include/linux/nfs4.h @@ -263,6 +263,12 @@ enum why_no_delegation4 { /* new to v4.1 */ WND4_IS_DIR = 8, }; +enum stable_how4 { + UNSTABLE4 = 0, + DATA_SYNC4 = 1, + FILE_SYNC4 = 2, +}; + enum lock_type4 { NFS4_UNLOCK_LT = 0, NFS4_READ_LT = 1, From 678b6fba6342a60b90bd6ef9df8563f7ccdf396a Mon Sep 17 00:00:00 2001 From: Ameer Hamza Date: Sun, 26 Jul 2026 17:46:58 +0500 Subject: [PATCH 317/857] nfsd: fix race between client_info_show() and free_client() client_info_show() renders /proc/fs/nfsd/clients//info and walks clp->cl_sessions under clp->cl_lock to print each session's slot counts. free_client() tears down the same list without taking cl_lock, and is the only unlocked mutator of cl_sessions. A reader can observe a client mid-teardown because get_nfsdfs_clp() pins the nfs4_client but not its sessions: free_client() frees every session before calling nfsd_client_rmdir(), so an in-flight seq_file reader can follow a list_del()'d node whose ->next now holds LIST_POISON1 and take a general protection fault: Oops: general protection fault, probably for non-canonical address 0xdead00000000014c CPU: 1 UID: 0 PID: 132488 Comm: cat RIP: 0010:client_info_show+0x2bf/0x3d0 RAX: dead000000000100 Call Trace: seq_read_iter+0x12a/0x4b0 seq_read+0xf1/0x130 vfs_read+0xbf/0x350 ksys_read+0x6f/0xf0 do_syscall_64+0x8b/0xcb0 entry_SYSCALL_64_after_hwframe+0x76/0x7e Kernel panic - not syncing: Fatal exception Detach cl_sessions onto a local reaplist under cl_lock, then free the sessions after dropping the lock. Removing entries from cl_sessions under cl_lock matches unhash_session(), and the detach-then-reap shape matches how __destroy_client() reaps cl_delegations. The sessions cannot be freed while cl_lock is held, since free_session() calls nfsd4_del_conns(), which re-acquires it. Reported-by: Nicholas Wolff Fixes: 601c8cb349c2 ("nfsd: add session slot count to /proc/fs/nfsd/clients/*/info") Cc: stable@vger.kernel.org Signed-off-by: Ameer Hamza Link: https://patch.msgid.link/20260726124658.1715711-1-ameer.hamza@truenas.com Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index d7731bb4959c7a..d93672e6fa2690 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -2791,10 +2791,16 @@ void nfsd4_put_client(struct nfs4_client *clp) static void free_client(struct nfs4_client *clp) { - while (!list_empty(&clp->cl_sessions)) { + LIST_HEAD(reaplist); + + /* client_info_show() walks cl_sessions under cl_lock */ + spin_lock(&clp->cl_lock); + list_splice_init(&clp->cl_sessions, &reaplist); + spin_unlock(&clp->cl_lock); + while (!list_empty(&reaplist)) { struct nfsd4_session *ses; - ses = list_entry(clp->cl_sessions.next, struct nfsd4_session, - se_perclnt); + ses = list_entry(reaplist.next, struct nfsd4_session, + se_perclnt); list_del(&ses->se_perclnt); WARN_ON_ONCE(atomic_read(&ses->se_ref)); free_session(ses); From e57a0217c20b2246127b44f287b98851ae2044e7 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:07 -0400 Subject: [PATCH 318/857] NFSD: Move the RPC program definition for LOCALIO Clean up: The definitions for the LOCALIO program are not needed by most files that include linux/nfs.h. Following the convention used by most other in-kernel RPC program implementations, relocate the LOCALIO program definitions to a localio-specific header. Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-2-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/nfs.h | 7 ------- include/linux/nfslocalio.h | 8 ++++++++ 2 files changed, 8 insertions(+), 7 deletions(-) diff --git a/include/linux/nfs.h b/include/linux/nfs.h index 0e2b210c103b69..8c2818db43c51f 100644 --- a/include/linux/nfs.h +++ b/include/linux/nfs.h @@ -15,11 +15,4 @@ #include -/* The LOCALIO program is entirely private to Linux and is - * NOT part of the uapi. - */ -#define NFS_LOCALIO_PROGRAM 400122 -#define LOCALIOPROC_NULL 0 -#define LOCALIOPROC_UUID_IS_LOCAL 1 - #endif /* _LINUX_NFS_H */ diff --git a/include/linux/nfslocalio.h b/include/linux/nfslocalio.h index 3d91043254e64a..d2b39e6e6c6acf 100644 --- a/include/linux/nfslocalio.h +++ b/include/linux/nfslocalio.h @@ -16,6 +16,14 @@ #include #include +/* + * The LOCALIO program is entirely private to Linux and is NOT part of + * the uapi. + */ +#define NFS_LOCALIO_PROGRAM 400122 +#define LOCALIOPROC_NULL 0 +#define LOCALIOPROC_UUID_IS_LOCAL 1 + struct nfs_client; struct nfs_file_localio; From e5098ededa9426542143ab3ce3378912dca6689f Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:08 -0400 Subject: [PATCH 319/857] nfs_common: Remove "#include " from linux/nfslocalio.h Clean up: linux/nfslocalio.h pulls in linux/nfs.h only for the definition of struct nfs_fh, which now lives in linux/nfs_fh.h. Replace linux/nfs.h with linux/nfs_fh.h so that nfslocalio.h no longer carries uapi/linux/nfs.h into its consumers. Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-3-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/nfslocalio.h | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/include/linux/nfslocalio.h b/include/linux/nfslocalio.h index d2b39e6e6c6acf..8ce4d978a6367e 100644 --- a/include/linux/nfslocalio.h +++ b/include/linux/nfslocalio.h @@ -13,7 +13,8 @@ #include #include #include -#include +#include + #include /* From cef7b75684020a75375347546cbc566a6e1bf268 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:09 -0400 Subject: [PATCH 320/857] NFSD: Tighten header includes in localio.c As a prerequisite to converting NFSD to use xdrgen more broadly, NFSD source files should not depend on NFS client headers. fs/nfsd/localio.c is server-side LOCALIO code, yet it pulled in three of them: , the client inode header (struct nfs_inode, NFS_I(), writeback helpers), which server code never uses; , whose only referenced symbol is decode_opaque_fixed(), a static inline that exists to remap the error return to -EIO for client call sites; and the catch-all . Convert the UUID decoder to call the canonical SUNRPC primitive xdr_stream_decode_opaque_fixed() directly. It is shared by client and server, performs the identical bounds check, and is already reachable through . With the wrapper gone, localio.c references no symbol from , and with that header gone, none of the NFSv3 definitions its structs embed are needed here. Drop all three client includes and add what the file actually uses: struct nfs_fh comes from , included directly rather than through nfslocalio.h's conditional re-export, and NFS4_FHSIZE from . enum nfs_stat and nfs_stat_to_errno continue to come from the already-included . Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-4-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/localio.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/nfsd/localio.c b/fs/nfsd/localio.c index c458c01e94783a..4110be02b75092 100644 --- a/fs/nfsd/localio.c +++ b/fs/nfsd/localio.c @@ -11,11 +11,10 @@ #include #include #include -#include +#include #include +#include #include -#include -#include #include #include "nfsd.h" @@ -179,7 +178,7 @@ static bool localio_decode_uuidarg(struct svc_rqst *rqstp, struct localio_uuidarg *argp = rqstp->rq_argp; u8 uuid[UUID_SIZE]; - if (decode_opaque_fixed(xdr, uuid, UUID_SIZE)) + if (xdr_stream_decode_opaque_fixed(xdr, uuid, UUID_SIZE) < 0) return false; import_uuid(&argp->uuid, uuid); From 32874394947e1253c9beba90b6fb498e0db39246 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:10 -0400 Subject: [PATCH 321/857] NFSD: Name the fh_maxsize value that carries no NFS version nfsd_set_fh_dentry() selects behavior specific to an NFS protocol version by matching fh_maxsize against NFS_FHSIZE, NFS3_FHSIZE, or NFS4_FHSIZE. A filehandle that reaches NFSD outside an NFS request has no such version. nlm_fopen() opts out of the switch by passing a bare 0, which matches no arm, and the literal says nothing about why, so an adjacent comment has to carry it. Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-5-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/lockd.c | 3 +-- fs/nfsd/nfsfh.c | 2 ++ fs/nfsd/nfsfh.h | 14 ++++++++++++++ 3 files changed, 17 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/lockd.c b/fs/nfsd/lockd.c index f5a4f352f8abf9..5ec0f545606323 100644 --- a/fs/nfsd/lockd.c +++ b/fs/nfsd/lockd.c @@ -34,8 +34,7 @@ static int nlm_fopen(struct svc_rqst *rqstp, struct nfs_fh *f, int access; struct svc_fh fh; - /* must initialize before using! but maxsize doesn't matter */ - fh_init(&fh,0); + fh_init(&fh, NFSD_FHSIZE_UNSPEC); fh.fh_handle.fh_size = f->size; memcpy(&fh.fh_handle.fh_raw, f->data, f->size); fh.fh_export = NULL; diff --git a/fs/nfsd/nfsfh.c b/fs/nfsd/nfsfh.c index fd721a5a6b37bd..b1f3c22af52586 100644 --- a/fs/nfsd/nfsfh.c +++ b/fs/nfsd/nfsfh.c @@ -335,6 +335,8 @@ static __be32 nfsd_set_fh_dentry(struct svc_rqst *rqstp, struct net *net, } switch (fhp->fh_maxsize) { + case NFSD_FHSIZE_UNSPEC: + break; case NFS4_FHSIZE: if (dentry->d_sb->s_export_op->flags & EXPORT_OP_NOATOMIC_ATTR) fhp->fh_no_atomic_attr = true; diff --git a/fs/nfsd/nfsfh.h b/fs/nfsd/nfsfh.h index ab15b59ac7b3ba..7d8e3f0153073f 100644 --- a/fs/nfsd/nfsfh.h +++ b/fs/nfsd/nfsfh.h @@ -246,6 +246,20 @@ fh_copy_shallow(struct knfsd_fh *dst, const struct knfsd_fh *src) memcpy(&dst->fh_raw, &src->fh_raw, src->fh_size); } +#define NFSD_FHSIZE_UNSPEC 0 + +/** + * fh_init - Prepare a file handle for fh_compose() or fh_verify() + * @fhp: File handle to initialize + * @maxsize: Largest file handle, in bytes, to build in @fhp + * + * @maxsize bounds the handle fh_compose() may build: NFS_FHSIZE, + * NFS3_FHSIZE, and NFS4_FHSIZE additionally select version-specific + * handling in fh_verify(). Callers that only verify an incoming + * handle pass NFSD_FHSIZE_UNSPEC, which cannot be composed. + * + * Return: @fhp + */ static __inline__ struct svc_fh * fh_init(struct svc_fh *fhp, int maxsize) { From 3b3300cdefda7ba040bffaf1412b28b410f8c6a1 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:11 -0400 Subject: [PATCH 322/857] NFSD: Don't apply NFS version-specific behavior to LOCALIO requests LOCALIO serves NFS clients of every version through one entry point, so no protocol version is associated with such a request. nfsd_set_fh_dentry() selects version-specific behavior anyway: its switch keys off fh_maxsize, and nfsd_open_local_fh() passes NFS4_FHSIZE because that is the size of the buffer it copies into, so LOCALIO lands in the NFSv4 arm. fh_getattr() keys off fh_maxsize too and does run on a LOCALIO open, adding STATX_BTIME and STATX_CHANGE_COOKIE to the mask it requests: work on filesystems that compute them for a caller that never reads them. nfsd_open_local_fh() only verifies a handle it received, so it has no maximum size to state. Pass NFSD_FHSIZE_UNSPEC as nlm_fopen() already does, which selects the switch arm that applies no version-specific behavior, and state the bound on the copy out of struct nfs_fh as NFS_MAXFHSIZE. Suggested-by: NeilBrown Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-6-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/localio.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/localio.c b/fs/nfsd/localio.c index 4110be02b75092..33b56d1b3f440b 100644 --- a/fs/nfsd/localio.c +++ b/fs/nfsd/localio.c @@ -11,7 +11,6 @@ #include #include #include -#include #include #include #include @@ -54,7 +53,7 @@ nfsd_open_local_fh(struct net *net, struct auth_domain *dom, struct nfsd_file *localio; __be32 beres; - if (nfs_fh->size > NFS4_FHSIZE) + if (nfs_fh->size > NFS_MAXFHSIZE) return ERR_PTR(-EINVAL); if (!nfsd_net_try_get(net)) @@ -67,7 +66,7 @@ nfsd_open_local_fh(struct net *net, struct auth_domain *dom, return localio; /* nfs_fh -> svc_fh */ - fh_init(&fh, NFS4_FHSIZE); + fh_init(&fh, NFSD_FHSIZE_UNSPEC); fh.fh_handle.fh_size = nfs_fh->size; memcpy(fh.fh_handle.fh_raw, nfs_fh->data, nfs_fh->size); From 4e9efa0577cd9e57c0debddc3b64124c76440989 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:28 -0400 Subject: [PATCH 323/857] NFSD: Budget the CB_SEQUENCE opcode and referring call array count cb_sequence_enc_sz counts the session ID, the four scalar fields, and one referring call list. encode_cb_sequence4args() also emits the CB_SEQUENCE opcode and the csa_referring_call_lists array count, so the macro falls two XDR words short. Every NFS4_enc_cb_*_sz built on it is short by the same two words. NFSD_CB_MAX_REQ_SZ derives from NFS4_enc_cb_recall_sz, so the two missing CB_SEQUENCE words shrink the ca_maxrequestsize that check_backchannel_attrs() accepts by eight bytes. Count both words. The minimum a client must advertise rises by those eight bytes. The short count cannot overrun the send buffer. The macro sizes p_arglen, and rq_callsize adds two credential slacks on top of that. The only client affected is one whose ca_maxrequestsize falls inside those eight bytes. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-1-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index b06d0170d7c43b..04d3e321a972db 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -6,14 +6,14 @@ #define cb_compound_enc_hdr_sz 4 #define cb_compound_dec_hdr_sz (3 + (NFS4_MAXTAGLEN >> 2)) #define sessionid_sz (NFS4_MAX_SESSIONID_LEN >> 2) +#define op_enc_sz 1 #define enc_referring_call4_sz (1 + 1) #define enc_referring_call_list4_sz (sessionid_sz + 1 + \ enc_referring_call4_sz) -#define cb_sequence_enc_sz (sessionid_sz + 4 + \ - enc_referring_call_list4_sz) +#define cb_sequence_enc_sz (op_enc_sz + sessionid_sz + 4 + \ + 1 + enc_referring_call_list4_sz) #define cb_sequence_dec_sz (op_dec_sz + sessionid_sz + 4) -#define op_enc_sz 1 #define op_dec_sz 2 #define enc_nfs4_fh_sz (1 + (NFS4_FHSIZE >> 2)) #define enc_stateid_sz (NFS4_STATEID_SIZE >> 2) From 2689d943520843153427257ad48af2c0b065d968 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:29 -0400 Subject: [PATCH 324/857] NFSD: Budget the CB_RECALL truncate field NFS4_enc_cb_recall_sz counts the CB_RECALL opcode, the stateid, and the file handle. encode_cb_recall4args() also emits the truncate field, so the macro falls one XDR word short. NFSD_CB_MAX_REQ_SZ derives from this macro, so the minimum ca_maxrequestsize a client must advertise rises by four bytes. The field has been unbudgeted since the macro was written. Neither consumer of the macro justifies a backport. rq_callsize covers p_arglen with two credential slacks. The only client affected is one whose ca_maxrequestsize falls inside those four bytes. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-2-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index 04d3e321a972db..a7c8dc355d1aa8 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -19,8 +19,8 @@ #define enc_stateid_sz (NFS4_STATEID_SIZE >> 2) #define NFS4_enc_cb_recall_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ - 1 + enc_stateid_sz + \ - enc_nfs4_fh_sz) + op_enc_sz + enc_stateid_sz + \ + 1 + enc_nfs4_fh_sz) #define NFS4_dec_cb_recall_sz (cb_compound_dec_hdr_sz + \ cb_sequence_dec_sz + \ From 4380e509e104e69e31237093a116ea126d8daaf1 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:30 -0400 Subject: [PATCH 325/857] NFSD: Budget the CB_LAYOUTRECALL recall stateid NFS4_enc_cb_layout_sz counts the opcode, the three scalar fields, the file handle, and the offset and length hypers. encode_cb_layout4args() also emits the layoutrecall4 discriminator and the recall stateid, so the macro falls five XDR words short. This macro sizes p_arglen and nothing else. rq_callsize pads that with two credential slacks, so the shortfall has never reached the send buffer. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-3-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index a7c8dc355d1aa8..5285934b3a54b9 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -27,8 +27,9 @@ op_dec_sz) #define NFS4_enc_cb_layout_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ - 1 + 3 + \ - enc_nfs4_fh_sz + 4) + op_enc_sz + 3 + 1 + \ + enc_nfs4_fh_sz + 4 + \ + enc_stateid_sz) #define NFS4_dec_cb_layout_sz (cb_compound_dec_hdr_sz + \ cb_sequence_dec_sz + \ op_dec_sz) From 657e65c68a57056f0e0eb3b91255fc1dfe4632d0 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:31 -0400 Subject: [PATCH 326/857] NFSD: Budget the CB_OFFLOAD opcode NFS4_enc_cb_offload_sz counts the file handle, the stateid, and the offload information. encode_cb_offload4args() emits an opcode ahead of all three, so the macro falls one XDR word short. This macro sizes p_arglen and nothing else. rq_callsize pads that with two credential slacks, so the shortfall has never reached the send buffer. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-4-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index 5285934b3a54b9..21d8280edead82 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -58,6 +58,7 @@ XDR_QUADLEN(NFS4_VERIFIER_SIZE)) #define NFS4_enc_cb_offload_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ + op_enc_sz + \ enc_nfs4_fh_sz + \ enc_stateid_sz + \ enc_cb_offload_info_sz) From ca01331408ecaab871890ec4c810b5ef1d89869f Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:32 -0400 Subject: [PATCH 327/857] NFSD: Budget the CB_NOTIFY_LOCK opcode NFS4_enc_cb_notify_lock_sz counts the lock owner and the file handle. nfs4_xdr_enc_cb_notify_lock() emits an opcode ahead of both, so the macro falls one XDR word short. This macro sizes p_arglen and nothing else. rq_callsize pads that with two credential slacks, so the shortfall has never reached the send buffer. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-5-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index 21d8280edead82..01bb54f03d2577 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -48,7 +48,7 @@ #define NFS4_enc_cb_notify_lock_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ - 2 + 1 + \ + op_enc_sz + 2 + 1 + \ XDR_QUADLEN(NFS4_OPAQUE_LIMIT) + \ enc_nfs4_fh_sz) #define NFS4_dec_cb_notify_lock_sz (cb_compound_dec_hdr_sz + \ From 7d4800626e120f83dfb79a646d301bfc72e5e9a8 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:33 -0400 Subject: [PATCH 328/857] NFSD: Budget the CB_RECALL_ANY opcode NFS4_enc_cb_recall_any_sz counts the objects-to-keep field, the bitmap array length, and the bitmap word. encode_cb_recallany4args() emits an opcode ahead of all three, so the macro falls one XDR word short. This macro sizes p_arglen and nothing else. rq_callsize pads that with two credential slacks, so the shortfall has never reached the send buffer. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-6-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index 01bb54f03d2577..838f8629821f48 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -67,7 +67,7 @@ op_dec_sz) #define NFS4_enc_cb_recall_any_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ - 1 + 1 + 1) + op_enc_sz + 1 + 1 + 1) #define NFS4_dec_cb_recall_any_sz (cb_compound_dec_hdr_sz + \ cb_sequence_dec_sz + \ op_dec_sz) From 6ad75e17d62d676afe859c50af598cbea8e784dd Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:34 -0400 Subject: [PATCH 329/857] NFSD: Correct locking documentation for delegation sc_status The comment above the SC_STATUS_ flags states that nn->deleg_lock protects sc_status for delegation stateids, but only the transitions made while a delegation is hashed are taken under that lock. This comment was accurate until commit c88c150a467f ("nfsd: fix possible badness in FREE_STATEID") set SC_STATUS_CLOSED under ->cl_lock. Commit 8dd91e8d31fe ("nfsd: fix race between laundromat and free_stateid") added the other two sites. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-7-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/state.h | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/fs/nfsd/state.h b/fs/nfsd/state.h index c4627dc91e2067..42d3320622eb9c 100644 --- a/fs/nfsd/state.h +++ b/fs/nfsd/state.h @@ -145,10 +145,13 @@ struct nfs4_stid { #define SC_TYPE_COPY BIT(4) unsigned short sc_type; -/* nn->deleg_lock protects sc_status for delegation stateids. - * ->cl_lock protects sc_status for open and lock stateids. - * ->st_mutex also protect sc_status for open stateids. - * ->ls_lock protects sc_status for layout stateids. +/* + * nn->deleg_lock protects sc_status for hashed delegation stateids. + * ->cl_lock protects the bits set as one is disposed of + * (SC_STATUS_CLOSED, SC_STATUS_FREEABLE, SC_STATUS_FREED) and + * sc_status for open and lock stateids. ->st_mutex also protects + * sc_status for open stateids. ->ls_lock protects sc_status for + * layout stateids. */ /* * For an open stateid kept around *only* to process close replays. From 69a4cf48f9c3d00ed4d1ce0a670471ee1015368c Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:35 -0400 Subject: [PATCH 330/857] NFSD: Destroy a recalled delegation the client does not hold A client that answers CB_RECALL with NFS4ERR_BADHANDLE or NFS4ERR_BAD_STATEID has no record of the delegation, so the FREE_STATEID that clears it from cl_revoked never arrives. Every later SEQUENCE reply carries SEQ4_STATUS_RECALLABLE_STATE_REVOKED, and the client loops issuing TEST_STATEID. Destroy such a delegation when it is reaped rather than revoking it onto cl_revoked. RFC 8881 Section 20.2.4 completes the recall at the reply when its status is neither NFS4_OK nor NFS4ERR_DELAY, so a rejected recall leaves nothing to revoke. An administrative revoke keeps that path, since NFS4ERR_ADMIN_REVOKED reports it. A destroyed stateid returns NFS4ERR_BAD_STATEID instead of NFS4ERR_DELEG_REVOKED. A client that rejects the recall but still holds the delegation gets no notice that its state was revoked. CB_RECALL can outrun the reply that granted the delegation, so honor a rejection only once the client has seen that grant. Per RFC 8881 Section 2.10.6.3, retirement of the slot that carried the grant is that proof; retry until then, and revoke when the retries lapse. Fixes: 3bd64a5ba171 ("nfsd4: implement SEQ4_STATUS_RECALLABLE_STATE_REVOKED") Cc: stable@vger.kernel.org # 6.14.x Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-8-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 198 +++++++++++++++++++++++++++++++++++++------- fs/nfsd/state.h | 13 ++- 2 files changed, 178 insertions(+), 33 deletions(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index d93672e6fa2690..5774c7a1b3ded9 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -94,6 +94,8 @@ static void nfsd4_end_grace(struct nfsd_net *nn); static void _free_cpntf_state_locked(struct nfsd_net *nn, struct nfs4_cpntf_state *cps); static void nfsd4_file_hash_remove(struct nfs4_file *fi); static void deleg_reaper(struct nfsd_net *nn); +static void nfsd4_drop_revoked_stid(struct nfs4_stid *s) + __releases(&s->sc_client->cl_lock); static const struct lease_manager_operations nfsd_lease_mng_ops; @@ -1281,6 +1283,9 @@ __alloc_init_deleg(struct nfs4_client *clp, struct nfs4_file *fp, dp->dl_type = dl_type; dp->dl_retries = 1; dp->dl_recalled = false; + dp->dl_recall_rejected = false; + dp->dl_recall_grant.valid = false; + dp->dl_recall_grant.retired_at_send = false; get_nfs4_file(fp); dp->dl_stid.sc_file = fp; nfsd4_init_cb(&dp->dl_recall, dp->dl_stid.sc_client, @@ -1565,27 +1570,22 @@ static void destroy_delegation(struct nfs4_delegation *dp) } /** - * revoke_delegation - perform nfs4 delegation structure cleanup - * @dp: pointer to the delegation + * revoke_delegation - dispose of a delegation the server has revoked + * @dp: delegation to dispose of + * + * The caller holds a reference on @dp, which this function consumes. + * On NFSv4.1 and newer, @dp's sc_status must already carry + * SC_STATUS_REVOKED or SC_STATUS_ADMIN_REVOKED. * - * This function assumes that it's called either from the administrative - * interface (nfsd4_revoke_states()) that's revoking a specific delegation - * stateid or it's called from a laundromat thread (nfsd4_landromat()) that - * determined that this specific state has expired and needs to be revoked - * (both mark state with the appropriate stid sc_status mode). It is also - * assumed that a reference was taken on the @dp state. This function - * consumes that reference. + * @dp is parked on the client's cl_revoked list to await a FREE_STATEID. + * Where none can arrive, @dp is destroyed here instead: FREE_STATEID has + * already freed it, or the client rejected the recall with + * NFS4ERR_BADHANDLE or NFS4ERR_BAD_STATEID and holds no record of the + * delegation. NFS4ERR_ADMIN_REVOKED still prompts one, so an + * administrative revoke waits on cl_revoked. * - * If this function finds that the @dp state is SC_STATUS_FREED it means - * that a FREE_STATEID operation for this stateid has been processed and - * we can proceed to removing it from recalled list. However, if @dp state - * isn't marked SC_STATUS_FREED, it means we need place it on the cl_revoked - * list and wait for the FREE_STATEID to arrive from the client. At the same - * time, we need to mark it as SC_STATUS_FREEABLE to indicate to the - * nfsd4_free_stateid() function that this stateid has already been added - * to the cl_revoked list and that nfsd4_free_stateid() is now responsible - * for removing it from the list. Inspection of where the delegation state - * in the revocation process is protected by the clp->cl_lock. + * Context: Takes and releases the client's cl_lock; may sleep after + * dropping it. */ static void revoke_delegation(struct nfs4_delegation *dp) { @@ -1603,6 +1603,19 @@ static void revoke_delegation(struct nfs4_delegation *dp) list_del_init(&dp->dl_recall_lru); goto out; } + if (dp->dl_recall_rejected && + !(dp->dl_stid.sc_status & SC_STATUS_ADMIN_REVOKED)) { + /* + * SC_STATUS_CLOSED, set under cl_lock, makes a racing + * FREE_STATEID bail out rather than drop this reference + * too. The put releases what cl_revoked would have held. + */ + dp->dl_stid.sc_status |= SC_STATUS_CLOSED; + spin_unlock(&clp->cl_lock); + nfs4_put_stid(&dp->dl_stid); + destroy_unhashed_deleg(dp); + return; + } list_add(&dp->dl_recall_lru, &clp->cl_revoked); dp->dl_stid.sc_status |= SC_STATUS_FREEABLE; out: @@ -2897,11 +2910,18 @@ __destroy_client(struct nfs4_client *clp) list_del_init(&dp->dl_recall_lru); destroy_unhashed_deleg(dp); } + /* + * A CB_RECALL reply can release revoked delegations concurrently: + * nfsd4_shutdown_callback() has not run yet. + */ + spin_lock(&clp->cl_lock); while (!list_empty(&clp->cl_revoked)) { dp = list_entry(clp->cl_revoked.next, struct nfs4_delegation, dl_recall_lru); - list_del_init(&dp->dl_recall_lru); - nfs4_put_stid(&dp->dl_stid); + /* this function drops ->cl_lock */ + nfsd4_drop_revoked_stid(&dp->dl_stid); + spin_lock(&clp->cl_lock); } + spin_unlock(&clp->cl_lock); while (!list_empty(&clp->cl_openowners)) { oo = list_entry(clp->cl_openowners.next, struct nfs4_openowner, oo_perclient); nfs4_get_stateowner(&oo->oo_owner); @@ -6071,6 +6091,60 @@ bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, struct inode *inode) return timeo > 0; } +static bool nfsd4_recall_grant_slot_retired(struct nfs4_delegation *dp) +{ + struct nfs4_client *clp = dp->dl_stid.sc_client; + struct nfsd_net *nn = net_generic(clp->net, nfsd_net_id); + struct nfsd4_session *ses; + struct nfsd4_sessionid sid; + bool retired = false; + void *entry; + + if (!dp->dl_recall_grant.valid) + return false; + + /* + * gen_sessionid() composes a sessionid from the client's clientid + * and a sequence counter, so the sequence alone identifies the + * granting session. + */ + sid.clientid = clp->cl_clientid; + sid.sequence = dp->dl_recall_grant.sessionid_seq; + sid.reserved = 0; + + /* + * A missing session does not prove the client saw the grant: a + * DESTROY_SESSION unhashes its own session before the reply to + * that compound is encoded. + */ + spin_lock(&nn->client_lock); + ses = __find_in_sessionid_hashtbl((struct nfs4_sessionid *)&sid, + clp->net); + entry = ses ? xa_load(&ses->se_slots, dp->dl_recall_grant.slotid) : NULL; + if (xa_is_value(entry)) { + /* + * A slot is freed only once the client has acknowledged + * the smaller slot table, which it cannot do while a + * request on that slot is outstanding. + */ + retired = true; + } else if (entry) { + struct nfsd4_slot *slot = entry; + + /* + * A reactivated slot was freed and rebuilt, so the same + * acknowledgment applies. The seqid test errs toward + * revoking: a rebuilt slot restarting at seqid 1 matches + * an old grant. + */ + retired = (slot->sl_flags & NFSD4_SLOT_REUSED) || + ((slot->sl_flags & NFSD4_SLOT_INITIALIZED) && + slot->sl_seqid != dp->dl_recall_grant.seqid); + } + spin_unlock(&nn->client_lock); + return retired; +} + static bool nfsd4_cb_recall_prepare(struct nfsd4_callback *cb) { struct nfs4_delegation *dp = cb_to_delegation(cb); @@ -6092,9 +6166,37 @@ static bool nfsd4_cb_recall_prepare(struct nfsd4_callback *cb) list_add_tail(&dp->dl_recall_lru, &nn->del_recall_lru); } spin_unlock(&nn->deleg_lock); + + dp->dl_recall_grant.retired_at_send = + nfsd4_recall_grant_slot_retired(dp); return true; } +/* + * cl_lock orders this against a laundromat reaping @dp: either + * revoke_delegation() observes dl_recall_rejected and destroys @dp, or + * it reached cl_revoked first and @dp is released here instead. + */ +static void nfsd4_deleg_recall_rejected(struct nfs4_delegation *dp) +{ + struct nfs4_client *clp = dp->dl_stid.sc_client; + + spin_lock(&clp->cl_lock); + if (dp->dl_stid.sc_status & (SC_STATUS_CLOSED | SC_STATUS_FREED | + SC_STATUS_ADMIN_REVOKED)) { + spin_unlock(&clp->cl_lock); + return; + } + if (dp->dl_stid.sc_status & SC_STATUS_FREEABLE) { + dp->dl_stid.sc_status |= SC_STATUS_CLOSED; + /* this function drops ->cl_lock */ + nfsd4_drop_revoked_stid(&dp->dl_stid); + return; + } + dp->dl_recall_rejected = true; + spin_unlock(&clp->cl_lock); +} + static int nfsd4_cb_recall_done(struct nfsd4_callback *cb, struct rpc_task *task) { @@ -6102,27 +6204,33 @@ static int nfsd4_cb_recall_done(struct nfsd4_callback *cb, trace_nfsd_cb_recall_done(&dp->dl_stid.sc_stateid, task); - if (dp->dl_stid.sc_status) - /* CLOSED or REVOKED */ - return 1; - switch (task->tk_status) { case 0: return 1; case -NFS4ERR_DELAY: + if (dp->dl_stid.sc_status) + /* CLOSED or REVOKED */ + return 1; rpc_delay(task, 2 * HZ); return 0; case -EBADHANDLE: case -NFS4ERR_BAD_STATEID: /* - * Race: client probably got cb_recall before open reply - * granting delegation. + * Retirement of the granting slot proves the client saw + * the grant. Trust the rejection only if the slot had + * retired when this recall was sent. */ - if (dp->dl_retries--) { + if (dp->dl_recall_grant.retired_at_send) { + nfsd4_deleg_recall_rejected(dp); + return 1; + } + if (!dp->dl_stid.sc_status && dp->dl_retries--) { + dp->dl_recall_grant.retired_at_send = + nfsd4_recall_grant_slot_retired(dp); rpc_delay(task, 2 * HZ); return 0; } - fallthrough; + return 1; default: return 1; } @@ -6716,9 +6824,25 @@ static bool nfsd4_want_deleg_timestamps(const struct nfsd4_open *open) return open->op_deleg_want & OPEN4_SHARE_ACCESS_WANT_DELEG_TIMESTAMPS; } +static void +nfs4_delegation_record_grant_slot(struct nfs4_delegation *dp, + const struct nfsd4_compound_state *cstate) +{ + const struct nfsd4_sessionid *sid; + + if (!cstate->session) + return; + sid = (struct nfsd4_sessionid *)cstate->session->se_sessionid.data; + dp->dl_recall_grant.sessionid_seq = sid->sequence; + dp->dl_recall_grant.slotid = cstate->slot->sl_index; + dp->dl_recall_grant.seqid = cstate->slot->sl_seqid; + dp->dl_recall_grant.valid = true; +} + static struct nfs4_delegation * -nfs4_set_delegation(struct nfsd4_open *open, struct nfs4_ol_stateid *stp, - struct svc_fh *parent) +nfs4_set_delegation(struct nfsd4_open *open, + const struct nfsd4_compound_state *cstate, + struct nfs4_ol_stateid *stp, struct svc_fh *parent) { bool deleg_ts = nfsd4_want_deleg_timestamps(open); struct nfs4_client *clp = stp->st_stid.sc_client; @@ -6808,6 +6932,14 @@ nfs4_set_delegation(struct nfsd4_open *open, struct nfs4_ol_stateid *stp, dp = alloc_init_deleg(clp, fp, odstate, dl_type); if (!dp) goto out_delegees; + + /* + * Record the granting slot before kernel_setlease() makes @dp + * visible to lease breakers. A conflicting open can drive + * CB_RECALL to completion from that point on. + */ + nfs4_delegation_record_grant_slot(dp, cstate); + if (stp->st_stid.sc_export) dp->dl_stid.sc_export = exp_get(stp->st_stid.sc_export); @@ -6972,6 +7104,7 @@ nfs4_open_delegation(struct svc_rqst *rqstp, struct nfsd4_open *open, struct nfs4_ol_stateid *stp, struct svc_fh *currentfh, struct svc_fh *fh) { + struct nfsd4_compoundres *resp = rqstp->rq_resp; struct nfs4_openowner *oo = openowner(stp->st_stateowner); bool deleg_ts = nfsd4_want_deleg_timestamps(open); struct nfs4_client *clp = stp->st_stid.sc_client; @@ -7008,7 +7141,7 @@ nfs4_open_delegation(struct svc_rqst *rqstp, struct nfsd4_open *open, default: goto out_no_deleg; } - dp = nfs4_set_delegation(open, stp, parent); + dp = nfs4_set_delegation(open, &resp->cstate, stp, parent); if (IS_ERR(dp)) goto out_no_deleg; @@ -10302,6 +10435,7 @@ nfsd_get_dir_deleg(struct nfsd4_compound_state *cstate, dp = alloc_init_dir_deleg(clp, fp); if (!dp) goto out_delegees; + nfs4_delegation_record_grant_slot(dp, cstate); if (cstate->current_fh.fh_export) dp->dl_stid.sc_export = exp_get(cstate->current_fh.fh_export); diff --git a/fs/nfsd/state.h b/fs/nfsd/state.h index 42d3320622eb9c..ff1c9fa731aa25 100644 --- a/fs/nfsd/state.h +++ b/fs/nfsd/state.h @@ -292,7 +292,8 @@ struct nfsd4_cb_notify { * If the server attempts to recall a delegation and the client doesn't do so * before a timeout, the server may also revoke the delegation. In that case, * the object will either be destroyed (v4.0) or moved to a per-client list of - * revoked delegations (v4.1+). + * revoked delegations (v4.1+). A v4.1+ client that rejects the recall holds + * no record of the delegation, so the object is destroyed rather than listed. * * This object is a superset of the nfs4_stid. */ @@ -308,9 +309,19 @@ struct nfs4_delegation { int dl_retries; struct nfsd4_callback dl_recall; bool dl_recalled; + bool dl_recall_rejected; bool dl_written; bool dl_setattr; + /* Forward-channel slot that carried the granting request */ + struct { + u32 sessionid_seq; + u32 slotid; + u32 seqid; + bool valid; + bool retired_at_send; + } dl_recall_grant; + union { /* for CB_GETATTR */ struct nfs4_cb_fattr dl_cb_fattr; From e46b89d87f759b7c146c364447522d304068eaee Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:36 -0400 Subject: [PATCH 331/857] NFSD: Send referring calls with CB_RECALL When CB_RECALL races ahead of the reply that granted the delegation, the client has not yet recorded the delegation stateid and responds NFS4ERR_BADHANDLE or NFS4ERR_BAD_STATEID. The slot that carried the grant has not retired at that point, so NFSD cannot read the rejection as proof that the client never held the delegation. It retries the recall and, once the retries lapse, revokes a delegation the client is by then able to return. Remove the ambiguity with the referring call mechanism of RFC 8881 Section 2.10.6.3: until the slot that carried the grant retires, name that request as a referring call in the CB_SEQUENCE of each recall. A client that finds it still outstanding may respond NFS4ERR_DELAY, and the recall is retried until the client has processed the grant. A recall reuses one callback context across its retries, and ->prepare does not run on every send. The granting request does not change, so a send that inherits the previous list sends the right one. Retirement of the granting slot drops the list, and nfs4_free_deleg() releases what is left. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-9-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4callback.c | 7 ++++-- fs/nfsd/nfs4state.c | 54 ++++++++++++++++++++++++++++++++---------- 2 files changed, 47 insertions(+), 14 deletions(-) diff --git a/fs/nfsd/nfs4callback.c b/fs/nfsd/nfs4callback.c index 509195d488c9a3..9afe2d78d39d16 100644 --- a/fs/nfsd/nfs4callback.c +++ b/fs/nfsd/nfs4callback.c @@ -1530,12 +1530,14 @@ void nfsd41_cb_referring_call(struct nfsd4_callback *cb, /** * nfsd41_cb_destroy_referring_call_list - release referring call info - * @cb: context of a callback that has completed + * @cb: context of callback to release referring calls from * * Callers who allocate referring calls using nfsd41_cb_referring_call() must * release those resources by calling nfsd41_cb_destroy_referring_call_list. * - * Caller serializes access to @cb. + * Caller serializes access to @cb. No CB_COMPOUND for @cb may be in + * flight, because encode_cb_sequence4args() walks this list as it + * encodes. */ void nfsd41_cb_destroy_referring_call_list(struct nfsd4_callback *cb) { @@ -1557,6 +1559,7 @@ void nfsd41_cb_destroy_referring_call_list(struct nfsd4_callback *cb) list_del(&rcl->__list); kfree(rcl); } + cb->cb_nr_referring_call_list = 0; } static void nfsd4_cb_prepare(struct rpc_task *task, void *calldata) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 5774c7a1b3ded9..510380b6aa7a16 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -1167,6 +1167,8 @@ static void nfs4_free_deleg(struct nfs4_stid *stid) WARN_ON_ONCE(!list_empty(&dp->dl_perfile)); WARN_ON_ONCE(!list_empty(&dp->dl_perclnt)); WARN_ON_ONCE(!list_empty(&dp->dl_recall_lru)); + /* The list outlives one recall, so ->release() cannot free it. */ + nfsd41_cb_destroy_referring_call_list(&dp->dl_recall); kmem_cache_free(deleg_slab, stid); atomic_long_dec(&num_delegations); } @@ -6091,6 +6093,18 @@ bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, struct inode *inode) return timeo > 0; } +/* + * gen_sessionid() composes a sessionid from the client's clientid and a + * sequence counter, so the sequence alone identifies the granting session. + */ +static void nfsd4_recall_grant_sessionid(const struct nfs4_delegation *dp, + struct nfsd4_sessionid *sid) +{ + sid->clientid = dp->dl_stid.sc_client->cl_clientid; + sid->sequence = dp->dl_recall_grant.sessionid_seq; + sid->reserved = 0; +} + static bool nfsd4_recall_grant_slot_retired(struct nfs4_delegation *dp) { struct nfs4_client *clp = dp->dl_stid.sc_client; @@ -6103,14 +6117,7 @@ static bool nfsd4_recall_grant_slot_retired(struct nfs4_delegation *dp) if (!dp->dl_recall_grant.valid) return false; - /* - * gen_sessionid() composes a sessionid from the client's clientid - * and a sequence counter, so the sequence alone identifies the - * granting session. - */ - sid.clientid = clp->cl_clientid; - sid.sequence = dp->dl_recall_grant.sessionid_seq; - sid.reserved = 0; + nfsd4_recall_grant_sessionid(dp, &sid); /* * A missing session does not prove the client saw the grant: a @@ -6145,6 +6152,21 @@ static bool nfsd4_recall_grant_slot_retired(struct nfs4_delegation *dp) return retired; } +/* + * ->prepare does not run on every send: nfsd4_run_cb_work() skips it + * on a requeue, and a retry via rpc_restart_call_prepare() re-enters + * the RPC layer beneath it. The granting request does not change, so + * a send inherits a correct list. Retirement is the one transition + * the list has to follow. + */ +static void nfsd4_refresh_recall_grant(struct nfs4_delegation *dp) +{ + dp->dl_recall_grant.retired_at_send = + nfsd4_recall_grant_slot_retired(dp); + if (dp->dl_recall_grant.retired_at_send) + nfsd41_cb_destroy_referring_call_list(&dp->dl_recall); +} + static bool nfsd4_cb_recall_prepare(struct nfsd4_callback *cb) { struct nfs4_delegation *dp = cb_to_delegation(cb); @@ -6167,8 +6189,17 @@ static bool nfsd4_cb_recall_prepare(struct nfsd4_callback *cb) } spin_unlock(&nn->deleg_lock); - dp->dl_recall_grant.retired_at_send = - nfsd4_recall_grant_slot_retired(dp); + nfsd4_refresh_recall_grant(dp); + + if (dp->dl_recall_grant.valid && !dp->dl_recall_grant.retired_at_send) { + struct nfsd4_sessionid sid; + + nfsd4_recall_grant_sessionid(dp, &sid); + nfsd41_cb_referring_call(&dp->dl_recall, + (struct nfs4_sessionid *)&sid, + dp->dl_recall_grant.slotid, + dp->dl_recall_grant.seqid); + } return true; } @@ -6225,8 +6256,7 @@ static int nfsd4_cb_recall_done(struct nfsd4_callback *cb, return 1; } if (!dp->dl_stid.sc_status && dp->dl_retries--) { - dp->dl_recall_grant.retired_at_send = - nfsd4_recall_grant_slot_retired(dp); + nfsd4_refresh_recall_grant(dp); rpc_delay(task, 2 * HZ); return 0; } From f5dc2038906bb0c9627c99bea06dd7786b5ac2d1 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 4 Aug 2026 14:46:30 -0400 Subject: [PATCH 332/857] NFSD: Point contributors and sashiko.dev to the nfsd-testing branch Scripting and automation is sensitive to branch names in the subsystem entries in MAINTAINERS. Rather than pulling from cel.git/master, we really want CI to pull from cel.git/nfsd-testing. Link: https://patch.msgid.link/20260804184630.1395002-1-cel@kernel.org Signed-off-by: Chuck Lever --- MAINTAINERS | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/MAINTAINERS b/MAINTAINERS index 3a19da74d00c9d..a7db3141f75983 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -14191,7 +14191,8 @@ L: linux-nfs@vger.kernel.org S: Supported P: Documentation/filesystems/nfs/nfsd-maintainer-entry-profile.rst B: https://bugzilla.kernel.org -T: git git://git.kernel.org/pub/scm/linux/kernel/git/cel/linux.git +T: git git://git.kernel.org/pub/scm/linux/kernel/git/cel/linux.git nfsd-testing +T: git git://git.kernel.org/pub/scm/linux/kernel/git/cel/linux.git nfsd-next F: Documentation/filesystems/nfs/ F: fs/lockd/ F: fs/nfs_common/ From 26e7d6d7b5906d48f90ce8967a6bffdd8374e9ed Mon Sep 17 00:00:00 2001 From: Linmao Li Date: Mon, 31 Aug 2026 09:45:09 +0800 Subject: [PATCH 333/857] hwmon: (corsair-cpro) Create debugfs entries after hwmon registration ccp_debugfs_init() registers debugfs files whose private data is the devm allocated ccp. It runs before hwmon_device_register_with_info(), so when that registration fails, ccp_probe() returns with the files still in place. The HID core then frees ccp, and ccp_remove() is not called for a failed probe, so nothing removes them later either. Reading one of the files dereferences the freed pointer. Create the debugfs entries only after the hwmon device has been registered, so no failing path can leave them behind. The two version queries stay where they are. They send USB commands without holding ccp->mutex, which is only safe as long as nothing else can call send_usb_cmd(); once the hwmon device is registered its callbacks can do so concurrently. Only the debugfs creation moves, and it is told which queries succeeded. Reported-by: Sashiko Closes: https://lore.kernel.org/linux-hwmon/20260708031612.BD7E61F000E9@smtp.kernel.org/ Suggested-by: Guenter Roeck Fixes: 5997eb60f896 ("hwmon: (corsair-cpro) Add firmware and bootloader information") Signed-off-by: Linmao Li Link: https://patch.msgid.link/20260831014509.3352442-1-lilinmao@kylinos.cn Signed-off-by: Guenter Roeck --- drivers/hwmon/corsair-cpro.c | 20 +++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/drivers/hwmon/corsair-cpro.c b/drivers/hwmon/corsair-cpro.c index 8354a002f4c5e1..56de0fe0f544c5 100644 --- a/drivers/hwmon/corsair-cpro.c +++ b/drivers/hwmon/corsair-cpro.c @@ -566,21 +566,18 @@ static int bootloader_show(struct seq_file *seqf, void *unused) } DEFINE_SHOW_ATTRIBUTE(bootloader); -static void ccp_debugfs_init(struct ccp_device *ccp) +static void ccp_debugfs_init(struct ccp_device *ccp, bool fw_valid, bool bl_valid) { char name[32]; - int ret; scnprintf(name, sizeof(name), "corsaircpro-%s", dev_name(&ccp->hdev->dev)); ccp->debugfs = debugfs_create_dir(name, NULL); - ret = get_fw_version(ccp); - if (!ret) + if (fw_valid) debugfs_create_file("firmware_version", 0444, ccp->debugfs, ccp, &firmware_fops); - ret = get_bl_version(ccp); - if (!ret) + if (bl_valid) debugfs_create_file("bootloader_version", 0444, ccp->debugfs, ccp, &bootloader_fops); } @@ -588,6 +585,7 @@ static void ccp_debugfs_init(struct ccp_device *ccp) static int ccp_probe(struct hid_device *hdev, const struct hid_device_id *id) { struct ccp_device *ccp; + bool fw_valid, bl_valid; int ret; ccp = devm_kzalloc(&hdev->dev, sizeof(*ccp), GFP_KERNEL); @@ -632,7 +630,13 @@ static int ccp_probe(struct hid_device *hdev, const struct hid_device_id *id) if (ret) goto out_hw_close; - ccp_debugfs_init(ccp); + /* + * Query the versions before registering the hwmon device: they send + * USB commands without holding ccp->mutex, which is only safe while + * nothing else can call send_usb_cmd(). + */ + fw_valid = !get_fw_version(ccp); + bl_valid = !get_bl_version(ccp); ccp->hwmon_dev = hwmon_device_register_with_info(&hdev->dev, "corsaircpro", ccp, &ccp_chip_info, NULL); @@ -641,6 +645,8 @@ static int ccp_probe(struct hid_device *hdev, const struct hid_device_id *id) goto out_hw_close; } + ccp_debugfs_init(ccp, fw_valid, bl_valid); + return 0; out_hw_close: From eb656c5bb75434f89e87ea5c0321a0ec6eb7c67c Mon Sep 17 00:00:00 2001 From: Laxman Acharya Padhya Date: Mon, 31 Aug 2026 15:44:21 +0545 Subject: [PATCH 334/857] Bluetooth: btintel: validate version TLV value lengths btintel_parse_version_tlv() verifies that a complete TLV is present in the response, but it does not ensure that the value is long enough for the specific TLV type. A short value can therefore cause an out-of-bounds read through get_unaligned_le16(), get_unaligned_le32(), or memcpy(). Reject values shorter than the minimum required by each known TLV type. Also reject responses that do not contain the Command Complete Status field. Fixes: 57375beef71a ("Bluetooth: btintel: Add infrastructure to read controller information") Reviewed-by: Ali Ahmet Memis Signed-off-by: Laxman Acharya Padhya Tested-by: Kiran K Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btintel.c | 37 ++++++++++++++++++++++++++++++++++++- 1 file changed, 36 insertions(+), 1 deletion(-) diff --git a/drivers/bluetooth/btintel.c b/drivers/bluetooth/btintel.c index cbeb27033aa1e7..998c99b1799ccf 100644 --- a/drivers/bluetooth/btintel.c +++ b/drivers/bluetooth/btintel.c @@ -573,12 +573,44 @@ int btintel_version_info_tlv(struct hci_dev *hdev, } EXPORT_SYMBOL_GPL(btintel_version_info_tlv); +static u8 btintel_version_tlv_min_len(u8 type) +{ + switch (type) { + case INTEL_TLV_CNVI_TOP: + case INTEL_TLV_CNVR_TOP: + case INTEL_TLV_CNVI_BT: + case INTEL_TLV_CNVR_BT: + case INTEL_TLV_BUILD_NUM: + case INTEL_TLV_GIT_SHA1: + return sizeof(u32); + case INTEL_TLV_DEV_REV_ID: + case INTEL_TLV_TIME_STAMP: + return sizeof(u16); + case INTEL_TLV_IMAGE_TYPE: + case INTEL_TLV_BUILD_TYPE: + case INTEL_TLV_SECURE_BOOT: + case INTEL_TLV_OTP_LOCK: + case INTEL_TLV_API_LOCK: + case INTEL_TLV_DEBUG_LOCK: + case INTEL_TLV_LIMITED_CCE: + case INTEL_TLV_SBE_TYPE: + return sizeof(u8); + case INTEL_TLV_MIN_FW: + return 3; + case INTEL_TLV_OTP_BDADDR: + return sizeof(bdaddr_t); + default: + return 0; + } +} + int btintel_parse_version_tlv(struct hci_dev *hdev, struct intel_version_tlv *version, struct sk_buff *skb) { /* Consume Command Complete Status field */ - skb_pull(skb, 1); + if (!skb_pull(skb, 1)) + return -EINVAL; /* Event parameters contain multiple TLVs. Read each of them * and only keep the required data. Also, it use existing legacy @@ -598,6 +630,9 @@ int btintel_parse_version_tlv(struct hci_dev *hdev, if (skb->len < tlv->len + sizeof(*tlv)) return -EINVAL; + if (tlv->len < btintel_version_tlv_min_len(tlv->type)) + return -EINVAL; + switch (tlv->type) { case INTEL_TLV_CNVI_TOP: version->cnvi_top = get_unaligned_le32(tlv->val); From 58c6f5ec1d22b52a0b667838c38a53842ffae0e5 Mon Sep 17 00:00:00 2001 From: Laxman Acharya Padhya Date: Mon, 31 Aug 2026 15:44:22 +0545 Subject: [PATCH 335/857] Bluetooth: btintel: bound firmware ID by TLV length The firmware ID is treated as a NUL-terminated string even though the TLV length is its only boundary. If the value does not contain a NUL terminator, snprintf() can read beyond the received response. Limit the conversion to the advertised TLV value length. Fixes: 164c62f958f8 ("Bluetooth: btintel: Add firmware ID to firmware name") Reviewed-by: Ali Ahmet Memis Signed-off-by: Laxman Acharya Padhya Tested-by: Kiran K Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btintel.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/bluetooth/btintel.c b/drivers/bluetooth/btintel.c index 998c99b1799ccf..8871705343d4a9 100644 --- a/drivers/bluetooth/btintel.c +++ b/drivers/bluetooth/btintel.c @@ -704,7 +704,7 @@ int btintel_parse_version_tlv(struct hci_dev *hdev, break; case INTEL_TLV_FW_ID: snprintf(version->fw_id, sizeof(version->fw_id), - "%s", tlv->val); + "%.*s", tlv->len, tlv->val); break; default: /* Ignore rest of information */ From f8c8fa407aa0f99f4b486aee9d1f5bec284f52c8 Mon Sep 17 00:00:00 2001 From: Laxman Acharya Padhya Date: Mon, 31 Aug 2026 15:44:23 +0545 Subject: [PATCH 336/857] Bluetooth: btintel: propagate version TLV parsing errors btintel_read_version_tlv() ignores the parser return value, so setup continues with partially initialized version data after a malformed TLV causes parsing to stop. Return the parser error to the caller so an invalid response fails setup instead of being treated as successful. Keep this behavioral change separate from the bounds checks so it can be reverted independently if an existing controller sends malformed data. Signed-off-by: Laxman Acharya Padhya Reviewed-by: Ali Ahmet Memis Tested-by: Kiran K Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btintel.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/drivers/bluetooth/btintel.c b/drivers/bluetooth/btintel.c index 8871705343d4a9..964d2de30e6539 100644 --- a/drivers/bluetooth/btintel.c +++ b/drivers/bluetooth/btintel.c @@ -723,6 +723,7 @@ static int btintel_read_version_tlv(struct hci_dev *hdev, { struct sk_buff *skb; const u8 param[1] = { 0xFF }; + int err; if (!version) return -EINVAL; @@ -741,10 +742,10 @@ static int btintel_read_version_tlv(struct hci_dev *hdev, return -EIO; } - btintel_parse_version_tlv(hdev, version, skb); + err = btintel_parse_version_tlv(hdev, version, skb); kfree_skb(skb); - return 0; + return err; } /* ------- REGMAP IBT SUPPORT ------- */ From 4988456a75ea31a367a73b6e519158a0440036f5 Mon Sep 17 00:00:00 2001 From: Chris Lu Date: Tue, 25 Aug 2026 11:36:33 +0800 Subject: [PATCH 337/857] Bluetooth: btmtksdio: Remove redundant firmware filename override btmtksdio_setup() derives the firmware filename with btmtk_fw_get_filename() and then overwrites it with an snprintf() that open-codes that helper's fallback format. Commit 7f935b21bee4 ("Bluetooth: btmtk: apply the common btmtk_fw_get_filename") added the helper call without removing the snprintf() it was meant to replace. None of the device ids the helper special-cases can appear here: 0x6639, 0x7925 and the flavored 0x7961 belong to parts with no SDIO interface, and btmtksdio_setup() passes a flavor of 0 accordingly. The helper always falls through to the snprintf()'s own format, so both produce the same string and removing it is a no-op. Remove it anyway, since it silently defeats the helper for any device id the helper special-cases. Signed-off-by: Chris Lu Assisted-by: Claude:claude-opus-5 Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmtksdio.c | 3 --- 1 file changed, 3 deletions(-) diff --git a/drivers/bluetooth/btmtksdio.c b/drivers/bluetooth/btmtksdio.c index b7f0be7fc42a92..a5709cecd4b52a 100644 --- a/drivers/bluetooth/btmtksdio.c +++ b/drivers/bluetooth/btmtksdio.c @@ -1162,9 +1162,6 @@ static int btmtksdio_setup(struct hci_dev *hdev) btmtk_fw_get_filename(fwname, sizeof(fwname), dev_id, fw_version, 0); - snprintf(fwname, sizeof(fwname), - "mediatek/BT_RAM_CODE_MT%04x_1_%x_hdr.bin", - dev_id & 0xffff, (fw_version & 0xff) + 1); err = mt79xx_setup(hdev, fwname); if (err < 0) return err; From afe439f355464725801cde732446e0007f63d303 Mon Sep 17 00:00:00 2001 From: Chris Lu Date: Tue, 25 Aug 2026 11:36:34 +0800 Subject: [PATCH 338/857] Bluetooth: btmtksdio: Pass the hardware device id to mt79xx_setup() mt79xx_setup() passes a hardcoded 0 to btmtk_setup_firmware_79xx(), discarding the device id that btmtksdio_setup() has just read from register 0x70010200. That argument only gates the section filtering for MT6639, which has no SDIO interface, so this is a no-op on supported hardware and carries no Fixes: tag. Pass the value that has already been read, matching the USB path. Declare dev_id as u32 while at it, since that is what btmtksdio_mtk_reg_read() writes through the pointer. Signed-off-by: Chris Lu Assisted-by: Claude:claude-opus-5 Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmtksdio.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/drivers/bluetooth/btmtksdio.c b/drivers/bluetooth/btmtksdio.c index a5709cecd4b52a..fe4ca9395aa320 100644 --- a/drivers/bluetooth/btmtksdio.c +++ b/drivers/bluetooth/btmtksdio.c @@ -899,14 +899,14 @@ static int mt76xx_setup(struct hci_dev *hdev, const char *fwname) return 0; } -static int mt79xx_setup(struct hci_dev *hdev, const char *fwname) +static int mt79xx_setup(struct hci_dev *hdev, const char *fwname, u32 dev_id) { struct btmtksdio_dev *bdev = hci_get_drvdata(hdev); struct btmtk_hci_wmt_params wmt_params; u8 param = 0x1; int err; - err = btmtk_setup_firmware_79xx(hdev, fwname, mtk_hci_wmt_sync, 0); + err = btmtk_setup_firmware_79xx(hdev, fwname, mtk_hci_wmt_sync, dev_id); if (err < 0) { bt_dev_err(hdev, "Failed to setup 79xx firmware (%d)", err); return err; @@ -1119,8 +1119,8 @@ static int btmtksdio_setup(struct hci_dev *hdev) ktime_t calltime, delta, rettime; unsigned long long duration; char fwname[64]; - int err, dev_id; - u32 fw_version = 0, val; + int err; + u32 dev_id, fw_version = 0, val; calltime = ktime_get(); set_bit(BTMTKSDIO_HW_TX_READY, &bdev->tx_state); @@ -1162,7 +1162,7 @@ static int btmtksdio_setup(struct hci_dev *hdev) btmtk_fw_get_filename(fwname, sizeof(fwname), dev_id, fw_version, 0); - err = mt79xx_setup(hdev, fwname); + err = mt79xx_setup(hdev, fwname, dev_id); if (err < 0) return err; From 78f08df9649c18804ea935b9226395c97aac4a99 Mon Sep 17 00:00:00 2001 From: Aleksandr Nogikh Date: Fri, 28 Aug 2026 08:55:09 +0000 Subject: [PATCH 339/857] Bluetooth: hci_core: Fix race condition during device registration In hci_register_dev(), the power_on work item is queued to hdev->req_workqueue before initializing hdev->adv_monitors_idr and registering the MSFT extension via msft_register(). For devices marked with quirks such as HCI_QUIRK_RAW_DEVICE, the HCI_UNCONFIGURED flag is set on the device. When the power_on work item runs concurrently on another CPU, hci_power_on() detects that the device is unconfigured and immediately invokes hci_dev_do_close(), which calls msft_do_close(). Concurrently, msft_register() allocates the msft structure and exposes it to hdev->msft_data prior to calling mutex_init(&msft->filter_lock). If msft_do_close() executes while hdev->msft_data is already assigned but the mutex has not yet been initialized, mutex_lock(&msft->filter_lock) operates on an uninitialized mutex, triggering a DEBUG_LOCKS warning: DEBUG_LOCKS_WARN_ON(lock->magic != lock) WARNING: kernel/locking/mutex.c:625 at __mutex_lock_common kernel/locking/mutex.c:625 [inline] WARNING: kernel/locking/mutex.c:625 at __mutex_lock+0x12d8/0x1550 kernel/locking/mutex.c:821 ... Call Trace: msft_do_close+0x308/0x7b0 net/bluetooth/msft.c:693 hci_dev_close_sync+0x86b/0x10a0 net/bluetooth/hci_sync.c:5522 hci_dev_do_close net/bluetooth/hci_core.c:499 [inline] hci_power_on+0x32c/0x750 net/bluetooth/hci_core.c:937 process_one_work kernel/workqueue.c:3322 [inline] process_scheduled_works+0xa8e/0x14e0 kernel/workqueue.c:3405 worker_thread+0x92d/0xe10 kernel/workqueue.c:3486 kthread+0x388/0x470 kernel/kthread.c:436 ret_from_fork+0x514/0xb70 arch/x86/kernel/process.c:158 ret_from_fork_asm+0x1a/0x30 arch/x86/entry/entry_64.S:245 Fix this by moving the queue_work() call in hci_register_dev() to after idr_init(&hdev->adv_monitors_idr) and msft_register(hdev) so that device structures and extensions are fully initialized before asynchronous tasks can access them. Additionally, assign hdev->msft_data in msft_register() only after mutex_init(&msft->filter_lock) has completed. Fixes: 9e14606d8f38 ("Bluetooth: msft: Extended monitor tracking by address filter") Assisted-by: Gemini:gemini-3.7-flash Gemini:gemini-3.1-pro-preview syzbot Reported-by: syzbot+14ce1b05b7d5a989abbe@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=14ce1b05b7d5a989abbe Link: https://syzkaller.appspot.com/ai_job?id=2bc9e8aa-ca6d-43e2-be2c-fd5d9f649d7e Signed-off-by: Aleksandr Nogikh Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/hci_core.c | 4 ++-- net/bluetooth/msft.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/net/bluetooth/hci_core.c b/net/bluetooth/hci_core.c index 88df159d339371..66840df8c020f7 100644 --- a/net/bluetooth/hci_core.c +++ b/net/bluetooth/hci_core.c @@ -2632,11 +2632,11 @@ int hci_register_dev(struct hci_dev *hdev) if (error) BT_WARN("register suspend notifier failed error:%d\n", error); - queue_work(hdev->req_workqueue, &hdev->power_on); - idr_init(&hdev->adv_monitors_idr); msft_register(hdev); + queue_work(hdev->req_workqueue, &hdev->power_on); + return id; err_wqueue: diff --git a/net/bluetooth/msft.c b/net/bluetooth/msft.c index ded68568e6c9b5..d9dd722db3ebd5 100644 --- a/net/bluetooth/msft.c +++ b/net/bluetooth/msft.c @@ -769,8 +769,8 @@ void msft_register(struct hci_dev *hdev) INIT_LIST_HEAD(&msft->handle_map); INIT_LIST_HEAD(&msft->address_filters); - hdev->msft_data = msft; mutex_init(&msft->filter_lock); + hdev->msft_data = msft; } void msft_release(struct hci_dev *hdev) From b1f1766ef7691f6ae7478b9613e978adaaf693ae Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 30 Aug 2026 20:11:36 +0300 Subject: [PATCH 340/857] Bluetooth: L2CAP: fix chan mode for LE_CONN_REQ + EXT_FLOWCTL pchan l2cap_new_connection() sets default value of channel mode to match the parent channel. l2cap_le_connect_req() left this at the default, and created L2CAP_MODE_EXT_FLOWCTL channels if listening pchan has that mode. This causes FLAG_DEFER_SETUP channels to reply to L2CAP_LE_CONN_REQ with L2CAP_ECRED_CONN_RSP, which is incorrect. It can also result to stack OOB write (of l2cap_alloc_cid determined values) in l2cap_ecred_rsp_defer(), as l2cap_le_connect_req() does not limit maximum number of deferred channels or check for duplicate ident. Fix by setting chan->mode correctly in l2cap_le_connect_req(). Also check channel mode in l2cap_ecred_rsp_defer(), and do WARN_ON_ONCE instead of OOB write to make it less brittle. Fixes: 15f02b910562 ("Bluetooth: L2CAP: Add initial code for Enhanced Credit Based Mode") Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_core.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 358b11eabd4f5b..fa7dbf5f448eed 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -3886,6 +3886,9 @@ static void l2cap_ecred_rsp_defer(struct l2cap_chan *chan, void *data) struct l2cap_ecred_conn_rsp *rsp_flex = container_of(&rsp->pdu.rsp, struct l2cap_ecred_conn_rsp, hdr); + if (chan->mode != L2CAP_MODE_EXT_FLOWCTL) + return; + /* Check if channel for outgoing connection or if it wasn't deferred * since in those cases it must be skipped. */ @@ -3896,6 +3899,10 @@ static void l2cap_ecred_rsp_defer(struct l2cap_chan *chan, void *data) /* Reset ident so only one response is sent */ chan->ident = 0; + /* Unreachable, check in l2cap_ecred_conn_req. If reached, drop rest */ + if (WARN_ON_ONCE(rsp->count >= ARRAY_SIZE(rsp->pdu.scid))) + rsp->pdu.rsp.result = cpu_to_le16(L2CAP_CR_LE_NO_MEM); + /* Include all channels pending with the same ident */ if (!rsp->pdu.rsp.result) rsp_flex->dcid[rsp->count++] = cpu_to_le16(chan->scid); @@ -5064,6 +5071,7 @@ static int l2cap_le_connect_req(struct l2cap_conn *conn, __set_chan_timer(chan, chan->ops->get_sndtimeo(chan)); chan->ident = cmd->ident; + chan->mode = L2CAP_MODE_LE_FLOWCTL; if (test_bit(FLAG_DEFER_SETUP, &chan->flags)) { l2cap_state_change(chan, BT_CONNECT2); From ddaccd985bb01da7384af3f0524c29fa551c10e2 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 30 Aug 2026 15:04:01 +0300 Subject: [PATCH 341/857] Bluetooth: L2CAP: fix out-of-bounds write in l2cap_ecred_connect l2cap_chan_connect() tries to ensure there are no more than L2CAP_ECRED_CONN_SCID_MAX pending ECRED channels, so they fit in the same L2CAP_ECRED_CONN_REQ that l2cap_ecred_connect() constructs. However, the check only counts deferred channels. If 6 L2CAP sockets are connected at the same time in order DDDDND (D=deferred, N=non-deferred), the last can bump the total to max+1. It results to one __le16 written out of bounds of the scid array, and an invalid ECRED_CONN_REQ being sent. Fix by leaving room for the non-deferred pending ECRED channels in the counting in l2cap_chan_connect(), so the limit can't be exceeded. Move counting under same critical section where the channel is added. Although race conditions involving this appear unreachable, it's easier to see. Also add WARN_ON_ONCE check in l2cap_ecred_defer_connect() to make this less brittle. Fixes: da49b602f7f7 ("Bluetooth: L2CAP: Use DEFER_SETUP to group ECRED connections") Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_core.c | 20 ++++++++++++++------ 1 file changed, 14 insertions(+), 6 deletions(-) diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index fa7dbf5f448eed..84a6ef920ba383 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -1329,7 +1329,7 @@ static void l2cap_le_connect(struct l2cap_chan *chan) struct l2cap_ecred_conn_data { struct { struct l2cap_ecred_conn_req_hdr req; - __le16 scid[5]; + __le16 scid[L2CAP_ECRED_CONN_SCID_MAX]; } __packed pdu; struct l2cap_chan *chan; struct pid *pid; @@ -1357,6 +1357,10 @@ static void l2cap_ecred_defer_connect(struct l2cap_chan *chan, void *data) if (test_and_set_bit(FLAG_ECRED_CONN_REQ_SENT, &chan->flags)) return; + /* Unreachable, checked in l2cap_connect (+timer drops it if reached) */ + if (WARN_ON_ONCE(conn->count >= ARRAY_SIZE(conn->pdu.scid))) + return; + l2cap_ecred_init(chan, 0); /* Set the same ident so we can match on the rsp */ @@ -7382,6 +7386,9 @@ int l2cap_chan_connect(struct l2cap_chan *chan, __le16 psm, u16 cid, goto done; } + mutex_lock(&conn->lock); + l2cap_chan_lock(chan); + if (chan->mode == L2CAP_MODE_EXT_FLOWCTL) { struct l2cap_chan_data data; @@ -7389,19 +7396,20 @@ int l2cap_chan_connect(struct l2cap_chan *chan, __le16 psm, u16 cid, data.pid = chan->ops->get_peer_pid(chan); data.count = 1; - l2cap_chan_list(conn, l2cap_chan_by_pid, &data); + __l2cap_chan_list(conn, l2cap_chan_by_pid, &data); + + /* Leave room for non-deferred channel that ends the group. */ + if (test_bit(FLAG_DEFER_SETUP, &chan->flags)) + data.count += 1; /* Check if there isn't too many channels being connected */ if (data.count > L2CAP_ECRED_CONN_SCID_MAX) { hci_conn_drop(hcon); err = -EPROTO; - goto done; + goto chan_unlock; } } - mutex_lock(&conn->lock); - l2cap_chan_lock(chan); - if (cid && __l2cap_get_chan_by_dcid(conn, cid)) { hci_conn_drop(hcon); err = -EBUSY; From af04b0e3176e55baaf2c080401486cfc7eb957af Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sun, 30 Aug 2026 15:04:02 +0300 Subject: [PATCH 342/857] Bluetooth: L2CAP: clear FLAG_DEFER_SETUP only for same PID/PSM l2cap_ecred_defer_connect() clears FLAG_DEFER_SETUP also for channels with different PID/PSM, which will not be added to the same ECRED_CONN_REQ in any case. Consequently, only one ECRED connection group can work at a time although it appears intended they would be separate for each PID/PSM combination. Fix by clearing FLAG_DEFER_SETUP only for the connections that could be added in the request. Retain test_bit(FLAG_DEFER_SETUP) before calling get_peer_pid as it may be NULL otherwise. Fixes: da49b602f7f7 ("Bluetooth: L2CAP: Use DEFER_SETUP to group ECRED connections") Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_core.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 84a6ef920ba383..2410e8f6d58787 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -1344,7 +1344,7 @@ static void l2cap_ecred_defer_connect(struct l2cap_chan *chan, void *data) if (chan == conn->chan) return; - if (!test_and_clear_bit(FLAG_DEFER_SETUP, &chan->flags)) + if (!test_bit(FLAG_DEFER_SETUP, &chan->flags)) return; pid = chan->ops->get_peer_pid(chan); @@ -1354,6 +1354,9 @@ static void l2cap_ecred_defer_connect(struct l2cap_chan *chan, void *data) chan->mode != L2CAP_MODE_EXT_FLOWCTL || chan->state != BT_CONNECT) return; + if (!test_and_clear_bit(FLAG_DEFER_SETUP, &chan->flags)) + return; + if (test_and_set_bit(FLAG_ECRED_CONN_REQ_SENT, &chan->flags)) return; From 870187be2362118ce51f6d583881d381f2ffde81 Mon Sep 17 00:00:00 2001 From: Gongwei Li Date: Tue, 25 Aug 2026 10:01:45 +0800 Subject: [PATCH 343/857] Bluetooth: hci_mrvl: Fix wrong return value check of wait_on_bit_timeout() wait_on_bit_timeout() returns 0 if the bit was cleared, -EINTR if the process received a signal and the mode permitted wake up on that signal, or -EAGAIN if the timeout elapsed. It never returns 1. Hence the check "err == 1" in mrvl_load_firmware() is dead code: when the waiting task is interrupted by a signal (-EINTR), the code falls into the "else if (err)" branch and misreports it as "Firmware request timeout" with -ETIMEDOUT instead of propagating -EINTR. Fix this by testing for -EINTR so that an interrupted firmware load is properly detected and reported. Fixes: 162f812f23ba ("Bluetooth: hci_uart: Add Marvell support") Signed-off-by: Gongwei Li Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/hci_mrvl.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/drivers/bluetooth/hci_mrvl.c b/drivers/bluetooth/hci_mrvl.c index 516b8f74c43406..5798a8db016ee0 100644 --- a/drivers/bluetooth/hci_mrvl.c +++ b/drivers/bluetooth/hci_mrvl.c @@ -307,9 +307,8 @@ static int mrvl_load_firmware(struct hci_dev *hdev, const char *name) err = wait_on_bit_timeout(&mrvl->flags, STATE_FW_REQ_PENDING, TASK_INTERRUPTIBLE, msecs_to_jiffies(2000)); - if (err == 1) { + if (err == -EINTR) { bt_dev_err(hdev, "Firmware load interrupted"); - err = -EINTR; break; } else if (err) { bt_dev_err(hdev, "Firmware request timeout"); From 5497dcee4c8a42a220e5eab282899e5ebbd3fc0b Mon Sep 17 00:00:00 2001 From: Pengpeng Hou Date: Wed, 15 Jul 2026 16:43:25 +0800 Subject: [PATCH 344/857] nfc: nfcmrvl: validate helper command length before pull The firmware download receive path removes the NCI data header and reads the helper command before validating the remaining packet length. A short frame can therefore reach the data access before the malformed packet is rejected. Validate the complete helper command length before stripping the NCI data header. Fixes: 3194c6870158 ("NFC: nfcmrvl: add firmware download support") Signed-off-by: Pengpeng Hou Link: https://patch.msgid.link/20260715084325.40276-1-pengpeng@iscas.ac.cn Signed-off-by: David Heidelberg --- drivers/nfc/nfcmrvl/fw_dnld.c | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/drivers/nfc/nfcmrvl/fw_dnld.c b/drivers/nfc/nfcmrvl/fw_dnld.c index 2b8f401d8fd7a6..8b9d5257320dcc 100644 --- a/drivers/nfc/nfcmrvl/fw_dnld.c +++ b/drivers/nfc/nfcmrvl/fw_dnld.c @@ -263,9 +263,14 @@ static int process_state_fw_dnld(struct nfcmrvl_private *priv, * B8..N: payload */ - /* Remove NCI HDR */ - skb_pull(skb, 3); - if (skb->data[0] != HELPER_CMD_PACKET_FORMAT || skb->len != 5) { + if (skb->len != NCI_DATA_HDR_SIZE + 5) { + nfc_err(priv->dev, "bad command"); + return -EINVAL; + } + + /* Remove NCI header */ + skb_pull(skb, NCI_DATA_HDR_SIZE); + if (skb->data[0] != HELPER_CMD_PACKET_FORMAT) { nfc_err(priv->dev, "bad command"); return -EINVAL; } From 46a9b322403d3da8a7639db364b451bfb8c27a88 Mon Sep 17 00:00:00 2001 From: Pengpeng Hou Date: Wed, 15 Jul 2026 16:44:05 +0800 Subject: [PATCH 345/857] nfc: st21nfca: validate received frame size st21nfca_hci_i2c_repack() trims a received frame at its EOF marker before removing byte stuffing. It then assumes the truncated frame contains the LLC header and two CRC bytes, and it unconditionally reads the byte after an escape marker. A malformed frame can place EOF immediately after the start marker or can end its data portion with an escape marker. The former leaves too few bytes for check_crc(), while the latter makes the unstuffing loop read past the current skb length. Require the minimum framing bytes both before and after unstuffing. Use separate input and output cursors while removing byte stuffing, and reject an escape marker without its encoded byte. This keeps malformed frames within the received frame boundary before CRC processing. Fixes: 3096e25a3e40 ("NFC: st21nfca: Fix incorrect byte stuffing revocation") Signed-off-by: Pengpeng Hou Link: https://patch.msgid.link/20260715084405.41546-1-pengpeng@iscas.ac.cn Signed-off-by: David Heidelberg --- drivers/nfc/st21nfca/i2c.c | 29 +++++++++++++++++++---------- 1 file changed, 19 insertions(+), 10 deletions(-) diff --git a/drivers/nfc/st21nfca/i2c.c b/drivers/nfc/st21nfca/i2c.c index a4c93ff7c5b0c2..0f44c783bd0417 100644 --- a/drivers/nfc/st21nfca/i2c.c +++ b/drivers/nfc/st21nfca/i2c.c @@ -289,27 +289,36 @@ static int check_crc(u8 *buf, int buflen) */ static int st21nfca_hci_i2c_repack(struct sk_buff *skb) { - int i, j, r, size; + int read, write, r, size; - if (skb->len < 1 || (skb->len > 1 && skb->data[1] != 0)) + if (skb->len < ST21NFCA_FRAME_HEADROOM || + !IS_START_OF_FRAME(skb->data)) return -EBADMSG; size = get_frame_size(skb->data, skb->len); if (size > 0) { + if (size < ST21NFCA_FRAME_HEADROOM + 2) + return -EBADMSG; + skb_trim(skb, size); /* remove ST21NFCA byte stuffing for upper layer */ - for (i = 1, j = 0; i < skb->len; i++) { - if (skb->data[i + j] == + for (read = 1, write = 1; read < skb->len;) { + if (skb->data[read] == (u8) ST21NFCA_ESCAPE_BYTE_STUFFING) { - skb->data[i] = skb->data[i + j + 1] - | ST21NFCA_BYTE_STUFFING_MASK; - i++; - j++; + if (read + 1 == skb->len) + return -EBADMSG; + + skb->data[write++] = skb->data[read + 1] + | ST21NFCA_BYTE_STUFFING_MASK; + read += 2; + } else { + skb->data[write++] = skb->data[read++]; } - skb->data[i] = skb->data[i + j]; } /* remove byte stuffing useless byte */ - skb_trim(skb, i - j); + skb_trim(skb, write); + if (skb->len < ST21NFCA_FRAME_HEADROOM + 2) + return -EBADMSG; /* remove ST21NFCA_SOF_EOF from head */ skb_pull(skb, 1); From c20f4650e5efe36ccd69063c8b19587a619e2342 Mon Sep 17 00:00:00 2001 From: Aldo Ariel Panzardo Date: Thu, 16 Jul 2026 20:26:57 -0300 Subject: [PATCH 346/857] nfc: llcp: Fix list corruption / refcount desync in nfc_llcp_recv_dm() nfc_llcp_recv_dm() handles DM(NOBOUND)/DM(REJ) for a socket that is still linked on local->connecting_sockets: it looks the socket up with nfc_llcp_connecting_sock_get(), sets sk->sk_state = LLCP_CLOSED and returns, without taking the socket lock and without unlinking the socket from the connecting_sockets list. llcp_sock_release() selects the list to unlink from by sk_state: a socket in LLCP_CONNECTING is unlinked from connecting_sockets, otherwise from the sockets list. Because recv_dm left the socket physically on connecting_sockets but in the LLCP_CLOSED state, release() takes the else branch and calls nfc_llcp_sock_unlink(&local->sockets, sk). That runs sk_del_node_init() while holding sockets.lock, i.e. it removes the socket from the connecting_sockets hlist under the wrong lock. A concurrent connect() linking another socket onto connecting_sockets under connecting_sockets.lock then mutates the same hlist unserialized, which corrupts the list and desyncs the sk_add_node()/sk_del_node_init() sock_hold()/__sock_put() pairing. An unprivileged local process holding LLCP sockets, with the DM supplied by the remote peer over an established LLCP link, can drive this to leak kernel sockets without bound (the mis-decrement goes through the non-freeing __sock_put() path, so the object is never released), leading to memory exhaustion / DoS. This is the same class of bug that was fixed in the sibling handler nfc_llcp_recv_cc() by commit b493ea2765cc ("nfc: llcp: Fix use-after-free race in nfc_llcp_recv_cc()"); recv_dm did not receive the equivalent fix. Fix it the same way: take lock_sock(), re-check that the socket is still hashed (release() may have won the race), and for the NOBOUND/REJ case unlink it from connecting_sockets before moving it to LLCP_CLOSED. The unlink drops the connecting_sockets membership reference via sk_del_node_init(), leaving the socket unhashed, so the later nfc_llcp_sock_unlink() in llcp_sock_release() becomes a no-op and no double put occurs. Fixes: a69f32af86e3 ("NFC: Socket linked list") Signed-off-by: Aldo Ariel Panzardo Link: https://patch.msgid.link/20260716232657.203145-1-qwe.aldo@gmail.com Signed-off-by: David Heidelberg --- net/nfc/llcp_core.c | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/net/nfc/llcp_core.c b/net/nfc/llcp_core.c index cac1b5487064d0..bd6361e2efa4f1 100644 --- a/net/nfc/llcp_core.c +++ b/net/nfc/llcp_core.c @@ -1251,6 +1251,7 @@ static void nfc_llcp_recv_dm(struct nfc_llcp_local *local, struct nfc_llcp_sock *llcp_sock; struct sock *sk; u8 dsap, ssap, reason; + bool connecting = false; dsap = nfc_llcp_dsap(skb); ssap = nfc_llcp_ssap(skb); @@ -1262,6 +1263,7 @@ static void nfc_llcp_recv_dm(struct nfc_llcp_local *local, case LLCP_DM_NOBOUND: case LLCP_DM_REJ: llcp_sock = nfc_llcp_connecting_sock_get(local, dsap); + connecting = true; break; default: @@ -1276,10 +1278,33 @@ static void nfc_llcp_recv_dm(struct nfc_llcp_local *local, sk = &llcp_sock->sk; + lock_sock(sk); + + /* Check if socket was destroyed whilst waiting for the lock */ + if (!sk_hashed(sk)) { + release_sock(sk); + nfc_llcp_sock_put(llcp_sock); + return; + } + + /* + * For DM(NOBOUND)/DM(REJ) the socket is still linked on the + * connecting_sockets list. Unlink it here, under the socket lock, + * before moving it to LLCP_CLOSED: llcp_sock_release() selects the + * list to unlink from by sk_state, so leaving a connecting socket + * in the CLOSED state would make it unlink from the wrong list and + * corrupt the connecting_sockets list / desync the socket refcount. + * This mirrors nfc_llcp_recv_cc(). + */ + if (connecting) + nfc_llcp_sock_unlink(&local->connecting_sockets, sk); + sk->sk_err = ENXIO; sk->sk_state = LLCP_CLOSED; sk->sk_state_change(sk); + release_sock(sk); + nfc_llcp_sock_put(llcp_sock); } From a4916d08317de264d150b8c041d7fef751ed2d7d Mon Sep 17 00:00:00 2001 From: Doruk Tan Ozturk Date: Sat, 11 Jul 2026 14:36:51 +0200 Subject: [PATCH 347/857] nfc: port100: reject frames whose declared length exceeds the received data port100_recv_response() passes the URB transfer buffer to port100_rx_frame_is_valid(), which checksums le16_to_cpu(frame->datalen) bytes of frame->data. datalen is a 16-bit field supplied by the device and is never checked against the number of bytes actually received (urb->actual_length), so a device reporting a datalen larger than the received frame makes port100_data_checksum() read out of bounds past the transfer buffer. Reject a response whose declared frame size does not fit the received length before validating it. Found by 0sec (https://0sec.ai) using automated source analysis; the missing bound is evident from source. Compile-tested. Fixes: 562d4d59b8a1 ("NFC: Sony Port-100 Series driver") Cc: stable@vger.kernel.org Assisted-by: 0sec:claude-opus-4-8 Signed-off-by: Doruk Tan Ozturk Reviewed-by: Simon Horman Link: https://patch.msgid.link/20260711123651.32595-1-doruk@0sec.ai Signed-off-by: David Heidelberg --- drivers/nfc/port100.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/drivers/nfc/port100.c b/drivers/nfc/port100.c index b613f5e2fd57a4..e769a30b8b5c7e 100644 --- a/drivers/nfc/port100.c +++ b/drivers/nfc/port100.c @@ -636,6 +636,13 @@ static void port100_recv_response(struct urb *urb) in_frame = dev->in_urb->transfer_buffer; + if (urb->actual_length < PORT100_FRAME_HEADER_LEN || + urb->actual_length < port100_rx_frame_size(in_frame)) { + nfc_err(&dev->interface->dev, "Received a truncated frame\n"); + cmd->status = -EIO; + goto sched_wq; + } + if (!port100_rx_frame_is_valid(in_frame)) { nfc_err(&dev->interface->dev, "Received an invalid frame\n"); cmd->status = -EIO; From 50a0fd24027765eada2532ca27d2056cfe304711 Mon Sep 17 00:00:00 2001 From: Lei Zhu Date: Wed, 29 Jul 2026 15:24:26 +0800 Subject: [PATCH 348/857] selftests: nci: Correct pthread_create return value check The pthread_create() functions returns 0 on success and a positive value on failure. Modify the return value check to correctly detect failure cases. Fixes: 72696bd8a09d ("selftests: nci: Extract the start/stop discovery function") Signed-off-by: Lei Zhu Link: https://patch.msgid.link/20260729072426.303484-1-zhulei_szu@163.com Signed-off-by: David Heidelberg --- tools/testing/selftests/nci/nci_dev.c | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/tools/testing/selftests/nci/nci_dev.c b/tools/testing/selftests/nci/nci_dev.c index 312f84ee0444fd..c053f5cf2745d7 100644 --- a/tools/testing/selftests/nci/nci_dev.c +++ b/tools/testing/selftests/nci/nci_dev.c @@ -438,7 +438,7 @@ FIXTURE_SETUP(NCI) else rc = pthread_create(&thread_t, NULL, virtual_dev_open, (void *)&self->virtual_nci_fd); - ASSERT_GT(rc, -1); + ASSERT_EQ(rc, 0); rc = send_cmd_with_idx(self->sd, self->fid, self->pid, NFC_CMD_DEV_UP, self->dev_idex); @@ -509,7 +509,7 @@ FIXTURE_TEARDOWN(NCI) rc = pthread_create(&thread_t, NULL, virtual_deinit, (void *)&self->virtual_nci_fd); - ASSERT_GT(rc, -1); + ASSERT_EQ(rc, 0); rc = send_cmd_with_idx(self->sd, self->fid, self->pid, NFC_CMD_DEV_DOWN, self->dev_idex); EXPECT_EQ(rc, 0); @@ -590,7 +590,7 @@ int start_polling(int dev_idx, int proto, int virtual_fd, int sd, int fid, int p rc = pthread_create(&thread_t, NULL, virtual_poll_start, (void *)&virtual_fd); - if (rc < 0) + if (rc) return rc; rc = send_cmd_mt_nla(sd, fid, pid, NFC_CMD_START_POLL, 2, nla_start_poll_type, @@ -610,7 +610,7 @@ int stop_polling(int dev_idx, int virtual_fd, int sd, int fid, int pid) rc = pthread_create(&thread_t, NULL, virtual_poll_stop, (void *)&virtual_fd); - if (rc < 0) + if (rc) return rc; rc = send_cmd_with_idx(sd, fid, pid, @@ -830,6 +830,8 @@ int disconnect_tag(int nfc_sock, int virtual_fd) status = pthread_create(&thread_t, NULL, virtual_deactivate_proc, (void *)&virtual_fd); + if (status) + return status; close(nfc_sock); pthread_join(thread_t, (void **)&status); @@ -874,7 +876,7 @@ TEST_F(NCI, deinit) else rc = pthread_create(&thread_t, NULL, virtual_deinit, (void *)&self->virtual_nci_fd); - ASSERT_GT(rc, -1); + ASSERT_EQ(rc, 0); rc = send_cmd_with_idx(self->sd, self->fid, self->pid, NFC_CMD_DEV_DOWN, self->dev_idex); From a9498e13c18ae1eec2cc8c6771f9b5beb918669c Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Sat, 8 Aug 2026 14:13:55 +0800 Subject: [PATCH 349/857] btrfs: free unlinked replace target on initialization failure btrfs_init_dev_replace_tgtdev() allocates the replacement target before looking up its dev_t and initializing its zoned device information. If either lookup_bdev() or btrfs_get_dev_zone_info() fails, the device has not been linked into fs_devices->devices yet, but the error path only drops the block device file reference. Free the allocated device on this error path to release its name, allocation state, zone info, and the device itself. The issue was found by a failure-path metadata residual analyzer and verified with targeted failure injection on v6.14. Assisted-by: Codex:gpt-5 Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/dev-replace.c | 10 +++++++--- fs/btrfs/volumes.c | 2 +- fs/btrfs/volumes.h | 1 + 3 files changed, 9 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c index dc0834f920c3ba..43e46aea81fb01 100644 --- a/fs/btrfs/dev-replace.c +++ b/fs/btrfs/dev-replace.c @@ -235,7 +235,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, struct btrfs_device **device_out) { struct btrfs_fs_devices *fs_devices = fs_info->fs_devices; - struct btrfs_device *device; + struct btrfs_device *device = NULL; + struct btrfs_device *tmp_device; struct file *bdev_file; struct block_device *bdev; u64 devid = BTRFS_DEV_REPLACE_DEVID; @@ -264,8 +265,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, sync_blockdev(bdev); - list_for_each_entry(device, &fs_devices->devices, dev_list) { - if (device->bdev == bdev) { + list_for_each_entry(tmp_device, &fs_devices->devices, dev_list) { + if (tmp_device->bdev == bdev) { btrfs_err(fs_info, "target device is in the filesystem!"); ret = -EEXIST; @@ -285,6 +286,7 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, device = btrfs_alloc_device(NULL, &devid, NULL, device_path); if (IS_ERR(device)) { ret = PTR_ERR(device); + device = NULL; goto error; } @@ -328,6 +330,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, error: /* Undo the open-time freeze deny. */ + if (device) + btrfs_free_device(device); btrfs_release_device_allow_freeze(bdev_file); return ret; } diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 9b66eb584ecef0..50751fcdeff034 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -403,7 +403,7 @@ static struct btrfs_fs_devices *alloc_fs_devices(const u8 *fsid) return fs_devs; } -static void btrfs_free_device(struct btrfs_device *device) +void btrfs_free_device(struct btrfs_device *device) { WARN_ON(!list_empty(&device->post_commit_list)); /* diff --git a/fs/btrfs/volumes.h b/fs/btrfs/volumes.h index 0415d74cad9ba9..337d7007d9e225 100644 --- a/fs/btrfs/volumes.h +++ b/fs/btrfs/volumes.h @@ -799,6 +799,7 @@ void btrfs_rm_dev_replace_remove_srcdev(struct btrfs_device *srcdev); void btrfs_rm_dev_replace_free_srcdev(struct btrfs_device *srcdev); void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev, bool allow_freeze); +void btrfs_free_device(struct btrfs_device *device); unsigned long btrfs_full_stripe_len(struct btrfs_fs_info *fs_info, u64 logical); u64 btrfs_calc_stripe_length(const struct btrfs_chunk_map *map); From 505ab5430f8d6d63d4ca1b11e909054886aa4c98 Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Sat, 8 Aug 2026 14:38:32 +0800 Subject: [PATCH 350/857] btrfs: clean up target device if block group marking fails btrfs_dev_replace_start() adds the replacement target to the device list before marking block groups to copy. If marking fails, returning directly leaves the target linked and keeps the device accounting incremented. Jump to the existing cleanup path so the target device is removed and released on failure. The issue was found by a failure-path metadata residual analyzer and verified with targeted failure injection on v6.14. Assisted-by: Codex:gpt-5 Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/dev-replace.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c index 43e46aea81fb01..22eea188a32808 100644 --- a/fs/btrfs/dev-replace.c +++ b/fs/btrfs/dev-replace.c @@ -640,7 +640,7 @@ static int btrfs_dev_replace_start(struct btrfs_fs_info *fs_info, ret = mark_block_group_to_copy(fs_info, src_device); if (ret) - return ret; + goto leave; down_write(&dev_replace->rwsem); dev_replace->replace_task = current; From 7ce223d7c635791ff6a3c792de88bdaf7dc59c1e Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Mon, 10 Aug 2026 20:16:04 +0800 Subject: [PATCH 351/857] btrfs: detach failed sprout device from transaction update list When creating the first metadata chunk for a sprout filesystem, create_chunk() adds the new device to the transaction dev_update_list through device->post_commit_list. If the subsequent system chunk creation fails, btrfs_init_new_device() aborts the transaction and releases the device while post_commit_list is still linked. This triggers a warning in btrfs_free_device() and leaves the transaction list referencing freed memory. Detach the device while holding chunk_mutex before releasing it. Fixes: bbbf7243d62d ("btrfs: combine device update operations during transaction commit") Assisted-by: Codex:gpt-5 Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/volumes.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 50751fcdeff034..869a43ccfd8b81 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -3118,6 +3118,8 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path btrfs_sysfs_remove_device(device); mutex_lock(&fs_info->fs_devices->device_list_mutex); mutex_lock(&fs_info->chunk_mutex); + if (!list_empty(&device->post_commit_list)) + list_del_init(&device->post_commit_list); list_del_rcu(&device->dev_list); list_del(&device->dev_alloc_list); fs_info->fs_devices->num_devices--; From a4fba7b47261b28e4c85667b9934d99dd3d27143 Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Mon, 10 Aug 2026 20:16:05 +0800 Subject: [PATCH 352/857] btrfs: restore active device pointers after failed sprout btrfs_init_new_device() switches latest_dev and possibly s_bdev from the seed device to the new sprout device before creating the first writable chunks. If chunk creation or the subsequent sprout setup fails, the error path releases the new device without switching those pointers back. btrfs_show_devname() can then dereference the freed latest_dev and crash. Restore the active device pointers to the latest seed device before removing and releasing the failed sprout device. Fixes: b7cb29e666fe ("btrfs: update latest_dev when we create a sprout device") Assisted-by: Codex:gpt-5 Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/volumes.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 869a43ccfd8b81..7fb0bf742a2969 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -3117,6 +3117,8 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path error_sysfs: btrfs_sysfs_remove_device(device); mutex_lock(&fs_info->fs_devices->device_list_mutex); + if (seeding_dev) + btrfs_assign_next_active_device(device, seed_devices->latest_dev); mutex_lock(&fs_info->chunk_mutex); if (!list_empty(&device->post_commit_list)) list_del_init(&device->post_commit_list); From c41a2394a4b3585ab377803ad4235be1ed1c12d2 Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Tue, 11 Aug 2026 09:02:27 +0930 Subject: [PATCH 353/857] btrfs: roll back sprout setup after device add failure btrfs_init_new_device() calls btrfs_setup_sprout() before creating the first writable chunks for a seed filesystem. That moves the seed devices out of fs_info->fs_devices, clears the seeding state and installs a new fsid for the sprout filesystem. If a later step fails, the error path removes the new device but leaves fs_info->fs_devices in the partially initialized sprout state. The mounted filesystem can then be left with no open devices after the failed device add. Add the inverse of btrfs_setup_sprout() and use it from the error path so the mounted seed filesystem is restored before the temporary seed_devices copy is released. Fixes: 2b82032c34ec ("Btrfs: Seed device support") Assisted-by: Codex:gpt-5 Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> [ Fix a conflict with per-profile available space, revert sprout before updating per-profile available space estimation. ] Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/volumes.c | 37 +++++++++++++++++++++++++++++++++++++ 1 file changed, 37 insertions(+) diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 7fb0bf742a2969..949e40baff3343 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -2783,6 +2783,41 @@ static void btrfs_setup_sprout(struct btrfs_fs_info *fs_info, btrfs_set_super_flags(disk_super, super_flags); } +static void btrfs_rollback_sprout(struct btrfs_fs_info *fs_info, + struct btrfs_fs_devices *seed_devices) +{ + struct btrfs_fs_devices *fs_devices = fs_info->fs_devices; + struct btrfs_super_block *disk_super = fs_info->super_copy; + struct btrfs_device *device; + u64 super_flags; + + lockdep_assert_held(&uuid_mutex); + lockdep_assert_held(&fs_devices->device_list_mutex); + + list_del_init(&seed_devices->seed_list); + list_splice_init_rcu(&seed_devices->devices, &fs_devices->devices, synchronize_rcu); + list_for_each_entry(device, &fs_devices->devices, dev_list) { + device->fs_devices = fs_devices; + } + + fs_devices->seeding = true; + fs_devices->num_devices = seed_devices->num_devices; + fs_devices->open_devices = seed_devices->open_devices; + fs_devices->missing_devices = seed_devices->missing_devices; + fs_devices->rotating = seed_devices->rotating; + fs_devices->latest_dev = seed_devices->latest_dev; + + memcpy(fs_devices->fsid, seed_devices->fsid, BTRFS_FSID_SIZE); + memcpy(fs_devices->metadata_uuid, seed_devices->metadata_uuid, BTRFS_FSID_SIZE); + memcpy(disk_super->fsid, seed_devices->fsid, BTRFS_FSID_SIZE); + + super_flags = (btrfs_super_flags(disk_super) | BTRFS_SUPER_FLAG_SEEDING); + btrfs_set_super_flags(disk_super, super_flags); + + seed_devices->opened = 0; + free_fs_devices(seed_devices); +} + /* * Store the expected generation for seed devices in device items. */ @@ -3134,6 +3169,8 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path orig_super_total_bytes); btrfs_set_super_num_devices(fs_info->super_copy, orig_super_num_devices); + if (seeding_dev) + btrfs_rollback_sprout(fs_info, seed_devices); btrfs_update_per_profile_avail(fs_info); mutex_unlock(&fs_info->chunk_mutex); mutex_unlock(&fs_info->fs_devices->device_list_mutex); From 1450a68f30a163d5cc2b1f1009797b357bec33e3 Mon Sep 17 00:00:00 2001 From: Dongfang Zhao Date: Thu, 30 Jul 2026 22:53:49 -0700 Subject: [PATCH 354/857] clk: qcom: Add support for Hawi GPUCC Add the graphics clock controller driver for Hawi. This provides the clocks, resets and power domains required by the GPU driver to enable the graphics subsystem. Reuse the existing gxclkctl-kaanapali driver for GX clock controller support and build it with the Hawi GPUCC driver. Signed-off-by: Dongfang Zhao Reviewed-by: Konrad Dybcio Reviewed-by: Taniya Das Link: https://lore.kernel.org/r/20260730-gpucc-hawi-v2-2-7bf618ab8a34@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/Kconfig | 12 + drivers/clk/qcom/Makefile | 1 + drivers/clk/qcom/gpucc-hawi.c | 461 ++++++++++++++++++++++++++++++++++ 3 files changed, 474 insertions(+) create mode 100644 drivers/clk/qcom/gpucc-hawi.c diff --git a/drivers/clk/qcom/Kconfig b/drivers/clk/qcom/Kconfig index 330afe1a283136..aba035b8932209 100644 --- a/drivers/clk/qcom/Kconfig +++ b/drivers/clk/qcom/Kconfig @@ -472,6 +472,18 @@ config CLK_HAWI_GCC Say Y if you want to use peripheral devices such as UART, SPI, I2C, USB, UFS, SD/eMMC, PCIe, etc. +config CLK_HAWI_GPUCC + tristate "Hawi Graphics Clock Controller" + depends on ARM64 || COMPILE_TEST + select CLK_HAWI_GCC + default m if ARCH_QCOM + help + Support for the graphics clock controller on Hawi devices. + Say Y if you want to support graphics controller devices and + functionality such as 3D graphics. It also provides GX clock + domain control required for GPU power management on Hawi-based + devices. + config CLK_HAWI_TCSRCC tristate "Hawi TCSR Clock Controller" depends on ARM64 || COMPILE_TEST diff --git a/drivers/clk/qcom/Makefile b/drivers/clk/qcom/Makefile index 3a0976c038d58b..e867cf0fdd67ea 100644 --- a/drivers/clk/qcom/Makefile +++ b/drivers/clk/qcom/Makefile @@ -36,6 +36,7 @@ obj-$(CONFIG_CLK_GLYMUR_GPUCC) += gpucc-glymur.o gxclkctl-kaanapali.o obj-$(CONFIG_CLK_GLYMUR_TCSRCC) += tcsrcc-glymur.o obj-$(CONFIG_CLK_GLYMUR_VIDEOCC) += videocc-glymur.o obj-$(CONFIG_CLK_HAWI_GCC) += gcc-hawi.o +obj-$(CONFIG_CLK_HAWI_GPUCC) += gpucc-hawi.o gxclkctl-kaanapali.o obj-$(CONFIG_CLK_HAWI_TCSRCC) += tcsrcc-hawi.o obj-$(CONFIG_CLK_HAWI_VIDEOCC) += videocc-hawi.o obj-$(CONFIG_CLK_KAANAPALI_CAMCC) += cambistmclkcc-kaanapali.o camcc-kaanapali.o diff --git a/drivers/clk/qcom/gpucc-hawi.c b/drivers/clk/qcom/gpucc-hawi.c new file mode 100644 index 00000000000000..30b3ffb0638fb7 --- /dev/null +++ b/drivers/clk/qcom/gpucc-hawi.c @@ -0,0 +1,461 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) Qualcomm Technologies, Inc. and/or its subsidiaries. + */ + +#include +#include +#include +#include + +#include + +#include "clk-alpha-pll.h" +#include "clk-branch.h" +#include "clk-rcg.h" +#include "clk-regmap.h" +#include "clk-regmap-divider.h" +#include "common.h" +#include "gdsc.h" +#include "reset.h" + +enum { + DT_BI_TCXO, + DT_GPLL0_OUT_MAIN, + DT_GPLL0_OUT_MAIN_DIV, +}; + +enum { + P_BI_TCXO, + P_GPLL0_OUT_MAIN, + P_GPLL0_OUT_MAIN_DIV, + P_GPU_CC_PLL0_OUT_EVEN, + P_GPU_CC_PLL0_OUT_MAIN, + P_GPU_CC_PLL0_OUT_ODD, +}; + +static const struct pll_vco taycan_eha_t_vco[] = { + { 249600000, 2500000000, 0 }, +}; + +/* 788.0 MHz Configuration */ +static const struct alpha_pll_config gpu_cc_pll0_config = { + .l = 0x29, + .alpha = 0xaaa, + .config_ctl_val = 0xa5c400e7, + .config_ctl_hi_val = 0x0a8060e0, + .config_ctl_hi1_val = 0xf51dea20, + .user_ctl_val = 0x00000000, + .user_ctl_hi_val = 0x00000002, +}; + +static struct clk_alpha_pll gpu_cc_pll0 = { + .offset = 0x0, + .config = &gpu_cc_pll0_config, + .vco_table = taycan_eha_t_vco, + .num_vco = ARRAY_SIZE(taycan_eha_t_vco), + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_TAYCAN_EHA_T], + .clkr = { + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_pll0", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_taycan_eha_t_ops, + }, + }, +}; + +static const struct parent_map gpu_cc_parent_map_1[] = { + { P_BI_TCXO, 0 }, + { P_GPU_CC_PLL0_OUT_MAIN, 1 }, + { P_GPU_CC_PLL0_OUT_EVEN, 2 }, + { P_GPU_CC_PLL0_OUT_ODD, 3 }, + { P_GPLL0_OUT_MAIN, 5 }, + { P_GPLL0_OUT_MAIN_DIV, 6 }, +}; + +static const struct clk_parent_data gpu_cc_parent_data_1[] = { + { .index = DT_BI_TCXO }, + { .hw = &gpu_cc_pll0.clkr.hw }, + { .hw = &gpu_cc_pll0.clkr.hw }, + { .hw = &gpu_cc_pll0.clkr.hw }, + { .index = DT_GPLL0_OUT_MAIN }, + { .index = DT_GPLL0_OUT_MAIN_DIV }, +}; + +static const struct freq_tbl ftbl_gpu_cc_gmu_clk_src[] = { + F(19200000, P_BI_TCXO, 1, 0, 0), + F(788000000, P_GPU_CC_PLL0_OUT_MAIN, 1, 0, 0), + F(825000000, P_GPU_CC_PLL0_OUT_MAIN, 1, 0, 0), + F(880000000, P_GPU_CC_PLL0_OUT_MAIN, 1, 0, 0), + F(917000000, P_GPU_CC_PLL0_OUT_MAIN, 1, 0, 0), + { } +}; + +static struct clk_rcg2 gpu_cc_gmu_clk_src = { + .cmd_rcgr = 0x9318, + .mnd_width = 0, + .hid_width = 5, + .parent_map = gpu_cc_parent_map_1, + .freq_tbl = ftbl_gpu_cc_gmu_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_gmu_clk_src", + .parent_data = gpu_cc_parent_data_1, + .num_parents = ARRAY_SIZE(gpu_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_gpu_cc_hub_clk_src[] = { + F(150000000, P_GPLL0_OUT_MAIN_DIV, 2, 0, 0), + F(200000000, P_GPLL0_OUT_MAIN, 3, 0, 0), + F(300000000, P_GPLL0_OUT_MAIN, 2, 0, 0), + F(400000000, P_GPLL0_OUT_MAIN, 1.5, 0, 0), + { } +}; + +static struct clk_rcg2 gpu_cc_hub_clk_src = { + .cmd_rcgr = 0x93f0, + .mnd_width = 0, + .hid_width = 5, + .parent_map = gpu_cc_parent_map_1, + .freq_tbl = ftbl_gpu_cc_hub_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_hub_clk_src", + .parent_data = gpu_cc_parent_data_1, + .num_parents = ARRAY_SIZE(gpu_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_regmap_div gpu_cc_hub_div_clk_src = { + .reg = 0x9430, + .shift = 0, + .width = 4, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_hub_div_clk_src", + .parent_hws = (const struct clk_hw *[]) { + &gpu_cc_hub_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_regmap_div_ro_ops, + }, +}; + +static struct clk_branch gpu_cc_ahb_clk = { + .halt_reg = 0x90bc, + .halt_check = BRANCH_HALT_DELAY, + .clkr = { + .enable_reg = 0x90bc, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_ahb_clk", + .parent_hws = (const struct clk_hw *[]) { + &gpu_cc_hub_div_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_cx_accu_shift_clk = { + .halt_reg = 0x9104, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x9104, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_cx_accu_shift_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_cx_gmu_clk = { + .halt_reg = 0x90d4, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x90d4, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x90d4, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_cx_gmu_clk", + .parent_hws = (const struct clk_hw *[]) { + &gpu_cc_gmu_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_aon_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_demet_clk = { + .halt_reg = 0x9010, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x9010, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_demet_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_dpm_clk = { + .halt_reg = 0x9108, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x9108, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_dpm_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_freq_measure_clk = { + .halt_reg = 0x900c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x900c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_freq_measure_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_gpu_smmu_vote_clk = { + .halt_reg = 0x7000, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x7000, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_gpu_smmu_vote_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_gx_accu_shift_clk = { + .halt_reg = 0x9070, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x9070, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_gx_accu_shift_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_hub_aon_clk = { + .halt_reg = 0x93ec, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x93ec, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_hub_aon_clk", + .parent_hws = (const struct clk_hw *[]) { + &gpu_cc_hub_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_aon_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_hub_cx_int_clk = { + .halt_reg = 0x90e8, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x90e8, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x90e8, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_hub_cx_int_clk", + .parent_hws = (const struct clk_hw *[]) { + &gpu_cc_hub_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_aon_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_memnoc_gfx_clk = { + .halt_reg = 0x90ec, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x90ec, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x90ec, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_memnoc_gfx_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gpu_cc_mxg_ahb_clk = { + .halt_reg = 0x9744, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x9744, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpu_cc_mxg_ahb_clk", + .parent_hws = (const struct clk_hw *[]) { + &gpu_cc_hub_div_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct gdsc gpu_cc_cx_gdsc = { + .gdscr = 0x9080, + .gds_hw_ctrl = 0x9094, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "gpu_cc_cx_gdsc", + }, + .pwrsts = PWRSTS_OFF_ON, + .flags = VOTABLE | POLL_CFG_GDSCR | RETAIN_FF_ENABLE, +}; + +static struct gdsc gpu_cc_mxg_gdsc = { + .gdscr = 0x961c, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "gpu_cc_mxg_gdsc", + }, + .pwrsts = PWRSTS_OFF_ON, + .flags = POLL_CFG_GDSCR | RETAIN_FF_ENABLE, +}; + +static struct clk_regmap *gpu_cc_hawi_clocks[] = { + [GPU_CC_AHB_CLK] = &gpu_cc_ahb_clk.clkr, + [GPU_CC_CX_ACCU_SHIFT_CLK] = &gpu_cc_cx_accu_shift_clk.clkr, + [GPU_CC_CX_GMU_CLK] = &gpu_cc_cx_gmu_clk.clkr, + [GPU_CC_DEMET_CLK] = &gpu_cc_demet_clk.clkr, + [GPU_CC_DPM_CLK] = &gpu_cc_dpm_clk.clkr, + [GPU_CC_FREQ_MEASURE_CLK] = &gpu_cc_freq_measure_clk.clkr, + [GPU_CC_GMU_CLK_SRC] = &gpu_cc_gmu_clk_src.clkr, + [GPU_CC_GPU_SMMU_VOTE_CLK] = &gpu_cc_gpu_smmu_vote_clk.clkr, + [GPU_CC_GX_ACCU_SHIFT_CLK] = &gpu_cc_gx_accu_shift_clk.clkr, + [GPU_CC_HUB_AON_CLK] = &gpu_cc_hub_aon_clk.clkr, + [GPU_CC_HUB_CLK_SRC] = &gpu_cc_hub_clk_src.clkr, + [GPU_CC_HUB_CX_INT_CLK] = &gpu_cc_hub_cx_int_clk.clkr, + [GPU_CC_HUB_DIV_CLK_SRC] = &gpu_cc_hub_div_clk_src.clkr, + [GPU_CC_MEMNOC_GFX_CLK] = &gpu_cc_memnoc_gfx_clk.clkr, + [GPU_CC_MXG_AHB_CLK] = &gpu_cc_mxg_ahb_clk.clkr, + [GPU_CC_PLL0] = &gpu_cc_pll0.clkr, +}; + +static struct gdsc *gpu_cc_hawi_gdscs[] = { + [GPU_CC_CX_GDSC] = &gpu_cc_cx_gdsc, + [GPU_CC_MXG_GDSC] = &gpu_cc_mxg_gdsc, +}; + +static const struct qcom_reset_map gpu_cc_hawi_resets[] = { + [GPU_CC_CB_BCR] = { 0x93a0 }, + [GPU_CC_CX_BCR] = { 0x907c }, + [GPU_CC_FAST_HUB_BCR] = { 0x93e4 }, + [GPU_CC_GMU_BCR] = { 0x9314 }, + [GPU_CC_GX_BCR] = { 0x905c }, + [GPU_CC_MXG_BCR] = { 0x9618 }, + [GPU_CC_XO_BCR] = { 0x9000 }, +}; + +static struct clk_alpha_pll *gpu_cc_hawi_plls[] = { + &gpu_cc_pll0, +}; + +static u32 gpu_cc_hawi_critical_cbcrs[] = { + 0x93a4, /* GPU_CC_CB_CLK */ + 0x9704, /* GPU_CC_MXG_XO_CLK */ + 0x90e4, /* GPU_CC_CXO_CLK */ + 0x9008, /* GPU_CC_CXO_AON_CLK */ + 0x93e8, /* GPU_CC_RSCC_HUB_AON_CLK */ + 0x9004, /* GPU_CC_RSCC_XO_AON_CLK */ + 0x90cc, /* GPU_CC_SLEEP_CLK */ +}; + +static const struct regmap_config gpu_cc_hawi_regmap_config = { + .reg_bits = 32, + .reg_stride = 4, + .val_bits = 32, + .max_register = 0x9744, + .fast_io = true, +}; + +static struct qcom_cc_driver_data gpu_cc_hawi_driver_data = { + .alpha_plls = gpu_cc_hawi_plls, + .num_alpha_plls = ARRAY_SIZE(gpu_cc_hawi_plls), + .clk_cbcrs = gpu_cc_hawi_critical_cbcrs, + .num_clk_cbcrs = ARRAY_SIZE(gpu_cc_hawi_critical_cbcrs), +}; + +static const struct qcom_cc_desc gpu_cc_hawi_desc = { + .config = &gpu_cc_hawi_regmap_config, + .clks = gpu_cc_hawi_clocks, + .num_clks = ARRAY_SIZE(gpu_cc_hawi_clocks), + .resets = gpu_cc_hawi_resets, + .num_resets = ARRAY_SIZE(gpu_cc_hawi_resets), + .gdscs = gpu_cc_hawi_gdscs, + .num_gdscs = ARRAY_SIZE(gpu_cc_hawi_gdscs), + .use_rpm = true, + .driver_data = &gpu_cc_hawi_driver_data, +}; + +static const struct of_device_id gpu_cc_hawi_match_table[] = { + { .compatible = "qcom,hawi-gpucc" }, + { } +}; +MODULE_DEVICE_TABLE(of, gpu_cc_hawi_match_table); + +static int gpu_cc_hawi_probe(struct platform_device *pdev) +{ + return qcom_cc_probe(pdev, &gpu_cc_hawi_desc); +} + +static struct platform_driver gpu_cc_hawi_driver = { + .probe = gpu_cc_hawi_probe, + .driver = { + .name = "gpucc-hawi", + .of_match_table = gpu_cc_hawi_match_table, + }, +}; + +module_platform_driver(gpu_cc_hawi_driver); + +MODULE_DESCRIPTION("QTI GPUCC HAWI Driver"); +MODULE_LICENSE("GPL"); From 74b41b78aabfaf021ca4ca2aebae75fdb538b9a9 Mon Sep 17 00:00:00 2001 From: Jagadeesh Kona Date: Sat, 1 Aug 2026 00:42:45 +0530 Subject: [PATCH 355/857] dt-bindings: clock: qcom: Add Maili Graphics Clock Controllers Maili gpucc is a derivative of Hawi gpucc with pll0 configuration change, and hence reuses the Hawi gpucc driver with Maili fixup. Maili gxclkctl is the same as Kaanapali gxclkctl and falls back to it. Add the compatibles for the graphics clock controllers on the Qualcomm Maili SoC. Signed-off-by: Jagadeesh Kona Acked-by: Krzysztof Kozlowski Link: https://lore.kernel.org/r/20260801-maili_gpucc-v1-1-9d8c37bded69@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- Documentation/devicetree/bindings/clock/qcom,hawi-gpucc.yaml | 4 +++- .../devicetree/bindings/clock/qcom,kaanapali-gxclkctl.yaml | 1 + 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/Documentation/devicetree/bindings/clock/qcom,hawi-gpucc.yaml b/Documentation/devicetree/bindings/clock/qcom,hawi-gpucc.yaml index 3e40bc3c061d11..b7179d53166f8c 100644 --- a/Documentation/devicetree/bindings/clock/qcom,hawi-gpucc.yaml +++ b/Documentation/devicetree/bindings/clock/qcom,hawi-gpucc.yaml @@ -17,7 +17,9 @@ description: | properties: compatible: - const: qcom,hawi-gpucc + enum: + - qcom,hawi-gpucc + - qcom,maili-gpucc clocks: items: diff --git a/Documentation/devicetree/bindings/clock/qcom,kaanapali-gxclkctl.yaml b/Documentation/devicetree/bindings/clock/qcom,kaanapali-gxclkctl.yaml index 28b653f6b64d5c..94f84d5294c6df 100644 --- a/Documentation/devicetree/bindings/clock/qcom,kaanapali-gxclkctl.yaml +++ b/Documentation/devicetree/bindings/clock/qcom,kaanapali-gxclkctl.yaml @@ -28,6 +28,7 @@ properties: - items: - enum: - qcom,hawi-gxclkctl + - qcom,maili-gxclkctl - const: qcom,kaanapali-gxclkctl power-domains: From bc1cff907773a5d00475efa860d18def4f5f3b79 Mon Sep 17 00:00:00 2001 From: Jagadeesh Kona Date: Sat, 1 Aug 2026 00:42:46 +0530 Subject: [PATCH 356/857] clk: qcom: Add Graphics clock controller support on Qualcomm Maili SoC Maili gpucc is a derivative of Hawi gpucc with pll0 configuration change. Hence, reuse the Hawi graphics clock controller and extend it for Qualcomm Maili SoC. Signed-off-by: Jagadeesh Kona Reviewed-by: Abel Vesa Reviewed-by: Taniya Das Link: https://lore.kernel.org/r/20260801-maili_gpucc-v1-2-9d8c37bded69@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/gpucc-hawi.c | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/drivers/clk/qcom/gpucc-hawi.c b/drivers/clk/qcom/gpucc-hawi.c index 30b3ffb0638fb7..3deb7edf7810de 100644 --- a/drivers/clk/qcom/gpucc-hawi.c +++ b/drivers/clk/qcom/gpucc-hawi.c @@ -49,6 +49,18 @@ static const struct alpha_pll_config gpu_cc_pll0_config = { .user_ctl_hi_val = 0x00000002, }; +/* 788.0 MHz Configuration */ +static const struct alpha_pll_config gpu_cc_pll0_config_maili = { + .l = 0x29, + .cal_l = 0x42, + .alpha = 0xaaa, + .config_ctl_val = 0xa5c400e7, + .config_ctl_hi_val = 0x0a806160, + .config_ctl_hi1_val = 0xf51dea20, + .user_ctl_val = 0x00000000, + .user_ctl_hi_val = 0x00000002, +}; + static struct clk_alpha_pll gpu_cc_pll0 = { .offset = 0x0, .config = &gpu_cc_pll0_config, @@ -438,12 +450,16 @@ static const struct qcom_cc_desc gpu_cc_hawi_desc = { static const struct of_device_id gpu_cc_hawi_match_table[] = { { .compatible = "qcom,hawi-gpucc" }, + { .compatible = "qcom,maili-gpucc" }, { } }; MODULE_DEVICE_TABLE(of, gpu_cc_hawi_match_table); static int gpu_cc_hawi_probe(struct platform_device *pdev) { + if (device_is_compatible(&pdev->dev, "qcom,maili-gpucc")) + gpu_cc_pll0.config = &gpu_cc_pll0_config_maili; + return qcom_cc_probe(pdev, &gpu_cc_hawi_desc); } From c5d93fd3aac735fa643f5e1117bd34990d91b755 Mon Sep 17 00:00:00 2001 From: Imran Shaik Date: Wed, 5 Aug 2026 22:17:14 +0530 Subject: [PATCH 357/857] clk: qcom: gcc-qcs8300: Add support for TSCSS clocks and resets Add the GCC TSCSS clocks and reset support required for the Timestamp Counter Subsystem (TSCSS) functionality on Qualcomm QCS8300 SoC. Signed-off-by: Imran Shaik Reviewed-by: Taniya Das Link: https://lore.kernel.org/r/20260805-qcs8300-tsc-clks-v1-2-e7a5101ed479@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/gcc-qcs8300.c | 103 +++++++++++++++++++++++++++++++++ 1 file changed, 103 insertions(+) diff --git a/drivers/clk/qcom/gcc-qcs8300.c b/drivers/clk/qcom/gcc-qcs8300.c index 31fd870b10f7ae..146b3b5dfead31 100644 --- a/drivers/clk/qcom/gcc-qcs8300.c +++ b/drivers/clk/qcom/gcc-qcs8300.c @@ -42,6 +42,7 @@ enum { P_GCC_GPLL0_OUT_MAIN, P_GCC_GPLL1_OUT_MAIN, P_GCC_GPLL4_OUT_MAIN, + P_GCC_GPLL5_OUT_MAIN, P_GCC_GPLL7_OUT_MAIN, P_GCC_GPLL9_OUT_MAIN, P_PCIE_0_PIPE_CLK, @@ -128,6 +129,23 @@ static struct clk_alpha_pll gcc_gpll4 = { }, }; +static struct clk_alpha_pll gcc_gpll5 = { + .offset = 0x5000, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_EVO], + .clkr = { + .enable_reg = 0x4b028, + .enable_mask = BIT(5), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_gpll5", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_fixed_lucid_evo_ops, + }, + }, +}; + static struct clk_alpha_pll gcc_gpll7 = { .offset = 0x7000, .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_EVO], @@ -364,6 +382,22 @@ static const struct clk_parent_data gcc_parent_data_18[] = { { .index = DT_BI_TCXO }, }; +static const struct parent_map gcc_parent_map_19[] = { + { P_BI_TCXO, 0 }, + { P_GCC_GPLL7_OUT_MAIN, 2 }, + { P_GCC_GPLL5_OUT_MAIN, 3 }, + { P_GCC_GPLL4_OUT_MAIN, 5 }, + { P_GCC_GPLL0_OUT_EVEN, 6 }, +}; + +static const struct clk_parent_data gcc_parent_data_19[] = { + { .index = DT_BI_TCXO }, + { .hw = &gcc_gpll7.clkr.hw }, + { .hw = &gcc_gpll5.clkr.hw }, + { .hw = &gcc_gpll4.clkr.hw }, + { .hw = &gcc_gpll0_out_even.clkr.hw }, +}; + static struct clk_regmap_mux gcc_pcie_0_phy_aux_clk_src = { .reg = 0xa9074, .shift = 0, @@ -1094,6 +1128,25 @@ static struct clk_rcg2 gcc_sdcc1_ice_core_clk_src = { }, }; +static const struct freq_tbl ftbl_gcc_tscss_cntr_clk_src[] = { + F(15625000, P_GCC_GPLL7_OUT_MAIN, 16, 1, 4), + { } +}; + +static struct clk_rcg2 gcc_tscss_cntr_clk_src = { + .cmd_rcgr = 0x21008, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_19, + .freq_tbl = ftbl_gcc_tscss_cntr_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_tscss_cntr_clk_src", + .parent_data = gcc_parent_data_19, + .num_parents = ARRAY_SIZE(gcc_parent_data_19), + .ops = &clk_rcg2_shared_ops, + }, +}; + static const struct freq_tbl ftbl_gcc_ufs_phy_axi_clk_src[] = { F(25000000, P_GCC_GPLL0_OUT_EVEN, 12, 0, 0), F(75000000, P_GCC_GPLL0_OUT_EVEN, 4, 0, 0), @@ -2899,6 +2952,50 @@ static struct clk_branch gcc_sgmi_clkref_en = { }, }; +static struct clk_branch gcc_tscss_ahb_clk = { + .halt_reg = 0x21024, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x21024, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_tscss_ahb_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_tscss_etu_clk = { + .halt_reg = 0x21020, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x21020, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_tscss_etu_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_tscss_global_cntr_clk = { + .halt_reg = 0x21004, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x21004, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_tscss_global_cntr_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_tscss_cntr_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + static struct clk_branch gcc_ufs_phy_ahb_clk = { .halt_reg = 0x83020, .halt_check = BRANCH_HALT_VOTED, @@ -3360,6 +3457,7 @@ static struct clk_regmap *gcc_qcs8300_clocks[] = { [GCC_GPLL0_OUT_EVEN] = &gcc_gpll0_out_even.clkr, [GCC_GPLL1] = &gcc_gpll1.clkr, [GCC_GPLL4] = &gcc_gpll4.clkr, + [GCC_GPLL5] = &gcc_gpll5.clkr, [GCC_GPLL7] = &gcc_gpll7.clkr, [GCC_GPLL9] = &gcc_gpll9.clkr, [GCC_GPU_GPLL0_CLK_SRC] = &gcc_gpu_gpll0_clk_src.clkr, @@ -3464,6 +3562,10 @@ static struct clk_regmap *gcc_qcs8300_clocks[] = { [GCC_SDCC1_ICE_CORE_CLK] = &gcc_sdcc1_ice_core_clk.clkr, [GCC_SDCC1_ICE_CORE_CLK_SRC] = &gcc_sdcc1_ice_core_clk_src.clkr, [GCC_SGMI_CLKREF_EN] = &gcc_sgmi_clkref_en.clkr, + [GCC_TSCSS_AHB_CLK] = &gcc_tscss_ahb_clk.clkr, + [GCC_TSCSS_CNTR_CLK_SRC] = &gcc_tscss_cntr_clk_src.clkr, + [GCC_TSCSS_ETU_CLK] = &gcc_tscss_etu_clk.clkr, + [GCC_TSCSS_GLOBAL_CNTR_CLK] = &gcc_tscss_global_cntr_clk.clkr, [GCC_UFS_PHY_AHB_CLK] = &gcc_ufs_phy_ahb_clk.clkr, [GCC_UFS_PHY_AXI_CLK] = &gcc_ufs_phy_axi_clk.clkr, [GCC_UFS_PHY_AXI_CLK_SRC] = &gcc_ufs_phy_axi_clk_src.clkr, @@ -3523,6 +3625,7 @@ static const struct qcom_reset_map gcc_qcs8300_resets[] = { [GCC_PCIE_1_PHY_BCR] = { 0xae08c }, [GCC_PCIE_1_PHY_NOCSR_COM_PHY_BCR] = { 0xae094 }, [GCC_SDCC1_BCR] = { 0x20000 }, + [GCC_TSCSS_BCR] = { 0x21000 }, [GCC_UFS_PHY_BCR] = { 0x83000 }, [GCC_USB20_PRIM_BCR] = { 0x1c000 }, [GCC_USB2_PHY_PRIM_BCR] = { 0x5c01c }, From c07297f8e0e7dda7589c74be7ad570ef3a244efb Mon Sep 17 00:00:00 2001 From: Taniya Das Date: Sat, 1 Aug 2026 19:43:53 +0530 Subject: [PATCH 358/857] clk: qcom: nwgcc-nord: Add video axi clock resets for Nord The global clock controller video axi reset clocks are required by the video SW driver to assert and deassert the clock resets during their power down sequence. Hence add these clock resets. Fixes: a4f780cd5c7a ("clk: qcom: gcc: Add multiple global clock controller driver for Nord SoC") Signed-off-by: Taniya Das Reviewed-by: Dmitry Baryshkov Reviewed-by: Abel Vesa Link: https://lore.kernel.org/r/20260801-nord_videocc_camcc-v2-2-674d7718e41f@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/nwgcc-nord.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/drivers/clk/qcom/nwgcc-nord.c b/drivers/clk/qcom/nwgcc-nord.c index e061c0623e9a46..280558f15748c9 100644 --- a/drivers/clk/qcom/nwgcc-nord.c +++ b/drivers/clk/qcom/nwgcc-nord.c @@ -623,6 +623,9 @@ static const struct qcom_reset_map nw_gcc_nord_resets[] = { [NW_GCC_GPU_2_BCR] = { 0x24000 }, [NW_GCC_GPU_BCR] = { 0x23000 }, [NW_GCC_VIDEO_BCR] = { 0x1a000 }, + [NW_GCC_VIDEO_AXI0_CLK_ARES] = { 0x1a008, 2 }, + [NW_GCC_VIDEO_AXI0C_CLK_ARES] = { 0x1a01c, 2 }, + [NW_GCC_VIDEO_AXI1_CLK_ARES] = { 0x1a030, 2 }, }; static const u32 nw_gcc_nord_critical_cbcrs[] = { From 7f26939a78de32f9f2b0b9cd67c311ef4a0caf29 Mon Sep 17 00:00:00 2001 From: Taniya Das Date: Sat, 1 Aug 2026 19:43:56 +0530 Subject: [PATCH 359/857] clk: qcom: videocc-glymur: Add video clock controller support for Nord Nord shares the video clock controller topology with Glymur, differing only in the PLL0 hardware (Lucid-OLE vs Taycan-EKO-T) and the MVS0 frequency table. Extend the existing Glymur video clock controller driver to also probe on the qcom,nord-videocc compatible, switching to the Nord PLL configuration, VCO table, and MVS0 frequency table at probe time. Reviewed-by: Konrad Dybcio Signed-off-by: Taniya Das Link: https://lore.kernel.org/r/20260801-nord_videocc_camcc-v2-5-674d7718e41f@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/Kconfig | 11 +++++++++ drivers/clk/qcom/Makefile | 1 + drivers/clk/qcom/videocc-glymur.c | 38 +++++++++++++++++++++++++++++++ 3 files changed, 50 insertions(+) diff --git a/drivers/clk/qcom/Kconfig b/drivers/clk/qcom/Kconfig index aba035b8932209..cca4cb83a96647 100644 --- a/drivers/clk/qcom/Kconfig +++ b/drivers/clk/qcom/Kconfig @@ -273,6 +273,17 @@ config CLK_SHIKRA_AUDIOCORECC Say Y if you want to use AudioCoreCC clocks required to support audio devices and it's functionality. +config CLK_NORD_VIDEOCC + tristate "Nord VIDEO Clock Controller" + depends on ARM64 || COMPILE_TEST + select CLK_NORD_GCC + default m if ARCH_QCOM + help + Support for the video clock controller on Qualcomm Technologies, Inc. + Nord devices. + Say Y if you want to support video devices and functionality such as + video encode/decode. + config CLK_SHIKRA_GCC tristate "Shikra Global Clock Controller" depends on ARM64 || COMPILE_TEST diff --git a/drivers/clk/qcom/Makefile b/drivers/clk/qcom/Makefile index e867cf0fdd67ea..abab25fff91cba 100644 --- a/drivers/clk/qcom/Makefile +++ b/drivers/clk/qcom/Makefile @@ -50,6 +50,7 @@ obj-$(CONFIG_CLK_NORD_DISPCC) += dispcc0-nord.o dispcc1-nord.o obj-$(CONFIG_CLK_NORD_GCC) += gcc-nord.o negcc-nord.o nwgcc-nord.o segcc-nord.o obj-$(CONFIG_CLK_NORD_GPUCC) += gpucc-nord.o gpu2cc-nord.o obj-$(CONFIG_CLK_NORD_TCSRCC) += tcsrcc-nord.o +obj-$(CONFIG_CLK_NORD_VIDEOCC) += videocc-glymur.o obj-$(CONFIG_CLK_SHIKRA_AUDIOCORECC) += audiocorecc-shikra.o obj-$(CONFIG_CLK_SHIKRA_GCC) += gcc-shikra.o obj-$(CONFIG_CLK_X1E80100_CAMCC) += camcc-x1e80100.o diff --git a/drivers/clk/qcom/videocc-glymur.c b/drivers/clk/qcom/videocc-glymur.c index 18313a65e78d95..2776140de3555d 100644 --- a/drivers/clk/qcom/videocc-glymur.c +++ b/drivers/clk/qcom/videocc-glymur.c @@ -37,6 +37,10 @@ static const struct pll_vco taycan_eko_t_vco[] = { { 249600000, 2500000000, 0 }, }; +static const struct pll_vco lucid_ole_vco[] = { + { 249600000, 2300000000, 0 }, +}; + /* 720.0 MHz Configuration */ static const struct alpha_pll_config video_cc_pll0_config = { .l = 0x25, @@ -48,6 +52,21 @@ static const struct alpha_pll_config video_cc_pll0_config = { .user_ctl_hi_val = 0x00000002, }; +/* 720.0 MHz Configuration */ +static const struct alpha_pll_config video_cc_pll0_config_nord = { + .l = 0x25, + .alpha = 0x8000, + .config_ctl_val = 0x20485699, + .config_ctl_hi_val = 0x00182261, + .config_ctl_hi1_val = 0x82aa299c, + .test_ctl_val = 0x00000000, + .test_ctl_hi_val = 0x00000003, + .test_ctl_hi1_val = 0x00009000, + .test_ctl_hi2_val = 0x00000034, + .user_ctl_val = 0x00000000, + .user_ctl_hi_val = 0x00400005, +}; + static struct clk_alpha_pll video_cc_pll0 = { .offset = 0x0, .config = &video_cc_pll0_config, @@ -112,6 +131,15 @@ static struct clk_rcg2 video_cc_ahb_clk_src = { }, }; +static const struct freq_tbl ftbl_video_cc_mvs0_clk_src_nord[] = { + F(720000000, P_VIDEO_CC_PLL0_OUT_MAIN, 1, 0, 0), + F(1305000000, P_VIDEO_CC_PLL0_OUT_MAIN, 1, 0, 0), + F(1440000000, P_VIDEO_CC_PLL0_OUT_MAIN, 1, 0, 0), + F(1600000000, P_VIDEO_CC_PLL0_OUT_MAIN, 1, 0, 0), + F(1680000000, P_VIDEO_CC_PLL0_OUT_MAIN, 1, 0, 0), + { } +}; + static const struct freq_tbl ftbl_video_cc_mvs0_clk_src[] = { F(720000000, P_VIDEO_CC_PLL0_OUT_MAIN, 1, 0, 0), F(1014000000, P_VIDEO_CC_PLL0_OUT_MAIN, 1, 0, 0), @@ -508,12 +536,22 @@ static const struct qcom_cc_desc video_cc_glymur_desc = { static const struct of_device_id video_cc_glymur_match_table[] = { { .compatible = "qcom,glymur-videocc" }, + { .compatible = "qcom,nord-videocc" }, { } }; MODULE_DEVICE_TABLE(of, video_cc_glymur_match_table); static int video_cc_glymur_probe(struct platform_device *pdev) { + if (of_device_is_compatible(pdev->dev.of_node, "qcom,nord-videocc")) { + video_cc_pll0.vco_table = lucid_ole_vco; + video_cc_pll0.num_vco = ARRAY_SIZE(lucid_ole_vco); + video_cc_pll0.regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE]; + video_cc_pll0.config = &video_cc_pll0_config_nord; + + video_cc_mvs0_clk_src.freq_tbl = ftbl_video_cc_mvs0_clk_src_nord; + } + return qcom_cc_probe(pdev, &video_cc_glymur_desc); } From 7647815dfc8f7282ca37bb28fa22599edd9bb597 Mon Sep 17 00:00:00 2001 From: Taniya Das Date: Sat, 1 Aug 2026 19:43:57 +0530 Subject: [PATCH 360/857] clk: qcom: camcc: Add support for camera clock controller for Nord Add support for the Camera Clock Controller (CAMCC) on the Nord platform for camera SW drivers to request for these clocks. Reviewed-by: Dmitry Baryshkov Reviewed-by: Konrad Dybcio Signed-off-by: Taniya Das Link: https://lore.kernel.org/r/20260801-nord_videocc_camcc-v2-6-674d7718e41f@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/Kconfig | 11 + drivers/clk/qcom/Makefile | 1 + drivers/clk/qcom/camcc-nord.c | 2941 +++++++++++++++++++++++++++++++++ 3 files changed, 2953 insertions(+) create mode 100644 drivers/clk/qcom/camcc-nord.c diff --git a/drivers/clk/qcom/Kconfig b/drivers/clk/qcom/Kconfig index cca4cb83a96647..b373ecbb2befb4 100644 --- a/drivers/clk/qcom/Kconfig +++ b/drivers/clk/qcom/Kconfig @@ -233,6 +233,17 @@ config CLK_NORD_DISPCC Say Y if you want to support display devices and functionality such as splash screen. +config CLK_NORD_CAMCC + tristate "Nord Camera Clock Controller" + depends on ARM64 || COMPILE_TEST + select CLK_NORD_GCC + default m if ARCH_QCOM + help + Support for the camera clock controller on Qualcomm Technologies, Inc + Nord devices. + Say Y if you want to support camera devices and functionality such as + capturing pictures. + config CLK_NORD_GCC tristate "Nord Global Clock Controller" depends on ARM64 || COMPILE_TEST diff --git a/drivers/clk/qcom/Makefile b/drivers/clk/qcom/Makefile index abab25fff91cba..a7674aef8a67f2 100644 --- a/drivers/clk/qcom/Makefile +++ b/drivers/clk/qcom/Makefile @@ -46,6 +46,7 @@ obj-$(CONFIG_CLK_KAANAPALI_GPUCC) += gpucc-kaanapali.o gxclkctl-kaanapali.o obj-$(CONFIG_CLK_KAANAPALI_TCSRCC) += tcsrcc-kaanapali.o obj-$(CONFIG_CLK_KAANAPALI_VIDEOCC) += videocc-kaanapali.o obj-$(CONFIG_CLK_MAILI_VIDEOCC) += videocc-maili.o +obj-$(CONFIG_CLK_NORD_CAMCC) += camcc-nord.o obj-$(CONFIG_CLK_NORD_DISPCC) += dispcc0-nord.o dispcc1-nord.o obj-$(CONFIG_CLK_NORD_GCC) += gcc-nord.o negcc-nord.o nwgcc-nord.o segcc-nord.o obj-$(CONFIG_CLK_NORD_GPUCC) += gpucc-nord.o gpu2cc-nord.o diff --git a/drivers/clk/qcom/camcc-nord.c b/drivers/clk/qcom/camcc-nord.c new file mode 100644 index 00000000000000..9e3c40cb3ad5f0 --- /dev/null +++ b/drivers/clk/qcom/camcc-nord.c @@ -0,0 +1,2941 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) Qualcomm Technologies, Inc. and/or its subsidiaries. + */ + +#include +#include +#include +#include + +#include + +#include "clk-alpha-pll.h" +#include "clk-branch.h" +#include "clk-pll.h" +#include "clk-rcg.h" +#include "clk-regmap.h" +#include "clk-regmap-divider.h" +#include "clk-regmap-mux.h" +#include "common.h" +#include "gdsc.h" +#include "reset.h" + +enum { + DT_IFACE, + DT_BI_TCXO, + DT_BI_TCXO_AO, + DT_SLEEP_CLK +}; + +enum { + P_BI_TCXO, + P_CAM_CC_PLL0_OUT_EVEN, + P_CAM_CC_PLL0_OUT_MAIN, + P_CAM_CC_PLL0_OUT_ODD, + P_CAM_CC_PLL2_OUT_EVEN, + P_CAM_CC_PLL3_OUT_EVEN, + P_CAM_CC_PLL4_OUT_EVEN, + P_CAM_CC_PLL5_OUT_EVEN, + P_CAM_CC_PLL6_OUT_EVEN, + P_SLEEP_CLK, +}; + +static const struct pll_vco lucid_ole_vco[] = { + { 249600000, 2300000000, 0 }, +}; + +/* 1200.0 MHz Configuration */ +static const struct alpha_pll_config cam_cc_pll0_config = { + .l = 0x3e, + .alpha = 0x8000, + .config_ctl_val = 0x20485699, + .config_ctl_hi_val = 0x00182261, + .config_ctl_hi1_val = 0x82aa299c, + .test_ctl_val = 0x00000000, + .test_ctl_hi_val = 0x00000003, + .test_ctl_hi1_val = 0x00009000, + .test_ctl_hi2_val = 0x00000034, + .user_ctl_val = 0x00008400, + .user_ctl_hi_val = 0x00400005, +}; + +static struct clk_alpha_pll cam_cc_pll0 = { + .offset = 0x0, + .config = &cam_cc_pll0_config, + .vco_table = lucid_ole_vco, + .num_vco = ARRAY_SIZE(lucid_ole_vco), + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr = { + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll0", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO_AO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_lucid_evo_ops, + }, + }, +}; + +static const struct clk_div_table post_div_table_cam_cc_pll0_out_even[] = { + { 0x1, 2 }, + { } +}; + +static struct clk_alpha_pll_postdiv cam_cc_pll0_out_even = { + .offset = 0x0, + .post_div_shift = 10, + .post_div_table = post_div_table_cam_cc_pll0_out_even, + .num_post_div = ARRAY_SIZE(post_div_table_cam_cc_pll0_out_even), + .width = 4, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll0_out_even", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_pll0.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_alpha_pll_postdiv_lucid_evo_ops, + }, +}; + +static const struct clk_div_table post_div_table_cam_cc_pll0_out_odd[] = { + { 0x2, 3 }, + { } +}; + +static struct clk_alpha_pll_postdiv cam_cc_pll0_out_odd = { + .offset = 0x0, + .post_div_shift = 14, + .post_div_table = post_div_table_cam_cc_pll0_out_odd, + .num_post_div = ARRAY_SIZE(post_div_table_cam_cc_pll0_out_odd), + .width = 4, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll0_out_odd", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_pll0.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_alpha_pll_postdiv_lucid_evo_ops, + }, +}; + +/* 720.0 MHz Configuration */ +static const struct alpha_pll_config cam_cc_pll2_config = { + .l = 0x25, + .alpha = 0x8000, + .config_ctl_val = 0x20485699, + .config_ctl_hi_val = 0x00182261, + .config_ctl_hi1_val = 0x82aa299c, + .test_ctl_val = 0x00000000, + .test_ctl_hi_val = 0x00000003, + .test_ctl_hi1_val = 0x00009000, + .test_ctl_hi2_val = 0x00000034, + .user_ctl_val = 0x00000400, + .user_ctl_hi_val = 0x00400005, +}; + +static struct clk_alpha_pll cam_cc_pll2 = { + .offset = 0x2000, + .config = &cam_cc_pll2_config, + .vco_table = lucid_ole_vco, + .num_vco = ARRAY_SIZE(lucid_ole_vco), + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr = { + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll2", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_lucid_evo_ops, + }, + }, +}; + +static const struct clk_div_table post_div_table_cam_cc_pll2_out_even[] = { + { 0x1, 2 }, + { } +}; + +static struct clk_alpha_pll_postdiv cam_cc_pll2_out_even = { + .offset = 0x2000, + .post_div_shift = 10, + .post_div_table = post_div_table_cam_cc_pll2_out_even, + .num_post_div = ARRAY_SIZE(post_div_table_cam_cc_pll2_out_even), + .width = 4, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll2_out_even", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_pll2.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_alpha_pll_postdiv_lucid_evo_ops, + }, +}; + +/* 720.0 MHz Configuration */ +static const struct alpha_pll_config cam_cc_pll3_config = { + .l = 0x25, + .alpha = 0x8000, + .config_ctl_val = 0x20485699, + .config_ctl_hi_val = 0x00182261, + .config_ctl_hi1_val = 0x82aa299c, + .test_ctl_val = 0x00000000, + .test_ctl_hi_val = 0x00000003, + .test_ctl_hi1_val = 0x00009000, + .test_ctl_hi2_val = 0x00000034, + .user_ctl_val = 0x00000400, + .user_ctl_hi_val = 0x00400005, +}; + +static struct clk_alpha_pll cam_cc_pll3 = { + .offset = 0x3000, + .config = &cam_cc_pll3_config, + .vco_table = lucid_ole_vco, + .num_vco = ARRAY_SIZE(lucid_ole_vco), + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr = { + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll3", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_lucid_evo_ops, + }, + }, +}; + +static const struct clk_div_table post_div_table_cam_cc_pll3_out_even[] = { + { 0x1, 2 }, + { } +}; + +static struct clk_alpha_pll_postdiv cam_cc_pll3_out_even = { + .offset = 0x3000, + .post_div_shift = 10, + .post_div_table = post_div_table_cam_cc_pll3_out_even, + .num_post_div = ARRAY_SIZE(post_div_table_cam_cc_pll3_out_even), + .width = 4, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll3_out_even", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_pll3.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_alpha_pll_postdiv_lucid_evo_ops, + }, +}; + +/* 720.0 MHz Configuration */ +static const struct alpha_pll_config cam_cc_pll4_config = { + .l = 0x25, + .alpha = 0x8000, + .config_ctl_val = 0x20485699, + .config_ctl_hi_val = 0x00182261, + .config_ctl_hi1_val = 0x82aa299c, + .test_ctl_val = 0x00000000, + .test_ctl_hi_val = 0x00000003, + .test_ctl_hi1_val = 0x00009000, + .test_ctl_hi2_val = 0x00000034, + .user_ctl_val = 0x00000400, + .user_ctl_hi_val = 0x00400005, +}; + +static struct clk_alpha_pll cam_cc_pll4 = { + .offset = 0x4000, + .config = &cam_cc_pll4_config, + .vco_table = lucid_ole_vco, + .num_vco = ARRAY_SIZE(lucid_ole_vco), + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr = { + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll4", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_lucid_evo_ops, + }, + }, +}; + +static const struct clk_div_table post_div_table_cam_cc_pll4_out_even[] = { + { 0x1, 2 }, + { } +}; + +static struct clk_alpha_pll_postdiv cam_cc_pll4_out_even = { + .offset = 0x4000, + .post_div_shift = 10, + .post_div_table = post_div_table_cam_cc_pll4_out_even, + .num_post_div = ARRAY_SIZE(post_div_table_cam_cc_pll4_out_even), + .width = 4, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll4_out_even", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_pll4.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_alpha_pll_postdiv_lucid_evo_ops, + }, +}; + +/* 720.0 MHz Configuration */ +static const struct alpha_pll_config cam_cc_pll5_config = { + .l = 0x25, + .alpha = 0x8000, + .config_ctl_val = 0x20485699, + .config_ctl_hi_val = 0x00182261, + .config_ctl_hi1_val = 0x82aa299c, + .test_ctl_val = 0x00000000, + .test_ctl_hi_val = 0x00000003, + .test_ctl_hi1_val = 0x00009000, + .test_ctl_hi2_val = 0x00000034, + .user_ctl_val = 0x00000400, + .user_ctl_hi_val = 0x00400005, +}; + +static struct clk_alpha_pll cam_cc_pll5 = { + .offset = 0x5000, + .config = &cam_cc_pll5_config, + .vco_table = lucid_ole_vco, + .num_vco = ARRAY_SIZE(lucid_ole_vco), + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr = { + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll5", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_lucid_evo_ops, + }, + }, +}; + +static const struct clk_div_table post_div_table_cam_cc_pll5_out_even[] = { + { 0x1, 2 }, + { } +}; + +static struct clk_alpha_pll_postdiv cam_cc_pll5_out_even = { + .offset = 0x5000, + .post_div_shift = 10, + .post_div_table = post_div_table_cam_cc_pll5_out_even, + .num_post_div = ARRAY_SIZE(post_div_table_cam_cc_pll5_out_even), + .width = 4, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll5_out_even", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_pll5.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_alpha_pll_postdiv_lucid_evo_ops, + }, +}; + +/* 720.0 MHz Configuration */ +static const struct alpha_pll_config cam_cc_pll6_config = { + .l = 0x25, + .alpha = 0x8000, + .config_ctl_val = 0x20485699, + .config_ctl_hi_val = 0x00182261, + .config_ctl_hi1_val = 0x82aa299c, + .test_ctl_val = 0x00000000, + .test_ctl_hi_val = 0x00000003, + .test_ctl_hi1_val = 0x00009000, + .test_ctl_hi2_val = 0x00000034, + .user_ctl_val = 0x00000400, + .user_ctl_hi_val = 0x00400005, +}; + +static struct clk_alpha_pll cam_cc_pll6 = { + .offset = 0x6000, + .config = &cam_cc_pll6_config, + .vco_table = lucid_ole_vco, + .num_vco = ARRAY_SIZE(lucid_ole_vco), + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr = { + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll6", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_lucid_evo_ops, + }, + }, +}; + +static const struct clk_div_table post_div_table_cam_cc_pll6_out_even[] = { + { 0x1, 2 }, + { } +}; + +static struct clk_alpha_pll_postdiv cam_cc_pll6_out_even = { + .offset = 0x6000, + .post_div_shift = 10, + .post_div_table = post_div_table_cam_cc_pll6_out_even, + .num_post_div = ARRAY_SIZE(post_div_table_cam_cc_pll6_out_even), + .width = 4, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE], + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_pll6_out_even", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_pll6.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_alpha_pll_postdiv_lucid_evo_ops, + }, +}; + +static const struct parent_map cam_cc_parent_map_0[] = { + { P_BI_TCXO, 0 }, + { P_CAM_CC_PLL0_OUT_MAIN, 1 }, + { P_CAM_CC_PLL0_OUT_EVEN, 2 }, + { P_CAM_CC_PLL0_OUT_ODD, 3 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_0[] = { + { .index = DT_BI_TCXO }, + { .hw = &cam_cc_pll0.clkr.hw }, + { .hw = &cam_cc_pll0_out_even.clkr.hw }, + { .hw = &cam_cc_pll0_out_odd.clkr.hw }, +}; + +static const struct parent_map cam_cc_parent_map_1[] = { + { P_BI_TCXO, 0 }, + { P_CAM_CC_PLL0_OUT_MAIN, 1 }, + { P_CAM_CC_PLL0_OUT_EVEN, 2 }, + { P_CAM_CC_PLL0_OUT_ODD, 3 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_1[] = { + { .index = DT_BI_TCXO }, + { .hw = &cam_cc_pll0.clkr.hw }, + { .hw = &cam_cc_pll0_out_even.clkr.hw }, + { .hw = &cam_cc_pll0_out_odd.clkr.hw }, +}; + +static const struct parent_map cam_cc_parent_map_2[] = { + { P_BI_TCXO, 0 }, + { P_CAM_CC_PLL0_OUT_MAIN, 1 }, + { P_CAM_CC_PLL0_OUT_EVEN, 2 }, + { P_CAM_CC_PLL3_OUT_EVEN, 3 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_2[] = { + { .index = DT_BI_TCXO }, + { .hw = &cam_cc_pll0.clkr.hw }, + { .hw = &cam_cc_pll0_out_even.clkr.hw }, + { .hw = &cam_cc_pll3_out_even.clkr.hw }, +}; + +static const struct parent_map cam_cc_parent_map_3[] = { + { P_BI_TCXO, 0 }, + { P_CAM_CC_PLL2_OUT_EVEN, 2 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_3[] = { + { .index = DT_BI_TCXO }, + { .hw = &cam_cc_pll2_out_even.clkr.hw }, +}; + +static const struct parent_map cam_cc_parent_map_4[] = { + { P_BI_TCXO, 0 }, + { P_CAM_CC_PLL4_OUT_EVEN, 6 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_4[] = { + { .index = DT_BI_TCXO }, + { .hw = &cam_cc_pll4_out_even.clkr.hw }, +}; + +static const struct parent_map cam_cc_parent_map_5[] = { + { P_BI_TCXO, 0 }, + { P_CAM_CC_PLL5_OUT_EVEN, 6 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_5[] = { + { .index = DT_BI_TCXO }, + { .hw = &cam_cc_pll5_out_even.clkr.hw }, +}; + +static const struct parent_map cam_cc_parent_map_6[] = { + { P_BI_TCXO, 0 }, + { P_CAM_CC_PLL6_OUT_EVEN, 6 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_6[] = { + { .index = DT_BI_TCXO }, + { .hw = &cam_cc_pll6_out_even.clkr.hw }, +}; + +static const struct parent_map cam_cc_parent_map_7[] = { + { P_BI_TCXO, 0 }, + { P_CAM_CC_PLL0_OUT_MAIN, 1 }, + { P_CAM_CC_PLL0_OUT_EVEN, 2 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_7[] = { + { .index = DT_BI_TCXO }, + { .hw = &cam_cc_pll0.clkr.hw }, + { .hw = &cam_cc_pll0_out_even.clkr.hw }, +}; + +static const struct parent_map cam_cc_parent_map_8[] = { + { P_SLEEP_CLK, 0 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_8[] = { + { .index = DT_SLEEP_CLK }, +}; + +static const struct parent_map cam_cc_parent_map_9[] = { + { P_BI_TCXO, 0 }, +}; + +static const struct clk_parent_data cam_cc_parent_data_9[] = { + { .index = DT_BI_TCXO }, +}; + +static const struct freq_tbl ftbl_cam_cc_camnoc_rt_axi_clk_src[] = { + F(300000000, P_CAM_CC_PLL0_OUT_EVEN, 2, 0, 0), + F(400000000, P_CAM_CC_PLL0_OUT_MAIN, 3, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_camnoc_rt_axi_clk_src = { + .cmd_rcgr = 0x13244, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_camnoc_rt_axi_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_camnoc_rt_axi_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_cci_0_clk_src[] = { + F(37500000, P_CAM_CC_PLL0_OUT_EVEN, 16, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_cci_0_clk_src = { + .cmd_rcgr = 0x13110, + .mnd_width = 8, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_cci_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_0_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_cci_1_clk_src = { + .cmd_rcgr = 0x13130, + .mnd_width = 8, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_cci_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_1_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_cci_2_clk_src = { + .cmd_rcgr = 0x13150, + .mnd_width = 8, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_cci_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_2_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_cci_3_clk_src = { + .cmd_rcgr = 0x13170, + .mnd_width = 8, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_cci_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_3_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_cci_4_clk_src = { + .cmd_rcgr = 0x13190, + .mnd_width = 8, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_cci_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_4_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_cphy_rx_clk_src[] = { + F(300000000, P_CAM_CC_PLL0_OUT_EVEN, 2, 0, 0), + F(400000000, P_CAM_CC_PLL0_OUT_ODD, 1, 0, 0), + F(480000000, P_CAM_CC_PLL0_OUT_MAIN, 2.5, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_cphy_rx_clk_src = { + .cmd_rcgr = 0x120ac, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_1, + .freq_tbl = ftbl_cam_cc_cphy_rx_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cphy_rx_clk_src", + .parent_data = cam_cc_parent_data_1, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_csi0phytimer_clk_src[] = { + F(300000000, P_CAM_CC_PLL0_OUT_EVEN, 2, 0, 0), + F(400000000, P_CAM_CC_PLL0_OUT_ODD, 1, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_csi0phytimer_clk_src = { + .cmd_rcgr = 0x10000, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_1, + .freq_tbl = ftbl_cam_cc_csi0phytimer_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi0phytimer_clk_src", + .parent_data = cam_cc_parent_data_1, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_csi1phytimer_clk_src = { + .cmd_rcgr = 0x10028, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_1, + .freq_tbl = ftbl_cam_cc_csi0phytimer_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi1phytimer_clk_src", + .parent_data = cam_cc_parent_data_1, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_csi2phytimer_clk_src = { + .cmd_rcgr = 0x1004c, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_1, + .freq_tbl = ftbl_cam_cc_csi0phytimer_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi2phytimer_clk_src", + .parent_data = cam_cc_parent_data_1, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_csi3phytimer_clk_src = { + .cmd_rcgr = 0x10070, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_1, + .freq_tbl = ftbl_cam_cc_csi0phytimer_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi3phytimer_clk_src", + .parent_data = cam_cc_parent_data_1, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_csi4phytimer_clk_src = { + .cmd_rcgr = 0x10094, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_1, + .freq_tbl = ftbl_cam_cc_csi0phytimer_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi4phytimer_clk_src", + .parent_data = cam_cc_parent_data_1, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_csid_clk_src[] = { + F(300000000, P_CAM_CC_PLL0_OUT_MAIN, 4, 0, 0), + F(400000000, P_CAM_CC_PLL0_OUT_MAIN, 3, 0, 0), + F(480000000, P_CAM_CC_PLL0_OUT_MAIN, 2.5, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_csid_clk_src = { + .cmd_rcgr = 0x13214, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_1, + .freq_tbl = ftbl_cam_cc_csid_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csid_clk_src", + .parent_data = cam_cc_parent_data_1, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_fast_ahb_clk_src[] = { + F(200000000, P_CAM_CC_PLL0_OUT_EVEN, 3, 0, 0), + F(300000000, P_CAM_CC_PLL0_OUT_MAIN, 4, 0, 0), + F(400000000, P_CAM_CC_PLL0_OUT_MAIN, 3, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_fast_ahb_clk_src = { + .cmd_rcgr = 0x131dc, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_fast_ahb_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_fast_ahb_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_icp_0_clk_src[] = { + F(360000000, P_CAM_CC_PLL3_OUT_EVEN, 1, 0, 0), + F(480000000, P_CAM_CC_PLL3_OUT_EVEN, 1, 0, 0), + F(600000000, P_CAM_CC_PLL3_OUT_EVEN, 1, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_icp_0_clk_src = { + .cmd_rcgr = 0x130a4, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_2, + .freq_tbl = ftbl_cam_cc_icp_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_icp_0_clk_src", + .parent_data = cam_cc_parent_data_2, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_2), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_icp_1_clk_src = { + .cmd_rcgr = 0x130dc, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_2, + .freq_tbl = ftbl_cam_cc_icp_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_icp_1_clk_src", + .parent_data = cam_cc_parent_data_2, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_2), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_ife_0_main_clk_src[] = { + F(360000000, P_CAM_CC_PLL4_OUT_EVEN, 1, 0, 0), + F(480000000, P_CAM_CC_PLL4_OUT_EVEN, 1, 0, 0), + F(650000000, P_CAM_CC_PLL4_OUT_EVEN, 1, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_ife_0_main_clk_src = { + .cmd_rcgr = 0x12018, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_4, + .freq_tbl = ftbl_cam_cc_ife_0_main_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_0_main_clk_src", + .parent_data = cam_cc_parent_data_4, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_4), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_ife_1_main_clk_src[] = { + F(360000000, P_CAM_CC_PLL5_OUT_EVEN, 1, 0, 0), + F(480000000, P_CAM_CC_PLL5_OUT_EVEN, 1, 0, 0), + F(650000000, P_CAM_CC_PLL5_OUT_EVEN, 1, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_ife_1_main_clk_src = { + .cmd_rcgr = 0x120dc, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_5, + .freq_tbl = ftbl_cam_cc_ife_1_main_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_1_main_clk_src", + .parent_data = cam_cc_parent_data_5, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_5), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_ife_2_main_clk_src[] = { + F(360000000, P_CAM_CC_PLL6_OUT_EVEN, 1, 0, 0), + F(480000000, P_CAM_CC_PLL6_OUT_EVEN, 1, 0, 0), + F(650000000, P_CAM_CC_PLL6_OUT_EVEN, 1, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_ife_2_main_clk_src = { + .cmd_rcgr = 0x12188, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_6, + .freq_tbl = ftbl_cam_cc_ife_2_main_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_2_main_clk_src", + .parent_data = cam_cc_parent_data_6, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_6), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_ife_lite_clk_src = { + .cmd_rcgr = 0x13000, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_1, + .freq_tbl = ftbl_cam_cc_csid_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_lite_clk_src", + .parent_data = cam_cc_parent_data_1, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_ife_lite_csid_clk_src = { + .cmd_rcgr = 0x13024, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_1, + .freq_tbl = ftbl_cam_cc_csid_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_lite_csid_clk_src", + .parent_data = cam_cc_parent_data_1, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_1), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_ipe_0_clk_src[] = { + F(360000000, P_CAM_CC_PLL2_OUT_EVEN, 1, 0, 0), + F(480000000, P_CAM_CC_PLL2_OUT_EVEN, 1, 0, 0), + F(650000000, P_CAM_CC_PLL2_OUT_EVEN, 1, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_ipe_0_clk_src = { + .cmd_rcgr = 0x11018, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_3, + .freq_tbl = ftbl_cam_cc_ipe_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ipe_0_clk_src", + .parent_data = cam_cc_parent_data_3, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_3), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_ipe_1_clk_src = { + .cmd_rcgr = 0x11074, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_3, + .freq_tbl = ftbl_cam_cc_ipe_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ipe_1_clk_src", + .parent_data = cam_cc_parent_data_3, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_3), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_qdss_debug_clk_src[] = { + F(200000000, P_CAM_CC_PLL0_OUT_EVEN, 3, 0, 0), + F(300000000, P_CAM_CC_PLL0_OUT_EVEN, 2, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_qdss_debug_clk_src = { + .cmd_rcgr = 0x13330, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_qdss_debug_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qdss_debug_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_qup_ahbm_clk_src[] = { + F(150000000, P_CAM_CC_PLL0_OUT_EVEN, 4, 0, 0), + F(240000000, P_CAM_CC_PLL0_OUT_MAIN, 5, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_qup_ahbm_clk_src = { + .cmd_rcgr = 0x132e8, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_qup_ahbm_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qup_ahbm_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_qup_core_2x_clk_src[] = { + F(200000000, P_CAM_CC_PLL0_OUT_ODD, 2, 0, 0), + F(300000000, P_CAM_CC_PLL0_OUT_EVEN, 2, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_qup_core_2x_clk_src = { + .cmd_rcgr = 0x132a0, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_qup_core_2x_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qup_core_2x_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_rcg2 cam_cc_qup_se_clk_src = { + .cmd_rcgr = 0x13308, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_7, + .freq_tbl = ftbl_cam_cc_cci_0_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qup_se_clk_src", + .parent_data = cam_cc_parent_data_7, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_7), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_sleep_clk_src[] = { + F(32000, P_SLEEP_CLK, 1, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_sleep_clk_src = { + .cmd_rcgr = 0x13384, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_8, + .freq_tbl = ftbl_cam_cc_sleep_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_sleep_clk_src", + .parent_data = cam_cc_parent_data_8, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_8), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_slow_ahb_clk_src[] = { + F(80000000, P_CAM_CC_PLL0_OUT_EVEN, 7.5, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_slow_ahb_clk_src = { + .cmd_rcgr = 0x131f8, + .mnd_width = 8, + .hid_width = 5, + .parent_map = cam_cc_parent_map_0, + .freq_tbl = ftbl_cam_cc_slow_ahb_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_slow_ahb_clk_src", + .parent_data = cam_cc_parent_data_0, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_0), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static const struct freq_tbl ftbl_cam_cc_xo_clk_src[] = { + F(19200000, P_BI_TCXO, 1, 0, 0), + { } +}; + +static struct clk_rcg2 cam_cc_xo_clk_src = { + .cmd_rcgr = 0x13368, + .mnd_width = 0, + .hid_width = 5, + .parent_map = cam_cc_parent_map_9, + .freq_tbl = ftbl_cam_cc_xo_clk_src, + .hw_clk_ctrl = true, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_xo_clk_src", + .parent_data = cam_cc_parent_data_9, + .num_parents = ARRAY_SIZE(cam_cc_parent_data_9), + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_rcg2_shared_ops, + }, +}; + +static struct clk_regmap_div cam_cc_qup_core_2x_div_clk_src = { + .reg = 0x132cc, + .shift = 0, + .width = 4, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qup_core_2x_div_clk_src", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_qup_core_2x_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_regmap_div_ro_ops, + }, +}; + +static struct clk_branch cam_cc_camnoc_dcd_xo_clk = { + .halt_reg = 0x13290, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13290, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_camnoc_dcd_xo_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_xo_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_camnoc_nrt_axi_clk = { + .halt_reg = 0x13278, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13278, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_camnoc_nrt_axi_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_camnoc_rt_axi_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_camnoc_rt_axi_clk = { + .halt_reg = 0x13260, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13260, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_camnoc_rt_axi_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_camnoc_rt_axi_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_cci_0_clk = { + .halt_reg = 0x1312c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1312c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_0_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cci_0_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_cci_1_clk = { + .halt_reg = 0x1314c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1314c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_1_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cci_1_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_cci_2_clk = { + .halt_reg = 0x1316c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1316c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_2_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cci_2_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_cci_3_clk = { + .halt_reg = 0x1318c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1318c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_3_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cci_3_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_cci_4_clk = { + .halt_reg = 0x131ac, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x131ac, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_cci_4_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cci_4_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ccu_fast_ahb_clk = { + .halt_reg = 0x1329c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1329c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ccu_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csi0phytimer_clk = { + .halt_reg = 0x1001c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1001c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi0phytimer_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_csi0phytimer_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csi1phytimer_clk = { + .halt_reg = 0x10044, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x10044, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi1phytimer_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_csi1phytimer_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csi2phytimer_clk = { + .halt_reg = 0x10068, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x10068, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi2phytimer_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_csi2phytimer_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csi3phytimer_clk = { + .halt_reg = 0x1008c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1008c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi3phytimer_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_csi3phytimer_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csi4phytimer_clk = { + .halt_reg = 0x100b0, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x100b0, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csi4phytimer_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_csi4phytimer_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csid_clk = { + .halt_reg = 0x13230, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13230, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csid_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_csid_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csid_csiphy_rx_clk = { + .halt_reg = 0x10024, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x10024, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csid_csiphy_rx_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cphy_rx_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csiphy0_clk = { + .halt_reg = 0x10020, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x10020, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csiphy0_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cphy_rx_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csiphy1_clk = { + .halt_reg = 0x10048, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x10048, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csiphy1_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cphy_rx_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csiphy2_clk = { + .halt_reg = 0x1006c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1006c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csiphy2_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cphy_rx_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csiphy3_clk = { + .halt_reg = 0x10090, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x10090, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csiphy3_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cphy_rx_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_csiphy4_clk = { + .halt_reg = 0x100b4, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x100b4, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_csiphy4_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cphy_rx_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_icp_0_ahb_clk = { + .halt_reg = 0x130d4, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x130d4, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_icp_0_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_slow_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_icp_0_clk = { + .halt_reg = 0x130c0, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x130c0, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_icp_0_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_icp_0_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_icp_1_ahb_clk = { + .halt_reg = 0x1310c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1310c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_icp_1_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_slow_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_icp_1_clk = { + .halt_reg = 0x130f8, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x130f8, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_icp_1_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_icp_1_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_0_main_clk = { + .halt_reg = 0x12034, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12034, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_0_main_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_0_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_0_main_fast_ahb_clk = { + .halt_reg = 0x12054, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12054, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_0_main_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_0_pcp_clk = { + .halt_reg = 0x12058, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12058, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_0_pcp_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_0_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_0_pcp_fast_ahb_clk = { + .halt_reg = 0x12070, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12070, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_0_pcp_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_0_scalar_clk = { + .halt_reg = 0x12074, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12074, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_0_scalar_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_0_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_0_scalar_fast_ahb_clk = { + .halt_reg = 0x1208c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1208c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_0_scalar_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_0_tmc_clk = { + .halt_reg = 0x12090, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12090, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_0_tmc_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_0_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_0_tmc_fast_ahb_clk = { + .halt_reg = 0x120a8, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x120a8, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_0_tmc_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_1_main_clk = { + .halt_reg = 0x120f8, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x120f8, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_1_main_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_1_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_1_main_fast_ahb_clk = { + .halt_reg = 0x12118, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12118, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_1_main_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_1_pcp_clk = { + .halt_reg = 0x1211c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1211c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_1_pcp_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_1_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_1_pcp_fast_ahb_clk = { + .halt_reg = 0x12134, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12134, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_1_pcp_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_1_scalar_clk = { + .halt_reg = 0x12138, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12138, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_1_scalar_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_1_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_1_scalar_fast_ahb_clk = { + .halt_reg = 0x12150, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12150, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_1_scalar_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_1_tmc_clk = { + .halt_reg = 0x12154, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12154, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_1_tmc_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_1_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_1_tmc_fast_ahb_clk = { + .halt_reg = 0x1216c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1216c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_1_tmc_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_2_main_clk = { + .halt_reg = 0x121a4, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x121a4, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_2_main_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_2_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_2_main_fast_ahb_clk = { + .halt_reg = 0x121c4, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x121c4, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_2_main_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_2_pcp_clk = { + .halt_reg = 0x121c8, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x121c8, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_2_pcp_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_2_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_2_pcp_fast_ahb_clk = { + .halt_reg = 0x121e0, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x121e0, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_2_pcp_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_2_scalar_clk = { + .halt_reg = 0x121e4, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x121e4, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_2_scalar_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_2_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_2_scalar_fast_ahb_clk = { + .halt_reg = 0x121fc, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x121fc, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_2_scalar_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_2_tmc_clk = { + .halt_reg = 0x12200, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12200, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_2_tmc_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_2_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_2_tmc_fast_ahb_clk = { + .halt_reg = 0x12218, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12218, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_2_tmc_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_lite_ahb_clk = { + .halt_reg = 0x13054, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13054, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_lite_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_slow_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_lite_clk = { + .halt_reg = 0x1301c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1301c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_lite_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_lite_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_lite_cphy_rx_clk = { + .halt_reg = 0x13050, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13050, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_lite_cphy_rx_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cphy_rx_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ife_lite_csid_clk = { + .halt_reg = 0x1303c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1303c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ife_lite_csid_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_lite_csid_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ipe_0_ahb_clk = { + .halt_reg = 0x11054, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x11054, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ipe_0_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_slow_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ipe_0_clk = { + .halt_reg = 0x11034, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x11034, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ipe_0_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ipe_0_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ipe_0_fast_ahb_clk = { + .halt_reg = 0x11058, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x11058, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ipe_0_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ipe_1_ahb_clk = { + .halt_reg = 0x110b0, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x110b0, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ipe_1_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_slow_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ipe_1_clk = { + .halt_reg = 0x11090, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x11090, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ipe_1_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ipe_1_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_ipe_1_fast_ahb_clk = { + .halt_reg = 0x110b4, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x110b4, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_ipe_1_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_qdss_debug_clk = { + .halt_reg = 0x13348, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13348, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qdss_debug_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_qdss_debug_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_qdss_debug_xo_clk = { + .halt_reg = 0x1334c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1334c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qdss_debug_xo_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_xo_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_qup_ahbm_clk = { + .halt_reg = 0x13300, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13300, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qup_ahbm_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_qup_ahbm_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_qup_core_2x_clk = { + .halt_reg = 0x132b8, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x132b8, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qup_core_2x_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_qup_core_2x_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_qup_core_clk = { + .halt_reg = 0x132d0, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x132d0, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qup_core_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_qup_core_2x_div_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_qup_se_clk = { + .halt_reg = 0x13320, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13320, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_qup_se_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_qup_se_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_sfe_lite_0_clk = { + .halt_reg = 0x1305c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1305c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_sfe_lite_0_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_0_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_sfe_lite_0_fast_ahb_clk = { + .halt_reg = 0x1306c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1306c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_sfe_lite_0_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_sfe_lite_1_clk = { + .halt_reg = 0x13074, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13074, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_sfe_lite_1_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_1_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_sfe_lite_1_fast_ahb_clk = { + .halt_reg = 0x13084, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13084, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_sfe_lite_1_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_sfe_lite_2_clk = { + .halt_reg = 0x1308c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1308c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_sfe_lite_2_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_2_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_sfe_lite_2_fast_ahb_clk = { + .halt_reg = 0x1309c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1309c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_sfe_lite_2_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_sm_obs_clk = { + .halt_reg = 0x1401c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1401c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_sm_obs_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_ahb_clk = { + .halt_reg = 0x131b0, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x131b0, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_slow_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_fast_ahb_clk = { + .halt_reg = 0x131c4, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x131c4, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_fast_ahb_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_fast_ahb_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_ife_0_clk = { + .halt_reg = 0x12048, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x12048, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_ife_0_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_0_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_ife_1_clk = { + .halt_reg = 0x1210c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x1210c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_ife_1_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_1_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_ife_2_clk = { + .halt_reg = 0x121b8, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x121b8, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_ife_2_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_2_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_ife_lite_clk = { + .halt_reg = 0x13020, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13020, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_ife_lite_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_lite_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_ipe_0_clk = { + .halt_reg = 0x11048, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x11048, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_ipe_0_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ipe_0_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_ipe_1_clk = { + .halt_reg = 0x110a4, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x110a4, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_ipe_1_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ipe_1_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_qup_ahbm_clk = { + .halt_reg = 0x13304, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13304, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_qup_ahbm_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_qup_ahbm_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_sfe_lite_0_clk = { + .halt_reg = 0x13060, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13060, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_sfe_lite_0_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_0_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_sfe_lite_1_clk = { + .halt_reg = 0x13078, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13078, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_sfe_lite_1_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_1_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_top_sfe_lite_2_clk = { + .halt_reg = 0x13090, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x13090, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_top_sfe_lite_2_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_ife_2_main_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch cam_cc_tpg_csiphy_rx_clk = { + .halt_reg = 0x100b8, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x100b8, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "cam_cc_tpg_csiphy_rx_clk", + .parent_hws = (const struct clk_hw*[]) { + &cam_cc_cphy_rx_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct gdsc cam_cc_titan_top_gdsc = { + .gdscr = 0x13350, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "cam_cc_titan_top_gdsc", + }, + .pwrsts = PWRSTS_OFF_ON, + .flags = POLL_CFG_GDSCR | RETAIN_FF_ENABLE, +}; + +static struct gdsc cam_cc_ife_0_gdsc = { + .gdscr = 0x12004, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "cam_cc_ife_0_gdsc", + }, + .parent = &cam_cc_titan_top_gdsc.pd, + .pwrsts = PWRSTS_OFF_ON, + .flags = POLL_CFG_GDSCR | RETAIN_FF_ENABLE, +}; + +static struct gdsc cam_cc_ife_1_gdsc = { + .gdscr = 0x120c8, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "cam_cc_ife_1_gdsc", + }, + .parent = &cam_cc_titan_top_gdsc.pd, + .pwrsts = PWRSTS_OFF_ON, + .flags = POLL_CFG_GDSCR | RETAIN_FF_ENABLE, +}; + +static struct gdsc cam_cc_ife_2_gdsc = { + .gdscr = 0x12174, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "cam_cc_ife_2_gdsc", + }, + .parent = &cam_cc_titan_top_gdsc.pd, + .pwrsts = PWRSTS_OFF_ON, + .flags = POLL_CFG_GDSCR | RETAIN_FF_ENABLE, +}; + +static struct gdsc cam_cc_ipe_0_gdsc = { + .gdscr = 0x11004, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "cam_cc_ipe_0_gdsc", + }, + .parent = &cam_cc_titan_top_gdsc.pd, + .pwrsts = PWRSTS_OFF_ON, + .flags = HW_CTRL_TRIGGER | POLL_CFG_GDSCR | RETAIN_FF_ENABLE, +}; + +static struct gdsc cam_cc_ipe_1_gdsc = { + .gdscr = 0x11060, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "cam_cc_ipe_1_gdsc", + }, + .parent = &cam_cc_titan_top_gdsc.pd, + .pwrsts = PWRSTS_OFF_ON, + .flags = HW_CTRL_TRIGGER | POLL_CFG_GDSCR | RETAIN_FF_ENABLE, +}; + +static struct clk_regmap *cam_cc_nord_clocks[] = { + [CAM_CC_CAMNOC_DCD_XO_CLK] = &cam_cc_camnoc_dcd_xo_clk.clkr, + [CAM_CC_CAMNOC_NRT_AXI_CLK] = &cam_cc_camnoc_nrt_axi_clk.clkr, + [CAM_CC_CAMNOC_RT_AXI_CLK] = &cam_cc_camnoc_rt_axi_clk.clkr, + [CAM_CC_CAMNOC_RT_AXI_CLK_SRC] = &cam_cc_camnoc_rt_axi_clk_src.clkr, + [CAM_CC_CCI_0_CLK] = &cam_cc_cci_0_clk.clkr, + [CAM_CC_CCI_0_CLK_SRC] = &cam_cc_cci_0_clk_src.clkr, + [CAM_CC_CCI_1_CLK] = &cam_cc_cci_1_clk.clkr, + [CAM_CC_CCI_1_CLK_SRC] = &cam_cc_cci_1_clk_src.clkr, + [CAM_CC_CCI_2_CLK] = &cam_cc_cci_2_clk.clkr, + [CAM_CC_CCI_2_CLK_SRC] = &cam_cc_cci_2_clk_src.clkr, + [CAM_CC_CCI_3_CLK] = &cam_cc_cci_3_clk.clkr, + [CAM_CC_CCI_3_CLK_SRC] = &cam_cc_cci_3_clk_src.clkr, + [CAM_CC_CCI_4_CLK] = &cam_cc_cci_4_clk.clkr, + [CAM_CC_CCI_4_CLK_SRC] = &cam_cc_cci_4_clk_src.clkr, + [CAM_CC_CCU_FAST_AHB_CLK] = &cam_cc_ccu_fast_ahb_clk.clkr, + [CAM_CC_CPHY_RX_CLK_SRC] = &cam_cc_cphy_rx_clk_src.clkr, + [CAM_CC_CSI0PHYTIMER_CLK] = &cam_cc_csi0phytimer_clk.clkr, + [CAM_CC_CSI0PHYTIMER_CLK_SRC] = &cam_cc_csi0phytimer_clk_src.clkr, + [CAM_CC_CSI1PHYTIMER_CLK] = &cam_cc_csi1phytimer_clk.clkr, + [CAM_CC_CSI1PHYTIMER_CLK_SRC] = &cam_cc_csi1phytimer_clk_src.clkr, + [CAM_CC_CSI2PHYTIMER_CLK] = &cam_cc_csi2phytimer_clk.clkr, + [CAM_CC_CSI2PHYTIMER_CLK_SRC] = &cam_cc_csi2phytimer_clk_src.clkr, + [CAM_CC_CSI3PHYTIMER_CLK] = &cam_cc_csi3phytimer_clk.clkr, + [CAM_CC_CSI3PHYTIMER_CLK_SRC] = &cam_cc_csi3phytimer_clk_src.clkr, + [CAM_CC_CSI4PHYTIMER_CLK] = &cam_cc_csi4phytimer_clk.clkr, + [CAM_CC_CSI4PHYTIMER_CLK_SRC] = &cam_cc_csi4phytimer_clk_src.clkr, + [CAM_CC_CSID_CLK] = &cam_cc_csid_clk.clkr, + [CAM_CC_CSID_CLK_SRC] = &cam_cc_csid_clk_src.clkr, + [CAM_CC_CSID_CSIPHY_RX_CLK] = &cam_cc_csid_csiphy_rx_clk.clkr, + [CAM_CC_CSIPHY0_CLK] = &cam_cc_csiphy0_clk.clkr, + [CAM_CC_CSIPHY1_CLK] = &cam_cc_csiphy1_clk.clkr, + [CAM_CC_CSIPHY2_CLK] = &cam_cc_csiphy2_clk.clkr, + [CAM_CC_CSIPHY3_CLK] = &cam_cc_csiphy3_clk.clkr, + [CAM_CC_CSIPHY4_CLK] = &cam_cc_csiphy4_clk.clkr, + [CAM_CC_FAST_AHB_CLK_SRC] = &cam_cc_fast_ahb_clk_src.clkr, + [CAM_CC_ICP_0_AHB_CLK] = &cam_cc_icp_0_ahb_clk.clkr, + [CAM_CC_ICP_0_CLK] = &cam_cc_icp_0_clk.clkr, + [CAM_CC_ICP_0_CLK_SRC] = &cam_cc_icp_0_clk_src.clkr, + [CAM_CC_ICP_1_AHB_CLK] = &cam_cc_icp_1_ahb_clk.clkr, + [CAM_CC_ICP_1_CLK] = &cam_cc_icp_1_clk.clkr, + [CAM_CC_ICP_1_CLK_SRC] = &cam_cc_icp_1_clk_src.clkr, + [CAM_CC_IFE_0_MAIN_CLK] = &cam_cc_ife_0_main_clk.clkr, + [CAM_CC_IFE_0_MAIN_CLK_SRC] = &cam_cc_ife_0_main_clk_src.clkr, + [CAM_CC_IFE_0_MAIN_FAST_AHB_CLK] = &cam_cc_ife_0_main_fast_ahb_clk.clkr, + [CAM_CC_IFE_0_PCP_CLK] = &cam_cc_ife_0_pcp_clk.clkr, + [CAM_CC_IFE_0_PCP_FAST_AHB_CLK] = &cam_cc_ife_0_pcp_fast_ahb_clk.clkr, + [CAM_CC_IFE_0_SCALAR_CLK] = &cam_cc_ife_0_scalar_clk.clkr, + [CAM_CC_IFE_0_SCALAR_FAST_AHB_CLK] = &cam_cc_ife_0_scalar_fast_ahb_clk.clkr, + [CAM_CC_IFE_0_TMC_CLK] = &cam_cc_ife_0_tmc_clk.clkr, + [CAM_CC_IFE_0_TMC_FAST_AHB_CLK] = &cam_cc_ife_0_tmc_fast_ahb_clk.clkr, + [CAM_CC_IFE_1_MAIN_CLK] = &cam_cc_ife_1_main_clk.clkr, + [CAM_CC_IFE_1_MAIN_CLK_SRC] = &cam_cc_ife_1_main_clk_src.clkr, + [CAM_CC_IFE_1_MAIN_FAST_AHB_CLK] = &cam_cc_ife_1_main_fast_ahb_clk.clkr, + [CAM_CC_IFE_1_PCP_CLK] = &cam_cc_ife_1_pcp_clk.clkr, + [CAM_CC_IFE_1_PCP_FAST_AHB_CLK] = &cam_cc_ife_1_pcp_fast_ahb_clk.clkr, + [CAM_CC_IFE_1_SCALAR_CLK] = &cam_cc_ife_1_scalar_clk.clkr, + [CAM_CC_IFE_1_SCALAR_FAST_AHB_CLK] = &cam_cc_ife_1_scalar_fast_ahb_clk.clkr, + [CAM_CC_IFE_1_TMC_CLK] = &cam_cc_ife_1_tmc_clk.clkr, + [CAM_CC_IFE_1_TMC_FAST_AHB_CLK] = &cam_cc_ife_1_tmc_fast_ahb_clk.clkr, + [CAM_CC_IFE_2_MAIN_CLK] = &cam_cc_ife_2_main_clk.clkr, + [CAM_CC_IFE_2_MAIN_CLK_SRC] = &cam_cc_ife_2_main_clk_src.clkr, + [CAM_CC_IFE_2_MAIN_FAST_AHB_CLK] = &cam_cc_ife_2_main_fast_ahb_clk.clkr, + [CAM_CC_IFE_2_PCP_CLK] = &cam_cc_ife_2_pcp_clk.clkr, + [CAM_CC_IFE_2_PCP_FAST_AHB_CLK] = &cam_cc_ife_2_pcp_fast_ahb_clk.clkr, + [CAM_CC_IFE_2_SCALAR_CLK] = &cam_cc_ife_2_scalar_clk.clkr, + [CAM_CC_IFE_2_SCALAR_FAST_AHB_CLK] = &cam_cc_ife_2_scalar_fast_ahb_clk.clkr, + [CAM_CC_IFE_2_TMC_CLK] = &cam_cc_ife_2_tmc_clk.clkr, + [CAM_CC_IFE_2_TMC_FAST_AHB_CLK] = &cam_cc_ife_2_tmc_fast_ahb_clk.clkr, + [CAM_CC_IFE_LITE_AHB_CLK] = &cam_cc_ife_lite_ahb_clk.clkr, + [CAM_CC_IFE_LITE_CLK] = &cam_cc_ife_lite_clk.clkr, + [CAM_CC_IFE_LITE_CLK_SRC] = &cam_cc_ife_lite_clk_src.clkr, + [CAM_CC_IFE_LITE_CPHY_RX_CLK] = &cam_cc_ife_lite_cphy_rx_clk.clkr, + [CAM_CC_IFE_LITE_CSID_CLK] = &cam_cc_ife_lite_csid_clk.clkr, + [CAM_CC_IFE_LITE_CSID_CLK_SRC] = &cam_cc_ife_lite_csid_clk_src.clkr, + [CAM_CC_IPE_0_AHB_CLK] = &cam_cc_ipe_0_ahb_clk.clkr, + [CAM_CC_IPE_0_CLK] = &cam_cc_ipe_0_clk.clkr, + [CAM_CC_IPE_0_CLK_SRC] = &cam_cc_ipe_0_clk_src.clkr, + [CAM_CC_IPE_0_FAST_AHB_CLK] = &cam_cc_ipe_0_fast_ahb_clk.clkr, + [CAM_CC_IPE_1_AHB_CLK] = &cam_cc_ipe_1_ahb_clk.clkr, + [CAM_CC_IPE_1_CLK] = &cam_cc_ipe_1_clk.clkr, + [CAM_CC_IPE_1_CLK_SRC] = &cam_cc_ipe_1_clk_src.clkr, + [CAM_CC_IPE_1_FAST_AHB_CLK] = &cam_cc_ipe_1_fast_ahb_clk.clkr, + [CAM_CC_PLL0] = &cam_cc_pll0.clkr, + [CAM_CC_PLL0_OUT_EVEN] = &cam_cc_pll0_out_even.clkr, + [CAM_CC_PLL0_OUT_ODD] = &cam_cc_pll0_out_odd.clkr, + [CAM_CC_PLL2] = &cam_cc_pll2.clkr, + [CAM_CC_PLL2_OUT_EVEN] = &cam_cc_pll2_out_even.clkr, + [CAM_CC_PLL3] = &cam_cc_pll3.clkr, + [CAM_CC_PLL3_OUT_EVEN] = &cam_cc_pll3_out_even.clkr, + [CAM_CC_PLL4] = &cam_cc_pll4.clkr, + [CAM_CC_PLL4_OUT_EVEN] = &cam_cc_pll4_out_even.clkr, + [CAM_CC_PLL5] = &cam_cc_pll5.clkr, + [CAM_CC_PLL5_OUT_EVEN] = &cam_cc_pll5_out_even.clkr, + [CAM_CC_PLL6] = &cam_cc_pll6.clkr, + [CAM_CC_PLL6_OUT_EVEN] = &cam_cc_pll6_out_even.clkr, + [CAM_CC_QDSS_DEBUG_CLK] = &cam_cc_qdss_debug_clk.clkr, + [CAM_CC_QDSS_DEBUG_CLK_SRC] = &cam_cc_qdss_debug_clk_src.clkr, + [CAM_CC_QDSS_DEBUG_XO_CLK] = &cam_cc_qdss_debug_xo_clk.clkr, + [CAM_CC_QUP_AHBM_CLK] = &cam_cc_qup_ahbm_clk.clkr, + [CAM_CC_QUP_AHBM_CLK_SRC] = &cam_cc_qup_ahbm_clk_src.clkr, + [CAM_CC_QUP_CORE_2X_CLK] = &cam_cc_qup_core_2x_clk.clkr, + [CAM_CC_QUP_CORE_2X_CLK_SRC] = &cam_cc_qup_core_2x_clk_src.clkr, + [CAM_CC_QUP_CORE_2X_DIV_CLK_SRC] = &cam_cc_qup_core_2x_div_clk_src.clkr, + [CAM_CC_QUP_CORE_CLK] = &cam_cc_qup_core_clk.clkr, + [CAM_CC_QUP_SE_CLK] = &cam_cc_qup_se_clk.clkr, + [CAM_CC_QUP_SE_CLK_SRC] = &cam_cc_qup_se_clk_src.clkr, + [CAM_CC_SFE_LITE_0_CLK] = &cam_cc_sfe_lite_0_clk.clkr, + [CAM_CC_SFE_LITE_0_FAST_AHB_CLK] = &cam_cc_sfe_lite_0_fast_ahb_clk.clkr, + [CAM_CC_SFE_LITE_1_CLK] = &cam_cc_sfe_lite_1_clk.clkr, + [CAM_CC_SFE_LITE_1_FAST_AHB_CLK] = &cam_cc_sfe_lite_1_fast_ahb_clk.clkr, + [CAM_CC_SFE_LITE_2_CLK] = &cam_cc_sfe_lite_2_clk.clkr, + [CAM_CC_SFE_LITE_2_FAST_AHB_CLK] = &cam_cc_sfe_lite_2_fast_ahb_clk.clkr, + [CAM_CC_SLEEP_CLK_SRC] = &cam_cc_sleep_clk_src.clkr, + [CAM_CC_SLOW_AHB_CLK_SRC] = &cam_cc_slow_ahb_clk_src.clkr, + [CAM_CC_SM_OBS_CLK] = &cam_cc_sm_obs_clk.clkr, + [CAM_CC_TOP_AHB_CLK] = &cam_cc_top_ahb_clk.clkr, + [CAM_CC_TOP_FAST_AHB_CLK] = &cam_cc_top_fast_ahb_clk.clkr, + [CAM_CC_TOP_IFE_0_CLK] = &cam_cc_top_ife_0_clk.clkr, + [CAM_CC_TOP_IFE_1_CLK] = &cam_cc_top_ife_1_clk.clkr, + [CAM_CC_TOP_IFE_2_CLK] = &cam_cc_top_ife_2_clk.clkr, + [CAM_CC_TOP_IFE_LITE_CLK] = &cam_cc_top_ife_lite_clk.clkr, + [CAM_CC_TOP_IPE_0_CLK] = &cam_cc_top_ipe_0_clk.clkr, + [CAM_CC_TOP_IPE_1_CLK] = &cam_cc_top_ipe_1_clk.clkr, + [CAM_CC_TOP_QUP_AHBM_CLK] = &cam_cc_top_qup_ahbm_clk.clkr, + [CAM_CC_TOP_SFE_LITE_0_CLK] = &cam_cc_top_sfe_lite_0_clk.clkr, + [CAM_CC_TOP_SFE_LITE_1_CLK] = &cam_cc_top_sfe_lite_1_clk.clkr, + [CAM_CC_TOP_SFE_LITE_2_CLK] = &cam_cc_top_sfe_lite_2_clk.clkr, + [CAM_CC_TPG_CSIPHY_RX_CLK] = &cam_cc_tpg_csiphy_rx_clk.clkr, + [CAM_CC_XO_CLK_SRC] = &cam_cc_xo_clk_src.clkr, +}; + +static struct gdsc *cam_cc_nord_gdscs[] = { + [CAM_CC_TITAN_TOP_GDSC] = &cam_cc_titan_top_gdsc, + [CAM_CC_IFE_0_GDSC] = &cam_cc_ife_0_gdsc, + [CAM_CC_IFE_1_GDSC] = &cam_cc_ife_1_gdsc, + [CAM_CC_IFE_2_GDSC] = &cam_cc_ife_2_gdsc, + [CAM_CC_IPE_0_GDSC] = &cam_cc_ipe_0_gdsc, + [CAM_CC_IPE_1_GDSC] = &cam_cc_ipe_1_gdsc, +}; + +static const struct qcom_reset_map cam_cc_nord_resets[] = { + [CAM_CC_CCU_BCR] = { 0x13298 }, + [CAM_CC_ICP_0_BCR] = { 0x130a0 }, + [CAM_CC_ICP_1_BCR] = { 0x130d8 }, + [CAM_CC_IFE_0_BCR] = { 0x12000 }, + [CAM_CC_IFE_1_BCR] = { 0x120c4 }, + [CAM_CC_IFE_2_BCR] = { 0x12170 }, + [CAM_CC_IPE_0_BCR] = { 0x11000 }, + [CAM_CC_IPE_1_BCR] = { 0x1105c }, + [CAM_CC_QDSS_DEBUG_BCR] = { 0x1332c }, + [CAM_CC_SFE_LITE_0_BCR] = { 0x13058 }, + [CAM_CC_SFE_LITE_1_BCR] = { 0x13070 }, + [CAM_CC_SFE_LITE_2_BCR] = { 0x13088 }, +}; + +static struct clk_alpha_pll *cam_cc_nord_plls[] = { + &cam_cc_pll0, + &cam_cc_pll2, + &cam_cc_pll3, + &cam_cc_pll4, + &cam_cc_pll5, + &cam_cc_pll6, +}; + +static const u32 cam_cc_nord_critical_cbcrs[] = { + 0x13294, /* CAM_CC_CAMNOC_XO_CLK */ + 0x13364, /* CAM_CC_CORE_AHB_CLK */ + 0x13380, /* CAM_CC_GDSC_CLK */ + 0x132e4, /* CAM_CC_QUP_SLEEP_CLK */ + 0x1339c, /* CAM_CC_SLEEP_CLK */ +}; + +static const struct regmap_config cam_cc_nord_regmap_config = { + .reg_bits = 32, + .reg_stride = 4, + .val_bits = 32, + .max_register = 0x17000, + .fast_io = true, +}; + +static const struct qcom_cc_driver_data cam_cc_nord_driver_data = { + .alpha_plls = cam_cc_nord_plls, + .num_alpha_plls = ARRAY_SIZE(cam_cc_nord_plls), + .clk_cbcrs = cam_cc_nord_critical_cbcrs, + .num_clk_cbcrs = ARRAY_SIZE(cam_cc_nord_critical_cbcrs), +}; + +static const struct qcom_cc_desc cam_cc_nord_desc = { + .config = &cam_cc_nord_regmap_config, + .clks = cam_cc_nord_clocks, + .num_clks = ARRAY_SIZE(cam_cc_nord_clocks), + .resets = cam_cc_nord_resets, + .num_resets = ARRAY_SIZE(cam_cc_nord_resets), + .gdscs = cam_cc_nord_gdscs, + .num_gdscs = ARRAY_SIZE(cam_cc_nord_gdscs), + .use_rpm = true, + .driver_data = &cam_cc_nord_driver_data, +}; + +static const struct of_device_id cam_cc_nord_match_table[] = { + { .compatible = "qcom,nord-camcc" }, + { } +}; +MODULE_DEVICE_TABLE(of, cam_cc_nord_match_table); + +static int cam_cc_nord_probe(struct platform_device *pdev) +{ + return qcom_cc_probe(pdev, &cam_cc_nord_desc); +} + +static struct platform_driver cam_cc_nord_driver = { + .probe = cam_cc_nord_probe, + .driver = { + .name = "camcc-nord", + .of_match_table = cam_cc_nord_match_table, + }, +}; + +module_platform_driver(cam_cc_nord_driver); + +MODULE_DESCRIPTION("QTI CAMCC Nord Driver"); +MODULE_LICENSE("GPL"); From 7bc58e0aca2ac92967c265af67d3adc11d61ff6a Mon Sep 17 00:00:00 2001 From: Erikas Bitovtas Date: Fri, 31 Jul 2026 22:08:20 +0200 Subject: [PATCH 361/857] clk: qcom: gcc-msm8939: mark Venus core GDSCs as hardware controlled MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Allow Venus core GDSCs to have their control passed to hardware, so they can be powered on by Venus firmware and explicitly state that the vcodec clocks' halt bit should be checked. Signed-off-by: Erikas Bitovtas Reviewed-by: Konrad Dybcio Reviewed-by: Bryan O'Donoghue [André: Update commit message] Signed-off-by: André Apitzsch Link: https://lore.kernel.org/r/20260731-msm8939-venus-rfc-v10-1-9ac503250fc7@apitzsch.eu Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/gcc-msm8939.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/drivers/clk/qcom/gcc-msm8939.c b/drivers/clk/qcom/gcc-msm8939.c index ffd7f14fcbaf8b..3900a06189ff96 100644 --- a/drivers/clk/qcom/gcc-msm8939.c +++ b/drivers/clk/qcom/gcc-msm8939.c @@ -3665,6 +3665,7 @@ static struct clk_branch gcc_venus0_vcodec0_clk = { static struct clk_branch gcc_venus0_core0_vcodec0_clk = { .halt_reg = 0x4c02c, + .halt_check = BRANCH_HALT, .clkr = { .enable_reg = 0x4c02c, .enable_mask = BIT(0), @@ -3682,6 +3683,7 @@ static struct clk_branch gcc_venus0_core0_vcodec0_clk = { static struct clk_branch gcc_venus0_core1_vcodec0_clk = { .halt_reg = 0x4c034, + .halt_check = BRANCH_HALT, .clkr = { .enable_reg = 0x4c034, .enable_mask = BIT(0), @@ -3754,6 +3756,7 @@ static struct gdsc venus_core0_gdsc = { .pd = { .name = "venus_core0", }, + .flags = HW_CTRL_TRIGGER, .pwrsts = PWRSTS_OFF_ON, }; @@ -3762,6 +3765,7 @@ static struct gdsc venus_core1_gdsc = { .pd = { .name = "venus_core1", }, + .flags = HW_CTRL_TRIGGER, .pwrsts = PWRSTS_OFF_ON, }; From 73df1b79026d71bf5bba04a55266db7bbe744e75 Mon Sep 17 00:00:00 2001 From: Esteban Urrutia Date: Mon, 13 Jul 2026 23:28:17 -0400 Subject: [PATCH 362/857] clk: qcom: dispcc-sm8450: Fix disp_cc_mdss_mdp_clk_src ops If the clock frequency is changed at registration time, a flicker will be visible at boot. Switching to clk_rcg2_shared_no_init_park_ops fixes this. Fixes: 16fb89f92ec4 ("clk: qcom: Add support for Display Clock Controller on SM8450") Reviewed-by: Konrad Dybcio Signed-off-by: Esteban Urrutia Link: https://lore.kernel.org/r/20260713-sm8450-qol-dispcc-v3-1-56fd05822270@proton.me Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/dispcc-sm8450.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/clk/qcom/dispcc-sm8450.c b/drivers/clk/qcom/dispcc-sm8450.c index 3af120e54cddb2..c7e04bd315d5ac 100644 --- a/drivers/clk/qcom/dispcc-sm8450.c +++ b/drivers/clk/qcom/dispcc-sm8450.c @@ -613,7 +613,7 @@ static struct clk_rcg2 disp_cc_mdss_mdp_clk_src = { .parent_data = disp_cc_parent_data_5, .num_parents = ARRAY_SIZE(disp_cc_parent_data_5), .flags = CLK_SET_RATE_PARENT, - .ops = &clk_rcg2_shared_ops, + .ops = &clk_rcg2_shared_no_init_park_ops, }, }; From b1e06ad04819fc4842a96b22b0533a12648aeacc Mon Sep 17 00:00:00 2001 From: Esteban Urrutia Date: Mon, 13 Jul 2026 23:28:18 -0400 Subject: [PATCH 363/857] clk: qcom: dispcc-sm8450: Migrate to qcom_cc_driver_data Migrate to qcom_cc_driver_data, which is used by other clock controller drivers. Reviewed-by: Konrad Dybcio Signed-off-by: Esteban Urrutia Reviewed-by: Dmitry Baryshkov Link: https://lore.kernel.org/r/20260713-sm8450-qol-dispcc-v3-2-56fd05822270@proton.me Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/dispcc-sm8450.c | 38 +++++++++++++++++++++++--------- 1 file changed, 28 insertions(+), 10 deletions(-) diff --git a/drivers/clk/qcom/dispcc-sm8450.c b/drivers/clk/qcom/dispcc-sm8450.c index c7e04bd315d5ac..facbf040ab9a52 100644 --- a/drivers/clk/qcom/dispcc-sm8450.c +++ b/drivers/clk/qcom/dispcc-sm8450.c @@ -1778,6 +1778,29 @@ static const struct regmap_config disp_cc_sm8450_regmap_config = { .fast_io = true, }; +static struct clk_alpha_pll *disp_cc_sm8450_plls[] = { + &disp_cc_pll0, + &disp_cc_pll1, +}; + +static const u32 disp_cc_sm8450_critical_cbcrs[] = { + 0xe05c, /* DISP_CC_XO_CLK */ +}; + +static void disp_cc_sm8450_clk_regs_configure(struct device *dev, struct regmap *regmap) +{ + /* Enable clock gating for MDP clocks */ + regmap_set_bits(regmap, DISP_CC_MISC_CMD, BIT(4)); +} + +static const struct qcom_cc_driver_data disp_cc_sm8450_driver_data = { + .alpha_plls = disp_cc_sm8450_plls, + .num_alpha_plls = ARRAY_SIZE(disp_cc_sm8450_plls), + .clk_cbcrs = disp_cc_sm8450_critical_cbcrs, + .num_clk_cbcrs = ARRAY_SIZE(disp_cc_sm8450_critical_cbcrs), + .clk_regs_configure = disp_cc_sm8450_clk_regs_configure, +}; + static const struct qcom_cc_desc disp_cc_sm8450_desc = { .config = &disp_cc_sm8450_regmap_config, .clks = disp_cc_sm8450_clocks, @@ -1786,6 +1809,7 @@ static const struct qcom_cc_desc disp_cc_sm8450_desc = { .num_resets = ARRAY_SIZE(disp_cc_sm8450_resets), .gdscs = disp_cc_sm8450_gdscs, .num_gdscs = ARRAY_SIZE(disp_cc_sm8450_gdscs), + .driver_data = &disp_cc_sm8450_driver_data, }; static const struct of_device_id disp_cc_sm8450_match_table[] = { @@ -1823,19 +1847,13 @@ static int disp_cc_sm8450_probe(struct platform_device *pdev) disp_cc_pll1.regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_OLE]; disp_cc_pll1.clkr.hw.init = &sm8475_disp_cc_pll1_init; - clk_lucid_ole_pll_configure(&disp_cc_pll0, regmap, &sm8475_disp_cc_pll0_config); - clk_lucid_ole_pll_configure(&disp_cc_pll1, regmap, &sm8475_disp_cc_pll1_config); + disp_cc_pll0.config = &sm8475_disp_cc_pll0_config; + disp_cc_pll1.config = &sm8475_disp_cc_pll1_config; } else { - clk_lucid_evo_pll_configure(&disp_cc_pll0, regmap, &disp_cc_pll0_config); - clk_lucid_evo_pll_configure(&disp_cc_pll1, regmap, &disp_cc_pll1_config); + disp_cc_pll0.config = &disp_cc_pll0_config; + disp_cc_pll1.config = &disp_cc_pll1_config; } - /* Enable clock gating for MDP clocks */ - regmap_update_bits(regmap, DISP_CC_MISC_CMD, 0x10, 0x10); - - /* Keep some clocks always-on */ - qcom_branch_set_clk_en(regmap, 0xe05c); /* DISP_CC_XO_CLK */ - ret = qcom_cc_really_probe(&pdev->dev, &disp_cc_sm8450_desc, regmap); if (ret) goto err_put_rpm; From 2295d6a84179bfe853244c8ef26d6f54b29959cb Mon Sep 17 00:00:00 2001 From: Esteban Urrutia Date: Mon, 13 Jul 2026 23:28:19 -0400 Subject: [PATCH 364/857] clk: qcom: alpha-pll: Check Lucid Ole PLL status before configuring On some platforms such as SM8475, not doing this may result in graphical glitches when the mdss driver takes over. This fixes the aforementioned issue. Fixes: 3132a9a11e57 ("clk: qcom: clk-alpha-pll: Add support for lucid ole pll configure") Signed-off-by: Esteban Urrutia Reviewed-by: Konrad Dybcio Suggested-by: Konrad Dybcio Link: https://lore.kernel.org/r/20260713-sm8450-qol-dispcc-v3-3-56fd05822270@proton.me Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/clk-alpha-pll.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/drivers/clk/qcom/clk-alpha-pll.c b/drivers/clk/qcom/clk-alpha-pll.c index f8313f9d0e30ff..60173b076cc5f1 100644 --- a/drivers/clk/qcom/clk-alpha-pll.c +++ b/drivers/clk/qcom/clk-alpha-pll.c @@ -2373,6 +2373,15 @@ void clk_lucid_ole_pll_configure(struct clk_alpha_pll *pll, struct regmap *regma { u32 lval = config->l; + /* + * If the bootloader left the PLL enabled it's likely that there are + * RCGs that will lock up if we disable the PLL below. + */ + if (trion_pll_is_enabled(pll, regmap)) { + pr_debug("Lucid Ole PLL is already enabled, skipping configuration\n"); + return; + } + lval |= TRION_PLL_CAL_VAL << LUCID_EVO_PLL_CAL_L_VAL_SHIFT; lval |= TRION_PLL_CAL_VAL << LUCID_OLE_PLL_RINGOSC_CAL_L_VAL_SHIFT; clk_alpha_pll_write_config(regmap, PLL_L_VAL(pll), lval); From 9e697dcb8cc01f708cacc4ef3d8b6fb8a67dcd54 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Thomas=20Wei=C3=9Fschuh?= Date: Mon, 31 Aug 2026 18:04:59 +0200 Subject: [PATCH 365/857] tools/nolibc: verify that a directory is opened MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit If a non-directory is opened, ENODIR should be returned from opendir()/fdopendir() right away and not only during readdir_r(). Validate the type of opened file during open. Signed-off-by: Thomas Weißschuh Reviewed-by: Willy Tarreau Link: https://patch.msgid.link/20260831-nolibc-fdopendir-enotdir-v1-1-8cf0e79c4e6f@weissschuh.net --- tools/include/nolibc/dirent.h | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/tools/include/nolibc/dirent.h b/tools/include/nolibc/dirent.h index 4e02ef25e72d84..25fbff20899815 100644 --- a/tools/include/nolibc/dirent.h +++ b/tools/include/nolibc/dirent.h @@ -30,10 +30,23 @@ typedef struct { static __attribute__((unused)) DIR *fdopendir(int fd) { + struct stat buf; + int ret; + if (fd < 0) { SET_ERRNO(EBADF); return NULL; } + + ret = fstat(fd, &buf); + if (ret < 0) + return NULL; + + if (!S_ISDIR(buf.st_mode)) { + SET_ERRNO(ENOTDIR); + return NULL; + } + return (DIR *)(intptr_t)~fd; } From 6a6c14d2664e441f016b3b5f0d0b77d4aca2f54c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Thomas=20Wei=C3=9Fschuh?= Date: Mon, 31 Aug 2026 18:05:00 +0200 Subject: [PATCH 366/857] selftests/nolibc: validate ENOTDIR return values from opendir()/fdopendir() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Make sure that ENOTDIR is detected already during directory opening. Signed-off-by: Thomas Weißschuh Reviewed-by: Willy Tarreau Link: https://patch.msgid.link/20260831-nolibc-fdopendir-enotdir-v1-2-8cf0e79c4e6f@weissschuh.net --- tools/testing/selftests/nolibc/nolibc-test.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/nolibc/nolibc-test.c b/tools/testing/selftests/nolibc/nolibc-test.c index 0a8fe5100b7f58..7091f63f3b2509 100644 --- a/tools/testing/selftests/nolibc/nolibc-test.c +++ b/tools/testing/selftests/nolibc/nolibc-test.c @@ -1711,6 +1711,7 @@ int run_syscall(int min, int max) CASE_TEST(execve_root); EXPECT_SYSER(1, execve("/", (char*[]){ [0] = (char []){"/"}, [1] = NULL }, NULL), -1, EACCES); break; CASE_TEST(fchdir_stdin); EXPECT_SYSER(1, fchdir(STDIN_FILENO), -1, ENOTDIR); break; CASE_TEST(fchdir_badfd); EXPECT_SYSER(1, fchdir(-1), -1, EBADF); break; + CASE_TEST(fdopendir_notdir); EXPECT_SYSER(1, (uintptr_t)fdopendir(STDIN_FILENO), (uintptr_t)NULL, ENOTDIR); break; CASE_TEST(file_stream); EXPECT_SYSZR(1, test_file_stream()); break; CASE_TEST(file_stream_wsr); EXPECT_SYSZR(1, test_file_stream_wsr()); break; CASE_TEST(fork); EXPECT_SYSZR(1, test_fork(FORK_STANDARD)); break; @@ -1739,6 +1740,7 @@ int run_syscall(int min, int max) CASE_TEST(open_blah); EXPECT_SYSER(1, tmp = open("/proc/self/blah", O_RDONLY), -1, ENOENT); if (tmp != -1) close(tmp); break; CASE_TEST(openat_dir); EXPECT_SYSZR(1, test_openat()); break; CASE_TEST(open_mode); EXPECT_SYSZR(1, test_open_mode()); break; + CASE_TEST(opendir_notdir); EXPECT_SYSER(1, (uintptr_t)opendir("/dev/stdin"), (uintptr_t)NULL, ENOTDIR); break; CASE_TEST(pipe); EXPECT_SYSZR(1, test_pipe()); break; CASE_TEST(poll_null); EXPECT_SYSZR(1, poll(NULL, 0, 0)); break; CASE_TEST(poll_stdout); EXPECT_SYSNE(1, ({ struct pollfd fds = { 1, POLLOUT, 0}; poll(&fds, 1, 0); }), -1); break; From 6ef847683e7d294a4187a7253924b8f0d8911416 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Thomas=20Wei=C3=9Fschuh?= Date: Mon, 31 Aug 2026 18:05:01 +0200 Subject: [PATCH 367/857] tools/nolibc: validate directory with O_DIRECTORY in opendir() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fdopendir() requires a call to fstat() to determine if the opened file is a directory. Currently opendir() inherits this extra syscall. Switch to O_DIRECTORY and remove the call to fdopendir() in opendir() to make the directory type check cheaper. Signed-off-by: Thomas Weißschuh Reviewed-by: Willy Tarreau Link: https://patch.msgid.link/20260831-nolibc-fdopendir-enotdir-v1-3-8cf0e79c4e6f@weissschuh.net --- tools/include/nolibc/dirent.h | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/tools/include/nolibc/dirent.h b/tools/include/nolibc/dirent.h index 25fbff20899815..2dbf4052b85a0b 100644 --- a/tools/include/nolibc/dirent.h +++ b/tools/include/nolibc/dirent.h @@ -55,10 +55,11 @@ DIR *opendir(const char *name) { int fd; - fd = open(name, O_RDONLY); + fd = open(name, O_RDONLY | O_DIRECTORY); if (fd == -1) return NULL; - return fdopendir(fd); + + return (DIR *)(intptr_t)~fd; } static __attribute__((unused)) From e304e8919043712191462224ce9d40bb5f5a6bc1 Mon Sep 17 00:00:00 2001 From: Hardeep Sharma Date: Thu, 27 Aug 2026 23:21:07 +0530 Subject: [PATCH 368/857] dt-bindings: clock: qcom,rpmhcc: Add Kuno RPMh clock controller Document the RPMh clock controller compatible for the Qualcomm Kuno SoC. Acked-by: Krzysztof Kozlowski Signed-off-by: Hardeep Sharma Link: https://lore.kernel.org/r/20260827-kuno-soc-support-v5-4-6d47636a8f09@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- Documentation/devicetree/bindings/clock/qcom,rpmhcc.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/Documentation/devicetree/bindings/clock/qcom,rpmhcc.yaml b/Documentation/devicetree/bindings/clock/qcom,rpmhcc.yaml index 2b446aca5207c9..9d521ac9c33f5c 100644 --- a/Documentation/devicetree/bindings/clock/qcom,rpmhcc.yaml +++ b/Documentation/devicetree/bindings/clock/qcom,rpmhcc.yaml @@ -22,6 +22,7 @@ properties: - qcom,glymur-rpmh-clk - qcom,hawi-rpmh-clk - qcom,kaanapali-rpmh-clk + - qcom,kuno-rpmh-clk - qcom,milos-rpmh-clk - qcom,nord-rpmh-clk - qcom,qcs615-rpmh-clk From 12918d595ac96b819b9c3ba0df4c6a4bf50c43a9 Mon Sep 17 00:00:00 2001 From: Hardeep Sharma Date: Thu, 27 Aug 2026 23:21:08 +0530 Subject: [PATCH 369/857] clk: qcom: clk-rpmh: Add support for Kuno RPMh clocks Add the RPMh clock description for the Qualcomm Kuno SoC so the qcom,kuno-rpmh-clk compatible can provide the RPMh-managed clocks to consumers: the CXO div-2 (and always-on variant), RF clock 1 (and always-on variant), the QPIC BCM clock and the IPA clock. Reviewed-by: Dmitry Baryshkov Reviewed-by: Konrad Dybcio Reviewed-by: Abel Vesa Signed-off-by: Hardeep Sharma Link: https://lore.kernel.org/r/20260827-kuno-soc-support-v5-5-6d47636a8f09@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/clk-rpmh.c | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/drivers/clk/qcom/clk-rpmh.c b/drivers/clk/qcom/clk-rpmh.c index a224a94ae27337..ec8700b6186d8d 100644 --- a/drivers/clk/qcom/clk-rpmh.c +++ b/drivers/clk/qcom/clk-rpmh.c @@ -937,6 +937,20 @@ static const struct clk_rpmh_desc clk_rpmh_kaanapali = { .num_clks = ARRAY_SIZE(kaanapali_rpmh_clocks), }; +static struct clk_hw *kuno_rpmh_clocks[] = { + [RPMH_CXO_CLK] = &clk_rpmh_bi_tcxo_div2.hw, + [RPMH_CXO_CLK_A] = &clk_rpmh_bi_tcxo_div2_ao.hw, + [RPMH_RF_CLK1] = &clk_rpmh_rf_clk1_a.hw, + [RPMH_RF_CLK1_A] = &clk_rpmh_rf_clk1_a_ao.hw, + [RPMH_QPIC_CLK] = &clk_rpmh_qpic_clk.hw, + [RPMH_IPA_CLK] = &clk_rpmh_ipa.hw, +}; + +static const struct clk_rpmh_desc clk_rpmh_kuno = { + .clks = kuno_rpmh_clocks, + .num_clks = ARRAY_SIZE(kuno_rpmh_clocks), +}; + static struct clk_hw *eliza_rpmh_clocks[] = { [RPMH_CXO_CLK] = &clk_rpmh_bi_tcxo_div2.hw, [RPMH_CXO_CLK_A] = &clk_rpmh_bi_tcxo_div2_ao.hw, @@ -1097,6 +1111,7 @@ static const struct of_device_id clk_rpmh_match_table[] = { { .compatible = "qcom,glymur-rpmh-clk", .data = &clk_rpmh_glymur}, { .compatible = "qcom,hawi-rpmh-clk", .data = &clk_rpmh_hawi}, { .compatible = "qcom,kaanapali-rpmh-clk", .data = &clk_rpmh_kaanapali}, + { .compatible = "qcom,kuno-rpmh-clk", .data = &clk_rpmh_kuno}, { .compatible = "qcom,milos-rpmh-clk", .data = &clk_rpmh_milos}, { .compatible = "qcom,nord-rpmh-clk", .data = &clk_rpmh_nord}, { .compatible = "qcom,qcs615-rpmh-clk", .data = &clk_rpmh_qcs615}, From 7db06b5f624b583fa2206f98de8437819acada9b Mon Sep 17 00:00:00 2001 From: Hardeep Sharma Date: Thu, 27 Aug 2026 23:21:14 +0530 Subject: [PATCH 370/857] clk: qcom: Add Global Clock Controller driver for Kuno Add the global clock controller (GCC) driver for the Qualcomm Kuno SoC, providing the PLLs, root clock generators, gate/branch clocks and resets used by the peripheral devices such as UART, SPI, I2C, USB, SD, PCIe and Ethernet. Reviewed-by: Konrad Dybcio Reviewed-by: Abel Vesa Signed-off-by: Hardeep Sharma Link: https://lore.kernel.org/r/20260827-kuno-soc-support-v5-11-6d47636a8f09@oss.qualcomm.com Signed-off-by: Bjorn Andersson --- drivers/clk/qcom/Kconfig | 11 + drivers/clk/qcom/Makefile | 1 + drivers/clk/qcom/gcc-kuno.c | 1483 +++++++++++++++++++++++++++++++++++ 3 files changed, 1495 insertions(+) create mode 100644 drivers/clk/qcom/gcc-kuno.c diff --git a/drivers/clk/qcom/Kconfig b/drivers/clk/qcom/Kconfig index b373ecbb2befb4..6ef083f5d9e251 100644 --- a/drivers/clk/qcom/Kconfig +++ b/drivers/clk/qcom/Kconfig @@ -212,6 +212,17 @@ config CLK_KAANAPALI_VIDEOCC Say Y if you want to support video devices and functionality such as video encode/decode. +config CLK_KUNO_GCC + tristate "Kuno Global Clock Controller" + depends on ARM || COMPILE_TEST + select QCOM_GDSC + default ARCH_QCOM + help + Support for the global clock controller (GCC) on Kuno devices. + Say Y if you want to use peripheral devices such as UART, SPI, + I2C, USB, SD, PCIe and Ethernet on the Kuno SoC. This clock + controller supplies the clocks and resets to those peripherals. + config CLK_MAILI_VIDEOCC tristate "Maili Video Clock Controller" depends on ARM64 || COMPILE_TEST diff --git a/drivers/clk/qcom/Makefile b/drivers/clk/qcom/Makefile index a7674aef8a67f2..a6c676f68b79e4 100644 --- a/drivers/clk/qcom/Makefile +++ b/drivers/clk/qcom/Makefile @@ -45,6 +45,7 @@ obj-$(CONFIG_CLK_KAANAPALI_GCC) += gcc-kaanapali.o obj-$(CONFIG_CLK_KAANAPALI_GPUCC) += gpucc-kaanapali.o gxclkctl-kaanapali.o obj-$(CONFIG_CLK_KAANAPALI_TCSRCC) += tcsrcc-kaanapali.o obj-$(CONFIG_CLK_KAANAPALI_VIDEOCC) += videocc-kaanapali.o +obj-$(CONFIG_CLK_KUNO_GCC) += gcc-kuno.o obj-$(CONFIG_CLK_MAILI_VIDEOCC) += videocc-maili.o obj-$(CONFIG_CLK_NORD_CAMCC) += camcc-nord.o obj-$(CONFIG_CLK_NORD_DISPCC) += dispcc0-nord.o dispcc1-nord.o diff --git a/drivers/clk/qcom/gcc-kuno.c b/drivers/clk/qcom/gcc-kuno.c new file mode 100644 index 00000000000000..787111397df969 --- /dev/null +++ b/drivers/clk/qcom/gcc-kuno.c @@ -0,0 +1,1483 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) Qualcomm Technologies, Inc. and/or its subsidiaries. + */ + +#include +#include +#include +#include + +#include + +#include "clk-alpha-pll.h" +#include "clk-branch.h" +#include "clk-rcg.h" +#include "clk-regmap.h" +#include "clk-regmap-divider.h" +#include "clk-regmap-mux.h" +#include "clk-regmap-phy-mux.h" +#include "common.h" +#include "gdsc.h" +#include "reset.h" + +/* Need to match the order of clocks in DT binding */ +enum { + DT_BI_TCXO, + DT_BI_TCXO_AO, + DT_SLEEP_CLK, + DT_PCIE_PIPE_CLK, +}; + +enum { + P_BI_TCXO, + P_GPLL0_OUT_EVEN, + P_GPLL0_OUT_MAIN, + P_GPLL2_OUT_MAIN, + P_GPLL3_OUT_MAIN, + P_GPLL4_OUT_EVEN, + P_GPLL4_OUT_MAIN, + P_SLEEP_CLK, +}; + +static struct clk_alpha_pll gpll0 = { + .offset = 0x0, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_EVO], + .clkr = { + .enable_reg = 0x7d000, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gpll0", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_fixed_lucid_evo_ops, + }, + }, +}; + +static const struct clk_div_table post_div_table_gpll0_out_even[] = { + { 0x1, 2 }, + { } +}; + +static struct clk_alpha_pll_postdiv gpll0_out_even = { + .offset = 0x0, + .post_div_shift = 10, + .post_div_table = post_div_table_gpll0_out_even, + .num_post_div = ARRAY_SIZE(post_div_table_gpll0_out_even), + .width = 4, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_EVO], + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gpll0_out_even", + .parent_hws = (const struct clk_hw*[]) { + &gpll0.clkr.hw, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_postdiv_lucid_evo_ops, + }, +}; + +static struct clk_alpha_pll gpll2 = { + .offset = 0x2000, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_EVO], + .clkr = { + .enable_reg = 0x7d000, + .enable_mask = BIT(2), + .hw.init = &(const struct clk_init_data) { + .name = "gpll2", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_fixed_lucid_evo_ops, + }, + }, +}; + +static struct clk_alpha_pll gpll3 = { + .offset = 0x3000, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_EVO], + .clkr = { + .enable_reg = 0x7d000, + .enable_mask = BIT(3), + .hw.init = &(const struct clk_init_data) { + .name = "gpll3", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_fixed_lucid_evo_ops, + }, + }, +}; + +static struct clk_alpha_pll gpll4 = { + .offset = 0x4000, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_EVO], + .clkr = { + .enable_reg = 0x7d000, + .enable_mask = BIT(4), + .hw.init = &(const struct clk_init_data) { + .name = "gpll4", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_fixed_lucid_evo_ops, + }, + }, +}; + +static const struct clk_div_table post_div_table_gpll4_out_even[] = { + { 0x1, 2 }, + { } +}; + +static struct clk_alpha_pll_postdiv gpll4_out_even = { + .offset = 0x4000, + .post_div_shift = 10, + .post_div_table = post_div_table_gpll4_out_even, + .num_post_div = ARRAY_SIZE(post_div_table_gpll4_out_even), + .width = 4, + .regs = clk_alpha_pll_regs[CLK_ALPHA_PLL_TYPE_LUCID_EVO], + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gpll4_out_even", + .parent_hws = (const struct clk_hw*[]) { + &gpll4.clkr.hw, + }, + .num_parents = 1, + .ops = &clk_alpha_pll_postdiv_lucid_evo_ops, + }, +}; + +static const struct parent_map gcc_parent_map_0[] = { + { P_BI_TCXO, 0 }, + { P_GPLL0_OUT_MAIN, 1 }, + { P_GPLL4_OUT_EVEN, 2 }, + { P_GPLL0_OUT_EVEN, 6 }, +}; + +static const struct clk_parent_data gcc_parent_data_0[] = { + { .index = DT_BI_TCXO }, + { .hw = &gpll0.clkr.hw }, + { .hw = &gpll4_out_even.clkr.hw }, + { .hw = &gpll0_out_even.clkr.hw }, +}; + +static const struct parent_map gcc_parent_map_1[] = { + { P_BI_TCXO, 0 }, + { P_GPLL0_OUT_MAIN, 1 }, + { P_GPLL0_OUT_EVEN, 6 }, +}; + +static const struct clk_parent_data gcc_parent_data_1[] = { + { .index = DT_BI_TCXO }, + { .hw = &gpll0.clkr.hw }, + { .hw = &gpll0_out_even.clkr.hw }, +}; + +static const struct parent_map gcc_parent_map_2[] = { + { P_BI_TCXO, 0 }, + { P_GPLL0_OUT_MAIN, 1 }, + { P_GPLL4_OUT_EVEN, 2 }, + { P_SLEEP_CLK, 5 }, + { P_GPLL0_OUT_EVEN, 6 }, +}; + +static const struct clk_parent_data gcc_parent_data_2[] = { + { .index = DT_BI_TCXO }, + { .hw = &gpll0.clkr.hw }, + { .hw = &gpll4_out_even.clkr.hw }, + { .index = DT_SLEEP_CLK }, + { .hw = &gpll0_out_even.clkr.hw }, +}; + +static const struct parent_map gcc_parent_map_3[] = { + { P_BI_TCXO, 0 }, + { P_GPLL0_OUT_MAIN, 1 }, + { P_GPLL4_OUT_EVEN, 2 }, + { P_GPLL4_OUT_MAIN, 3 }, + { P_GPLL2_OUT_MAIN, 4 }, + { P_GPLL3_OUT_MAIN, 5 }, + { P_GPLL0_OUT_EVEN, 6 }, +}; + +static const struct clk_parent_data gcc_parent_data_3[] = { + { .index = DT_BI_TCXO }, + { .hw = &gpll0.clkr.hw }, + { .hw = &gpll4_out_even.clkr.hw }, + { .hw = &gpll4.clkr.hw }, + { .hw = &gpll2.clkr.hw }, + { .hw = &gpll3.clkr.hw }, + { .hw = &gpll0_out_even.clkr.hw }, +}; + +static const struct parent_map gcc_parent_map_4[] = { + { P_BI_TCXO, 0 }, + { P_SLEEP_CLK, 5 }, +}; + +static const struct clk_parent_data gcc_parent_data_4[] = { + { .index = DT_BI_TCXO }, + { .index = DT_SLEEP_CLK }, +}; + +static const struct parent_map gcc_parent_map_5[] = { + { P_BI_TCXO, 0 }, + { P_GPLL0_OUT_MAIN, 1 }, + { P_SLEEP_CLK, 5 }, +}; + +static const struct clk_parent_data gcc_parent_data_5[] = { + { .index = DT_BI_TCXO }, + { .hw = &gpll0.clkr.hw }, + { .index = DT_SLEEP_CLK }, +}; + +static const struct parent_map gcc_parent_map_6[] = { + { P_BI_TCXO, 2 }, +}; + +static struct clk_regmap_mux gcc_pcie_aux_clk_src = { + .reg = 0x5308c, + .shift = 0, + .width = 2, + .parent_map = gcc_parent_map_6, + .clkr = { + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_aux_clk_src", + .parent_data = &(const struct clk_parent_data) { + .index = DT_BI_TCXO, + }, + .num_parents = 1, + .ops = &clk_regmap_mux_closest_ops, + }, + }, +}; + +static struct clk_regmap_phy_mux gcc_pcie_pipe_clk_src = { + .reg = 0x53070, + .clkr = { + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_pipe_clk_src", + .parent_data = &(const struct clk_parent_data) { + .index = DT_PCIE_PIPE_CLK, + }, + .num_parents = 1, + .ops = &clk_regmap_phy_mux_ops, + }, + }, +}; + +static const struct freq_tbl ftbl_gcc_emac0_phy_aux_clk_src[] = { + F(19200000, P_BI_TCXO, 1, 0, 0), + { } +}; + +static struct clk_rcg2 gcc_emac0_phy_aux_clk_src = { + .cmd_rcgr = 0x7102c, + .mnd_width = 0, + .hid_width = 5, + .parent_map = gcc_parent_map_5, + .freq_tbl = ftbl_gcc_emac0_phy_aux_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_emac0_phy_aux_clk_src", + .parent_data = gcc_parent_data_5, + .num_parents = ARRAY_SIZE(gcc_parent_data_5), + .ops = &clk_rcg2_ops, + }, +}; + +static const struct freq_tbl ftbl_gcc_emac0_ptp_clk_src[] = { + F(62500000, P_GPLL4_OUT_EVEN, 4, 0, 0), + F(75000000, P_GPLL0_OUT_EVEN, 4, 0, 0), + F(125000000, P_GPLL2_OUT_MAIN, 4, 0, 0), + F(230400000, P_GPLL3_OUT_MAIN, 3.5, 0, 0), + { } +}; + +static struct clk_rcg2 gcc_emac0_ptp_clk_src = { + .cmd_rcgr = 0x71064, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_3, + .freq_tbl = ftbl_gcc_emac0_ptp_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_emac0_ptp_clk_src", + .parent_data = gcc_parent_data_3, + .num_parents = ARRAY_SIZE(gcc_parent_data_3), + .ops = &clk_rcg2_ops, + }, +}; + +static const struct freq_tbl ftbl_gcc_emac0_rgmii_clk_src[] = { + F(41666667, P_GPLL4_OUT_EVEN, 6, 0, 0), + F(50000000, P_GPLL0_OUT_EVEN, 6, 0, 0), + F(125000000, P_GPLL2_OUT_MAIN, 4, 0, 0), + F(250000000, P_GPLL2_OUT_MAIN, 2, 0, 0), + { } +}; + +static struct clk_rcg2 gcc_emac0_rgmii_clk_src = { + .cmd_rcgr = 0x7104c, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_3, + .freq_tbl = ftbl_gcc_emac0_rgmii_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_emac0_rgmii_clk_src", + .parent_data = gcc_parent_data_3, + .num_parents = ARRAY_SIZE(gcc_parent_data_3), + .ops = &clk_rcg2_ops, + }, +}; + +static const struct freq_tbl ftbl_gcc_gp1_clk_src[] = { + F(19200000, P_BI_TCXO, 1, 0, 0), + F(41666667, P_GPLL4_OUT_EVEN, 6, 0, 0), + F(50000000, P_GPLL0_OUT_EVEN, 6, 0, 0), + F(100000000, P_GPLL0_OUT_MAIN, 6, 0, 0), + F(200000000, P_GPLL0_OUT_MAIN, 3, 0, 0), + { } +}; + +static struct clk_rcg2 gcc_gp1_clk_src = { + .cmd_rcgr = 0x47004, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_2, + .freq_tbl = ftbl_gcc_gp1_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_gp1_clk_src", + .parent_data = gcc_parent_data_2, + .num_parents = ARRAY_SIZE(gcc_parent_data_2), + .ops = &clk_rcg2_ops, + }, +}; + +static struct clk_rcg2 gcc_gp2_clk_src = { + .cmd_rcgr = 0x48004, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_2, + .freq_tbl = ftbl_gcc_gp1_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_gp2_clk_src", + .parent_data = gcc_parent_data_2, + .num_parents = ARRAY_SIZE(gcc_parent_data_2), + .ops = &clk_rcg2_ops, + }, +}; + +static struct clk_rcg2 gcc_gp3_clk_src = { + .cmd_rcgr = 0x49004, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_2, + .freq_tbl = ftbl_gcc_gp1_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_gp3_clk_src", + .parent_data = gcc_parent_data_2, + .num_parents = ARRAY_SIZE(gcc_parent_data_2), + .ops = &clk_rcg2_ops, + }, +}; + +static struct clk_rcg2 gcc_pcie_aux_phy_clk_src = { + .cmd_rcgr = 0x53074, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_4, + .freq_tbl = ftbl_gcc_emac0_phy_aux_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_aux_phy_clk_src", + .parent_data = gcc_parent_data_4, + .num_parents = ARRAY_SIZE(gcc_parent_data_4), + .ops = &clk_rcg2_ops, + }, +}; + +static const struct freq_tbl ftbl_gcc_pcie_rchng_phy_clk_src[] = { + F(19200000, P_BI_TCXO, 1, 0, 0), + F(83333333, P_GPLL4_OUT_EVEN, 3, 0, 0), + F(100000000, P_GPLL0_OUT_EVEN, 3, 0, 0), + { } +}; + +static struct clk_rcg2 gcc_pcie_rchng_phy_clk_src = { + .cmd_rcgr = 0x53038, + .mnd_width = 0, + .hid_width = 5, + .parent_map = gcc_parent_map_2, + .freq_tbl = ftbl_gcc_pcie_rchng_phy_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_rchng_phy_clk_src", + .parent_data = gcc_parent_data_2, + .num_parents = ARRAY_SIZE(gcc_parent_data_2), + .ops = &clk_rcg2_ops, + }, +}; + +static const struct freq_tbl ftbl_gcc_pdm2_clk_src[] = { + F(19200000, P_BI_TCXO, 1, 0, 0), + F(60000000, P_GPLL0_OUT_MAIN, 10, 0, 0), + { } +}; + +static struct clk_rcg2 gcc_pdm2_clk_src = { + .cmd_rcgr = 0x34010, + .mnd_width = 0, + .hid_width = 5, + .parent_map = gcc_parent_map_1, + .freq_tbl = ftbl_gcc_pdm2_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_pdm2_clk_src", + .parent_data = gcc_parent_data_1, + .num_parents = ARRAY_SIZE(gcc_parent_data_1), + .ops = &clk_rcg2_ops, + }, +}; + +static const struct freq_tbl ftbl_gcc_qupv3_wrap0_s0_clk_src[] = { + F(7372800, P_GPLL0_OUT_EVEN, 1, 384, 15625), + F(14745600, P_GPLL0_OUT_EVEN, 1, 768, 15625), + F(19200000, P_BI_TCXO, 1, 0, 0), + F(29491200, P_GPLL0_OUT_EVEN, 1, 1536, 15625), + F(32000000, P_GPLL0_OUT_EVEN, 1, 8, 75), + F(48000000, P_GPLL0_OUT_EVEN, 1, 4, 25), + F(62500000, P_GPLL4_OUT_EVEN, 4, 0, 0), + F(64000000, P_GPLL0_OUT_EVEN, 1, 16, 75), + F(75000000, P_GPLL0_OUT_EVEN, 4, 0, 0), + F(80000000, P_GPLL0_OUT_EVEN, 1, 4, 15), + F(96000000, P_GPLL0_OUT_EVEN, 1, 8, 25), + F(100000000, P_GPLL0_OUT_MAIN, 6, 0, 0), + { } +}; + +static struct clk_init_data gcc_qupv3_wrap0_s0_clk_src_init = { + .name = "gcc_qupv3_wrap0_s0_clk_src", + .parent_data = gcc_parent_data_0, + .num_parents = ARRAY_SIZE(gcc_parent_data_0), + .ops = &clk_rcg2_ops, +}; + +static struct clk_rcg2 gcc_qupv3_wrap0_s0_clk_src = { + .cmd_rcgr = 0x6c004, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_0, + .freq_tbl = ftbl_gcc_qupv3_wrap0_s0_clk_src, + .clkr.hw.init = &gcc_qupv3_wrap0_s0_clk_src_init, +}; + +static struct clk_init_data gcc_qupv3_wrap0_s1_clk_src_init = { + .name = "gcc_qupv3_wrap0_s1_clk_src", + .parent_data = gcc_parent_data_0, + .num_parents = ARRAY_SIZE(gcc_parent_data_0), + .ops = &clk_rcg2_ops, +}; + +static struct clk_rcg2 gcc_qupv3_wrap0_s1_clk_src = { + .cmd_rcgr = 0x6c13c, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_0, + .freq_tbl = ftbl_gcc_qupv3_wrap0_s0_clk_src, + .clkr.hw.init = &gcc_qupv3_wrap0_s1_clk_src_init, +}; + +static struct clk_init_data gcc_qupv3_wrap0_s2_clk_src_init = { + .name = "gcc_qupv3_wrap0_s2_clk_src", + .parent_data = gcc_parent_data_0, + .num_parents = ARRAY_SIZE(gcc_parent_data_0), + .ops = &clk_rcg2_ops, +}; + +static struct clk_rcg2 gcc_qupv3_wrap0_s2_clk_src = { + .cmd_rcgr = 0x6c274, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_0, + .freq_tbl = ftbl_gcc_qupv3_wrap0_s0_clk_src, + .clkr.hw.init = &gcc_qupv3_wrap0_s2_clk_src_init, +}; + +static struct clk_init_data gcc_qupv3_wrap0_s3_clk_src_init = { + .name = "gcc_qupv3_wrap0_s3_clk_src", + .parent_data = gcc_parent_data_0, + .num_parents = ARRAY_SIZE(gcc_parent_data_0), + .ops = &clk_rcg2_ops, +}; + +static struct clk_rcg2 gcc_qupv3_wrap0_s3_clk_src = { + .cmd_rcgr = 0x6c3ac, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_0, + .freq_tbl = ftbl_gcc_qupv3_wrap0_s0_clk_src, + .clkr.hw.init = &gcc_qupv3_wrap0_s3_clk_src_init, +}; + +static struct clk_init_data gcc_qupv3_wrap0_s4_clk_src_init = { + .name = "gcc_qupv3_wrap0_s4_clk_src", + .parent_data = gcc_parent_data_0, + .num_parents = ARRAY_SIZE(gcc_parent_data_0), + .ops = &clk_rcg2_ops, +}; + +static struct clk_rcg2 gcc_qupv3_wrap0_s4_clk_src = { + .cmd_rcgr = 0x6c4e4, + .mnd_width = 16, + .hid_width = 5, + .parent_map = gcc_parent_map_0, + .freq_tbl = ftbl_gcc_qupv3_wrap0_s0_clk_src, + .clkr.hw.init = &gcc_qupv3_wrap0_s4_clk_src_init, +}; + +static const struct freq_tbl ftbl_gcc_sdcc4_apps_clk_src[] = { + F(400000, P_BI_TCXO, 12, 1, 4), + F(19200000, P_BI_TCXO, 1, 0, 0), + F(25000000, P_GPLL0_OUT_EVEN, 12, 1, 1), + F(50000000, P_GPLL0_OUT_EVEN, 6, 0, 0), + F(83333333, P_GPLL4_OUT_EVEN, 3, 0, 0), + F(100000000, P_GPLL0_OUT_EVEN, 3, 0, 0), + { } +}; + +static struct clk_rcg2 gcc_sdcc4_apps_clk_src = { + .cmd_rcgr = 0x6a01c, + .mnd_width = 8, + .hid_width = 5, + .parent_map = gcc_parent_map_0, + .freq_tbl = ftbl_gcc_sdcc4_apps_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_sdcc4_apps_clk_src", + .parent_data = gcc_parent_data_0, + .num_parents = ARRAY_SIZE(gcc_parent_data_0), + .ops = &clk_rcg2_ops, + }, +}; + +static const struct freq_tbl ftbl_gcc_usb20_master_clk_src[] = { + F(50000000, P_GPLL4_OUT_EVEN, 5, 0, 0), + F(60000000, P_GPLL0_OUT_EVEN, 5, 0, 0), + F(120000000, P_GPLL0_OUT_MAIN, 5, 0, 0), + { } +}; + +static struct clk_rcg2 gcc_usb20_master_clk_src = { + .cmd_rcgr = 0x27048, + .mnd_width = 8, + .hid_width = 5, + .parent_map = gcc_parent_map_0, + .freq_tbl = ftbl_gcc_usb20_master_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_usb20_master_clk_src", + .parent_data = gcc_parent_data_0, + .num_parents = ARRAY_SIZE(gcc_parent_data_0), + .ops = &clk_rcg2_ops, + }, +}; + +static struct clk_rcg2 gcc_usb20_mock_utmi_clk_src = { + .cmd_rcgr = 0x2702c, + .mnd_width = 0, + .hid_width = 5, + .parent_map = gcc_parent_map_1, + .freq_tbl = ftbl_gcc_emac0_phy_aux_clk_src, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_usb20_mock_utmi_clk_src", + .parent_data = gcc_parent_data_1, + .num_parents = ARRAY_SIZE(gcc_parent_data_1), + .ops = &clk_rcg2_ops, + }, +}; + +static struct clk_regmap_div gcc_usb20_mock_utmi_postdiv_clk_src = { + .reg = 0x27044, + .shift = 0, + .width = 4, + .clkr.hw.init = &(const struct clk_init_data) { + .name = "gcc_usb20_mock_utmi_postdiv_clk_src", + .parent_hws = (const struct clk_hw*[]) { + &gcc_usb20_mock_utmi_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_regmap_div_ro_ops, + }, +}; + +static struct clk_branch gcc_boot_rom_ahb_clk = { + .halt_reg = 0x37004, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x37004, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(26), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_boot_rom_ahb_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_emac0_axi_clk = { + .halt_reg = 0x71018, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x71018, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x71018, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_emac0_axi_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_emac0_phy_aux_clk = { + .halt_reg = 0x71028, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x71028, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_emac0_phy_aux_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_emac0_phy_aux_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_emac0_ptp_clk = { + .halt_reg = 0x71044, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x71044, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_emac0_ptp_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_emac0_ptp_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_emac0_rgmii_clk = { + .halt_reg = 0x71048, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x71048, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_emac0_rgmii_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_emac0_rgmii_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_emac0_slv_ahb_clk = { + .halt_reg = 0x71024, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x71024, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x71024, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_emac0_slv_ahb_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_emac_0_clkref_en = { + .halt_reg = 0x94004, + .halt_check = BRANCH_HALT_DELAY, + .clkr = { + .enable_reg = 0x94004, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_emac_0_clkref_en", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_gp1_clk = { + .halt_reg = 0x47000, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x47000, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_gp1_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_gp1_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_gp2_clk = { + .halt_reg = 0x48000, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x48000, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_gp2_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_gp2_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_gp3_clk = { + .halt_reg = 0x49000, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x49000, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_gp3_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_gp3_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pcie_0_clkref_en = { + .halt_reg = 0x94000, + .halt_check = BRANCH_HALT_DELAY, + .clkr = { + .enable_reg = 0x94000, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_0_clkref_en", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pcie_aux_clk = { + .halt_reg = 0x53054, + .halt_check = BRANCH_HALT_DELAY, + .hwcg_reg = 0x53054, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d010, + .enable_mask = BIT(15), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_aux_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_pcie_aux_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pcie_cfg_ahb_clk = { + .halt_reg = 0x53034, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x53034, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d010, + .enable_mask = BIT(13), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_cfg_ahb_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pcie_mstr_axi_clk = { + .halt_reg = 0x53028, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x53028, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d010, + .enable_mask = BIT(12), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_mstr_axi_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pcie_pipe_clk = { + .halt_reg = 0x53064, + .halt_check = BRANCH_HALT_DELAY, + .hwcg_reg = 0x53064, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d010, + .enable_mask = BIT(17), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_pipe_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_pcie_pipe_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pcie_rchng_phy_clk = { + .halt_reg = 0x53050, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x53050, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d010, + .enable_mask = BIT(14), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_rchng_phy_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_pcie_rchng_phy_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pcie_sleep_clk = { + .halt_reg = 0x53060, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x53060, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d010, + .enable_mask = BIT(16), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_sleep_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_pcie_aux_phy_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pcie_slv_axi_clk = { + .halt_reg = 0x5301c, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x7d010, + .enable_mask = BIT(11), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_slv_axi_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pcie_slv_q2a_axi_clk = { + .halt_reg = 0x53018, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x53018, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d010, + .enable_mask = BIT(10), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pcie_slv_q2a_axi_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pdm2_clk = { + .halt_reg = 0x3400c, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x3400c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pdm2_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_pdm2_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pdm_ahb_clk = { + .halt_reg = 0x34004, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x34004, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pdm_ahb_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_pdm_xo4_clk = { + .halt_reg = 0x34008, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x34008, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_pdm_xo4_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_qupv3_wrap0_core_2x_clk = { + .halt_reg = 0x2d018, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(15), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_qupv3_wrap0_core_2x_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_qupv3_wrap0_core_clk = { + .halt_reg = 0x2d008, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(14), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_qupv3_wrap0_core_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_qupv3_wrap0_s0_clk = { + .halt_reg = 0x6c130, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(16), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_qupv3_wrap0_s0_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_qupv3_wrap0_s0_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_qupv3_wrap0_s1_clk = { + .halt_reg = 0x6c268, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(17), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_qupv3_wrap0_s1_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_qupv3_wrap0_s1_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_qupv3_wrap0_s2_clk = { + .halt_reg = 0x6c3a0, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(18), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_qupv3_wrap0_s2_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_qupv3_wrap0_s2_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_qupv3_wrap0_s3_clk = { + .halt_reg = 0x6c4d8, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(19), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_qupv3_wrap0_s3_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_qupv3_wrap0_s3_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_qupv3_wrap0_s4_clk = { + .halt_reg = 0x6c610, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(20), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_qupv3_wrap0_s4_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_qupv3_wrap0_s4_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_qupv3_wrap_0_m_ahb_clk = { + .halt_reg = 0x2d000, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x2d000, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(12), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_qupv3_wrap_0_m_ahb_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_qupv3_wrap_0_s_ahb_clk = { + .halt_reg = 0x2d004, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x2d004, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x7d008, + .enable_mask = BIT(13), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_qupv3_wrap_0_s_ahb_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_sdcc4_ahb_clk = { + .halt_reg = 0x6a010, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x6a010, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_sdcc4_ahb_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_sdcc4_apps_clk = { + .halt_reg = 0x6a004, + .halt_check = BRANCH_HALT, + .clkr = { + .enable_reg = 0x6a004, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_sdcc4_apps_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_sdcc4_apps_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_snoc_cnoc_usb3_clk = { + .halt_reg = 0x27060, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x27060, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x27060, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_snoc_cnoc_usb3_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_usb20_master_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_sys_noc_usb_sf_axi_clk = { + .halt_reg = 0x27064, + .halt_check = BRANCH_HALT_VOTED, + .hwcg_reg = 0x27064, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x27064, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_sys_noc_usb_sf_axi_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_usb20_master_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_usb20_master_clk = { + .halt_reg = 0x27018, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x27018, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_usb20_master_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_usb20_master_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_usb20_mock_utmi_clk = { + .halt_reg = 0x27028, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x27028, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_usb20_mock_utmi_clk", + .parent_hws = (const struct clk_hw*[]) { + &gcc_usb20_mock_utmi_postdiv_clk_src.clkr.hw, + }, + .num_parents = 1, + .flags = CLK_SET_RATE_PARENT, + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_usb20_sleep_clk = { + .halt_reg = 0x27024, + .halt_check = BRANCH_HALT_VOTED, + .clkr = { + .enable_reg = 0x27024, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_usb20_sleep_clk", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_usb2_clkref_en = { + .halt_reg = 0x94008, + .halt_check = BRANCH_HALT_DELAY, + .clkr = { + .enable_reg = 0x94008, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_usb2_clkref_en", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_usb3_prim_clkref_en = { + .halt_reg = 0x9400c, + .halt_check = BRANCH_HALT_DELAY, + .clkr = { + .enable_reg = 0x9400c, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_usb3_prim_clkref_en", + .ops = &clk_branch2_ops, + }, + }, +}; + +static struct clk_branch gcc_usb_phy_cfg_ahb2phy_clk = { + .halt_reg = 0x29004, + .halt_check = BRANCH_HALT, + .hwcg_reg = 0x29004, + .hwcg_bit = 1, + .clkr = { + .enable_reg = 0x29004, + .enable_mask = BIT(0), + .hw.init = &(const struct clk_init_data) { + .name = "gcc_usb_phy_cfg_ahb2phy_clk", + .ops = &clk_branch2_aon_ops, + }, + }, +}; + +static struct gdsc gcc_emac0_gdsc = { + .gdscr = 0x71004, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "gcc_emac0_gdsc", + }, + .pwrsts = PWRSTS_OFF_ON, + .flags = RETAIN_FF_ENABLE, +}; + +static struct gdsc gcc_pcie_gdsc = { + .gdscr = 0x53004, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "gcc_pcie_gdsc", + }, + .pwrsts = PWRSTS_OFF_ON, + .flags = RETAIN_FF_ENABLE, +}; + +static struct gdsc gcc_usb20_gdsc = { + .gdscr = 0x27004, + .en_rest_wait_val = 0x2, + .en_few_wait_val = 0x2, + .clk_dis_wait_val = 0xf, + .pd = { + .name = "gcc_usb20_gdsc", + }, + .pwrsts = PWRSTS_OFF_ON, + .flags = RETAIN_FF_ENABLE, +}; + +static struct clk_regmap *gcc_kuno_clocks[] = { + [GCC_BOOT_ROM_AHB_CLK] = &gcc_boot_rom_ahb_clk.clkr, + [GCC_EMAC0_AXI_CLK] = &gcc_emac0_axi_clk.clkr, + [GCC_EMAC0_PHY_AUX_CLK] = &gcc_emac0_phy_aux_clk.clkr, + [GCC_EMAC0_PHY_AUX_CLK_SRC] = &gcc_emac0_phy_aux_clk_src.clkr, + [GCC_EMAC0_PTP_CLK] = &gcc_emac0_ptp_clk.clkr, + [GCC_EMAC0_PTP_CLK_SRC] = &gcc_emac0_ptp_clk_src.clkr, + [GCC_EMAC0_RGMII_CLK] = &gcc_emac0_rgmii_clk.clkr, + [GCC_EMAC0_RGMII_CLK_SRC] = &gcc_emac0_rgmii_clk_src.clkr, + [GCC_EMAC0_SLV_AHB_CLK] = &gcc_emac0_slv_ahb_clk.clkr, + [GCC_EMAC_0_CLKREF_EN] = &gcc_emac_0_clkref_en.clkr, + [GCC_GP1_CLK] = &gcc_gp1_clk.clkr, + [GCC_GP1_CLK_SRC] = &gcc_gp1_clk_src.clkr, + [GCC_GP2_CLK] = &gcc_gp2_clk.clkr, + [GCC_GP2_CLK_SRC] = &gcc_gp2_clk_src.clkr, + [GCC_GP3_CLK] = &gcc_gp3_clk.clkr, + [GCC_GP3_CLK_SRC] = &gcc_gp3_clk_src.clkr, + [GCC_PCIE_0_CLKREF_EN] = &gcc_pcie_0_clkref_en.clkr, + [GCC_PCIE_AUX_CLK] = &gcc_pcie_aux_clk.clkr, + [GCC_PCIE_AUX_CLK_SRC] = &gcc_pcie_aux_clk_src.clkr, + [GCC_PCIE_AUX_PHY_CLK_SRC] = &gcc_pcie_aux_phy_clk_src.clkr, + [GCC_PCIE_CFG_AHB_CLK] = &gcc_pcie_cfg_ahb_clk.clkr, + [GCC_PCIE_MSTR_AXI_CLK] = &gcc_pcie_mstr_axi_clk.clkr, + [GCC_PCIE_PIPE_CLK] = &gcc_pcie_pipe_clk.clkr, + [GCC_PCIE_PIPE_CLK_SRC] = &gcc_pcie_pipe_clk_src.clkr, + [GCC_PCIE_RCHNG_PHY_CLK] = &gcc_pcie_rchng_phy_clk.clkr, + [GCC_PCIE_RCHNG_PHY_CLK_SRC] = &gcc_pcie_rchng_phy_clk_src.clkr, + [GCC_PCIE_SLEEP_CLK] = &gcc_pcie_sleep_clk.clkr, + [GCC_PCIE_SLV_AXI_CLK] = &gcc_pcie_slv_axi_clk.clkr, + [GCC_PCIE_SLV_Q2A_AXI_CLK] = &gcc_pcie_slv_q2a_axi_clk.clkr, + [GCC_PDM2_CLK] = &gcc_pdm2_clk.clkr, + [GCC_PDM2_CLK_SRC] = &gcc_pdm2_clk_src.clkr, + [GCC_PDM_AHB_CLK] = &gcc_pdm_ahb_clk.clkr, + [GCC_PDM_XO4_CLK] = &gcc_pdm_xo4_clk.clkr, + [GCC_QUPV3_WRAP0_CORE_2X_CLK] = &gcc_qupv3_wrap0_core_2x_clk.clkr, + [GCC_QUPV3_WRAP0_CORE_CLK] = &gcc_qupv3_wrap0_core_clk.clkr, + [GCC_QUPV3_WRAP0_S0_CLK] = &gcc_qupv3_wrap0_s0_clk.clkr, + [GCC_QUPV3_WRAP0_S0_CLK_SRC] = &gcc_qupv3_wrap0_s0_clk_src.clkr, + [GCC_QUPV3_WRAP0_S1_CLK] = &gcc_qupv3_wrap0_s1_clk.clkr, + [GCC_QUPV3_WRAP0_S1_CLK_SRC] = &gcc_qupv3_wrap0_s1_clk_src.clkr, + [GCC_QUPV3_WRAP0_S2_CLK] = &gcc_qupv3_wrap0_s2_clk.clkr, + [GCC_QUPV3_WRAP0_S2_CLK_SRC] = &gcc_qupv3_wrap0_s2_clk_src.clkr, + [GCC_QUPV3_WRAP0_S3_CLK] = &gcc_qupv3_wrap0_s3_clk.clkr, + [GCC_QUPV3_WRAP0_S3_CLK_SRC] = &gcc_qupv3_wrap0_s3_clk_src.clkr, + [GCC_QUPV3_WRAP0_S4_CLK] = &gcc_qupv3_wrap0_s4_clk.clkr, + [GCC_QUPV3_WRAP0_S4_CLK_SRC] = &gcc_qupv3_wrap0_s4_clk_src.clkr, + [GCC_QUPV3_WRAP_0_M_AHB_CLK] = &gcc_qupv3_wrap_0_m_ahb_clk.clkr, + [GCC_QUPV3_WRAP_0_S_AHB_CLK] = &gcc_qupv3_wrap_0_s_ahb_clk.clkr, + [GCC_SDCC4_AHB_CLK] = &gcc_sdcc4_ahb_clk.clkr, + [GCC_SDCC4_APPS_CLK] = &gcc_sdcc4_apps_clk.clkr, + [GCC_SDCC4_APPS_CLK_SRC] = &gcc_sdcc4_apps_clk_src.clkr, + [GCC_SNOC_CNOC_USB3_CLK] = &gcc_snoc_cnoc_usb3_clk.clkr, + [GCC_SYS_NOC_USB_SF_AXI_CLK] = &gcc_sys_noc_usb_sf_axi_clk.clkr, + [GCC_USB20_MASTER_CLK] = &gcc_usb20_master_clk.clkr, + [GCC_USB20_MASTER_CLK_SRC] = &gcc_usb20_master_clk_src.clkr, + [GCC_USB20_MOCK_UTMI_CLK] = &gcc_usb20_mock_utmi_clk.clkr, + [GCC_USB20_MOCK_UTMI_CLK_SRC] = &gcc_usb20_mock_utmi_clk_src.clkr, + [GCC_USB20_MOCK_UTMI_POSTDIV_CLK_SRC] = &gcc_usb20_mock_utmi_postdiv_clk_src.clkr, + [GCC_USB20_SLEEP_CLK] = &gcc_usb20_sleep_clk.clkr, + [GCC_USB2_CLKREF_EN] = &gcc_usb2_clkref_en.clkr, + [GCC_USB3_PRIM_CLKREF_EN] = &gcc_usb3_prim_clkref_en.clkr, + [GCC_USB_PHY_CFG_AHB2PHY_CLK] = &gcc_usb_phy_cfg_ahb2phy_clk.clkr, + [GPLL0] = &gpll0.clkr, + [GPLL0_OUT_EVEN] = &gpll0_out_even.clkr, + [GPLL2] = &gpll2.clkr, + [GPLL3] = &gpll3.clkr, + [GPLL4] = &gpll4.clkr, + [GPLL4_OUT_EVEN] = &gpll4_out_even.clkr, +}; + +static const struct qcom_reset_map gcc_kuno_resets[] = { + [GCC_EMAC0_BCR] = { 0x71000 }, + [GCC_PCIE_BCR] = { 0x53000 }, + [GCC_PCIE_LINK_DOWN_BCR] = { 0x87000 }, + [GCC_PCIE_NOCSR_COM_PHY_BCR] = { 0x88008 }, + [GCC_PCIE_PHY_BCR] = { 0x54000 }, + [GCC_PCIE_PHY_CFG_AHB_BCR] = { 0x88000 }, + [GCC_PCIE_PHY_COM_BCR] = { 0x88004 }, + [GCC_PCIE_PHY_NOCSR_COM_PHY_BCR] = { 0x8800c }, + [GCC_PDM_BCR] = { 0x34000 }, + [GCC_QUPV3_WRAPPER_0_BCR] = { 0x6c000 }, + [GCC_QUSB2PHY_BCR] = { 0x2a000 }, + [GCC_SDCC4_BCR] = { 0x6a000 }, + [GCC_TCSR_PCIE_BCR] = { 0x84000 }, + [GCC_USB20_BCR] = { 0x27000 }, + [GCC_USB_PHY_CFG_AHB2PHY_BCR] = { 0x29000 }, +}; + +static struct gdsc *gcc_kuno_gdscs[] = { + [GCC_EMAC0_GDSC] = &gcc_emac0_gdsc, + [GCC_PCIE_GDSC] = &gcc_pcie_gdsc, + [GCC_USB20_GDSC] = &gcc_usb20_gdsc, +}; + +static const struct clk_rcg_dfs_data gcc_dfs_clocks[] = { + DEFINE_RCG_DFS(gcc_qupv3_wrap0_s0_clk_src), + DEFINE_RCG_DFS(gcc_qupv3_wrap0_s1_clk_src), + DEFINE_RCG_DFS(gcc_qupv3_wrap0_s2_clk_src), + DEFINE_RCG_DFS(gcc_qupv3_wrap0_s3_clk_src), + DEFINE_RCG_DFS(gcc_qupv3_wrap0_s4_clk_src), +}; + +static const struct regmap_config gcc_kuno_regmap_config = { + .reg_bits = 32, + .reg_stride = 4, + .val_bits = 32, + .max_register = 0x1f41f0, + .fast_io = true, +}; + +static const u32 gcc_kuno_critical_cbcrs[] = { + 0x3e004, /* GCC_AHB_PCIE_LINK_CLK */ + 0x3e008, /* GCC_XO_PCIE_LINK_CLK */ +}; + +static const struct qcom_cc_driver_data gcc_kuno_driver_data = { + .dfs_rcgs = gcc_dfs_clocks, + .num_dfs_rcgs = ARRAY_SIZE(gcc_dfs_clocks), + .clk_cbcrs = gcc_kuno_critical_cbcrs, + .num_clk_cbcrs = ARRAY_SIZE(gcc_kuno_critical_cbcrs), +}; + +static const struct qcom_cc_desc gcc_kuno_desc = { + .config = &gcc_kuno_regmap_config, + .clks = gcc_kuno_clocks, + .num_clks = ARRAY_SIZE(gcc_kuno_clocks), + .resets = gcc_kuno_resets, + .num_resets = ARRAY_SIZE(gcc_kuno_resets), + .gdscs = gcc_kuno_gdscs, + .num_gdscs = ARRAY_SIZE(gcc_kuno_gdscs), + .use_rpm = true, + .driver_data = &gcc_kuno_driver_data, +}; + +static const struct of_device_id gcc_kuno_match_table[] = { + { .compatible = "qcom,kuno-gcc" }, + { } +}; +MODULE_DEVICE_TABLE(of, gcc_kuno_match_table); + +static int gcc_kuno_probe(struct platform_device *pdev) +{ + return qcom_cc_probe(pdev, &gcc_kuno_desc); +} + +static struct platform_driver gcc_kuno_driver = { + .probe = gcc_kuno_probe, + .driver = { + .name = "gcc-kuno", + .of_match_table = gcc_kuno_match_table, + }, +}; + +static int __init gcc_kuno_init(void) +{ + return platform_driver_register(&gcc_kuno_driver); +} +subsys_initcall(gcc_kuno_init); + +static void __exit gcc_kuno_exit(void) +{ + platform_driver_unregister(&gcc_kuno_driver); +} +module_exit(gcc_kuno_exit); + +MODULE_DESCRIPTION("QTI GCC Kuno Driver"); +MODULE_LICENSE("GPL"); From c00a43cde446edd7c4efcea2278fa692ecbfa7a8 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Fri, 21 Aug 2026 04:53:27 +0200 Subject: [PATCH 371/857] keys: fix lost wakeup when reaping a dead key type clear_bit() is atomic with respect to the word it modifies, but it is an unordered operation: it implies no memory barrier on either side (Documentation/atomic_bitops.txt). key_garbage_collector() clears KEY_GC_REAPING_KEYTYPE with clear_bit() and calls wake_up_bit() after reaping a dead key type. wake_up_bit() uses a lockless waitqueue check and requires a full barrier after the clear. The existing smp_mb() is before clear_bit(), so nothing orders the clear against that check. The GC can see an empty waitqueue while unregister_key_type() still sees the bit set. The final wakeup is then lost, leaving module unload stuck in wait_on_bit(). Use clear_and_wake_up_bit(). Its clear_bit_unlock() has RELEASE semantics, so the completed GC work stays ordered before the clear, and its smp_mb__after_atomic() orders the clear before the waitqueue check. Fixes: 0c061b5707ab ("KEYS: Correctly destroy key payloads when their keytype is removed") Assisted-by: Claude:claude-fable-5 Signed-off-by: Karl Mehltretter Link: https://lore.kernel.org/r/20260821025327.61488-1-kmehltretter@gmail.com Reviewed-by: Jarkko Sakkinen Signed-off-by: Jarkko Sakkinen --- security/keys/gc.c | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/security/keys/gc.c b/security/keys/gc.c index 748e83818a7604..eda445f815d47b 100644 --- a/security/keys/gc.c +++ b/security/keys/gc.c @@ -318,9 +318,7 @@ static void key_garbage_collector(struct work_struct *work) if (unlikely(gc_state & KEY_GC_REAPING_DEAD_3)) { kdebug("dead wake"); - smp_mb(); - clear_bit(KEY_GC_REAPING_KEYTYPE, &key_gc_flags); - wake_up_bit(&key_gc_flags, KEY_GC_REAPING_KEYTYPE); + clear_and_wake_up_bit(KEY_GC_REAPING_KEYTYPE, &key_gc_flags); } if (gc_state & KEY_GC_REAP_AGAIN) From bf0d7882cd43d6b155d93e482893f685d6f08b89 Mon Sep 17 00:00:00 2001 From: Maoyi Xie Date: Fri, 21 Aug 2026 17:59:35 +0800 Subject: [PATCH 372/857] keys: translate request_key_auth pid for the reading procfs instance request_key_auth_describe() prints rka->pid into /proc/keys as a raw pid_t in the initial pid namespace. A reader can open /proc/keys through a mount in another pid namespace. That reader sees a number with no meaning there. The number can even name an unrelated task. The line needs VIEW on the key. So the reader either shares the key owner's uid or possesses the key. The fix keeps a struct pid. Commit 4f82f45730c6 ("net ip6 flowlabel: Make owner a union of struct pid * and kuid_t") gave /proc/net/ip6_flowlabel the same storage. The print goes through pid_nr_ns(). It renders against the pid namespace of the procfs instance the line is read through. Commit ad08978ab41c ("ipv6/flowlabel: simplify pid namespace lookup") moved that print to the same anchor. Output through an initial namespace /proc does not change. The line shows 0 for a requestor with no number in that namespace. Translating at read time was the alternative. find_pid_ns() can resolve a recycled number. The line would then name a live task with no connection to the key. A stored struct pid gives 0 instead when the requestor has no number there. Link: https://lore.kernel.org/keyrings/20260809110202.2180410-1-maoyixie.tju@gmail.com/ Fixes: 78b7280cce23 ("KEYS: Improve /proc/keys") Cc: stable@vger.kernel.org # v5.10+ Assisted-by: Claude:claude-opus-5 codeql Signed-off-by: Maoyi Xie Link: https://lore.kernel.org/r/20260821095935.1864998-1-maoyixie.tju@gmail.com Reviewed-by: Jarkko Sakkinen Signed-off-by: Jarkko Sakkinen --- include/keys/request_key_auth-type.h | 2 +- security/keys/request_key_auth.c | 12 +++++++++--- 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/include/keys/request_key_auth-type.h b/include/keys/request_key_auth-type.h index 01e42ee5f4099e..464636278c4f8b 100644 --- a/include/keys/request_key_auth-type.h +++ b/include/keys/request_key_auth-type.h @@ -22,7 +22,7 @@ struct request_key_auth { const struct cred *cred; void *callout_info; size_t callout_len; - pid_t pid; + struct pid *pid; char op[8]; } __randomize_layout; diff --git a/security/keys/request_key_auth.c b/security/keys/request_key_auth.c index 282e09d8fa46c2..ed6f55b9cdd93e 100644 --- a/security/keys/request_key_auth.c +++ b/security/keys/request_key_auth.c @@ -9,6 +9,8 @@ #include #include +#include +#include #include #include #include @@ -73,7 +75,10 @@ static void request_key_auth_describe(const struct key *key, seq_puts(m, "key:"); seq_puts(m, key->description); if (key_is_positive(key)) - seq_printf(m, " pid:%d ci:%zu", rka->pid, rka->callout_len); + seq_printf(m, " pid:%d ci:%zu", + pid_nr_ns(rka->pid, + proc_pid_ns(file_inode(m->file)->i_sb)), + rka->callout_len); } /* @@ -113,6 +118,7 @@ static void free_request_key_auth(struct request_key_auth *rka) if (rka->cred) put_cred(rka->cred); kfree(rka->callout_info); + put_pid(rka->pid); kfree(rka); } @@ -226,14 +232,14 @@ struct key *request_key_auth_new(struct key *target, const char *op, irka = cred->request_key_auth->payload.data[0]; rka->cred = get_cred(irka->cred); - rka->pid = irka->pid; + rka->pid = get_pid(irka->pid); up_read(&cred->request_key_auth->sem); } else { /* it isn't - use this process as the context */ rka->cred = get_cred(cred); - rka->pid = current->pid; + rka->pid = get_pid(task_pid(current)); } rka->target_key = key_get(target); From 38668489f135bbe5bfd07c23a528b905a8c488e5 Mon Sep 17 00:00:00 2001 From: Hongling Zeng Date: Tue, 25 Aug 2026 09:55:01 +0800 Subject: [PATCH 373/857] gfs2: move brelse() after buffer head accesses The brelse() call happens before all buffer head accesses are complete. While the reference counting prevents a real use-after-free in practice, this pattern is error-prone for future maintenance. Move the brelse() call to the end of the function to make the intent clearer and eliminate potential static analysis warnings. Suggested-by: Andreas Gruenbacher Suggested-by: Andrew Price Signed-off-by: Hongling Zeng Signed-off-by: Andreas Gruenbacher --- fs/gfs2/log.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/gfs2/log.c b/fs/gfs2/log.c index 78bba8cc10b8fd..a92c84146de918 100644 --- a/fs/gfs2/log.c +++ b/fs/gfs2/log.c @@ -1038,7 +1038,6 @@ void gfs2_remove_from_journal(struct buffer_head *bh, int meta) set_bit(TR_TOUCHED, &tr->tr_flags); } was_pinned = 1; - brelse(bh); } if (bd) { if (bd->bd_tr) { @@ -1056,6 +1055,8 @@ void gfs2_remove_from_journal(struct buffer_head *bh, int meta) } clear_buffer_dirty(bh); clear_buffer_uptodate(bh); + if (was_pinned) + brelse(bh); } /** From 0025112796287599dc79c822ed37124582a0d68a Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Wed, 26 Aug 2026 12:41:28 +0200 Subject: [PATCH 374/857] gfs2: No quotas for meta inodes Inodes on the meta filesystem (mount -o meta) can never have quotas, so skip them entirely in the quota code. This avoids allocating ip->i_qadata unnecessarily. Without this fix, unlinking meta inodes like "statfs" on the meta filesystem will result in a NULL pointer dereference at unmount time: gfs2_evict_inode() -> gfs2_dinode_dealloc() -> gfs2_quota_hold() -> qdsb_get() -> slot_get() -> find_first_zero_bit(sd_quota_bitmap) Unlinking inodes on the meta filesystem isn't a useful operation, but it still shouldn't cause the kernel to misbehave. Reported-by: syzbot+cb79de2cc8b76fbf474f@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=cb79de2cc8b76fbf474f Signed-off-by: Andreas Gruenbacher --- fs/gfs2/quota.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/fs/gfs2/quota.c b/fs/gfs2/quota.c index b62431724eea88..dbfc21693dd576 100644 --- a/fs/gfs2/quota.c +++ b/fs/gfs2/quota.c @@ -562,6 +562,8 @@ int gfs2_qa_get(struct gfs2_inode *ip) if (sdp->sd_args.ar_quota == GFS2_QUOTA_OFF) return 0; + if (ip->i_diskflags & GFS2_DIF_SYSTEM) + return 0; spin_lock(&inode->i_lock); if (ip->i_qadata == NULL) { @@ -605,7 +607,7 @@ int gfs2_quota_hold(struct gfs2_inode *ip, kuid_t uid, kgid_t gid) return 0; error = gfs2_qa_get(ip); - if (error) + if (error || !ip->i_qadata) return error; qd = ip->i_qadata->qa_qd; From 7d70a94339c247fac4c626cbd27c35765e8cf219 Mon Sep 17 00:00:00 2001 From: Jiaming Zhang Date: Mon, 31 Aug 2026 18:54:25 +0800 Subject: [PATCH 375/857] gfs2: reject invalid inode sizes A corrupted GFS2 image can store a dinode size that is larger than what VFS i_size can represent. gfs2_dinode_in() reads the on-disk di_size as a u64 and writes it directly into inode->i_size. If the value is larger than S64_MAX, the incore i_size becomes negative. That negative value can bypass the existing stuffed inode size check: inode->i_size > gfs2_max_stuffed_size(ip) Later, gfs2_quotad may try to sync the quota file and unstuff the quota inode. gfs2_unstuffer_folio() reads the negative i_size into an unsigned length and passes it to memcpy(), turning it into a huge copy size and triggering a out-of-bound issue. Reject dinodes whose size exceeds sb->s_maxbytes before storing the value in inode->i_size. Also make the stuffed inode check use the raw on-disk size while it is still unsigned. Fixes: 70376c7ff312 ("gfs2: Always check inode size of inline inodes") Closes: https://lore.kernel.org/lkml/CANypQFaF6bvORKKbRALvEL0k_epFaneFiOQqco4gjdmKVbdURg@mail.gmail.com/ Assisted-by: Codex:gpt-5.5-xhigh Cc: stable@vger.kernel.org Signed-off-by: Jiaming Zhang Signed-off-by: Andreas Gruenbacher --- fs/gfs2/glops.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/fs/gfs2/glops.c b/fs/gfs2/glops.c index 662d033fd2cafa..2597e1cf2315c7 100644 --- a/fs/gfs2/glops.c +++ b/fs/gfs2/glops.c @@ -393,6 +393,7 @@ static int gfs2_dinode_in(struct gfs2_inode *ip, const void *buf) umode_t mode = be32_to_cpu(str->di_mode); struct inode *inode = &ip->i_inode; bool is_new = inode_state_read_once(inode) & I_NEW; + u64 size; if (unlikely(ip->i_no_addr != be64_to_cpu(str->di_num.no_addr))) { gfs2_consist_inode(ip); @@ -418,7 +419,12 @@ static int gfs2_dinode_in(struct gfs2_inode *ip, const void *buf) i_uid_write(inode, be32_to_cpu(str->di_uid)); i_gid_write(inode, be32_to_cpu(str->di_gid)); set_nlink(inode, be32_to_cpu(str->di_nlink)); - i_size_write(inode, be64_to_cpu(str->di_size)); + size = be64_to_cpu(str->di_size); + if (unlikely(size > inode->i_sb->s_maxbytes)) { + gfs2_consist_inode(ip); + return -EIO; + } + i_size_write(inode, size); gfs2_set_inode_blocks(inode, be64_to_cpu(str->di_blocks)); atime.tv_sec = be64_to_cpu(str->di_atime); atime.tv_nsec = be32_to_cpu(str->di_atime_nsec); @@ -462,7 +468,7 @@ static int gfs2_dinode_in(struct gfs2_inode *ip, const void *buf) return -EIO; } - if (gfs2_is_stuffed(ip) && inode->i_size > gfs2_max_stuffed_size(ip)) { + if (gfs2_is_stuffed(ip) && size > gfs2_max_stuffed_size(ip)) { gfs2_consist_inode(ip); return -EIO; } From bd9b34572be0d5df542684231d69137eb11c5046 Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Sun, 30 Aug 2026 16:36:37 +0200 Subject: [PATCH 376/857] gfs2: Get rid of sd_async_glock_wait Get rid of the per-superblock wait queue for asynchronous locking requests. Instead, wait for the specific events we are interested in. Use multiple wait queue entries when waiting for multiple events at once. Without this patch, the per-superblock wait queue for asynchronous locking requests can become a bottleneck when many inodes are deleted remotely: in that case, we schedule delayed work with gfs2_queue_verify_delete(). When that work later runs, we end up in delete_work_func() -> iput() -> gfs2_evict_inode() -> gfs2_upgrade_iopen_glock(), which uses asynchronous locking. Each completing locking asynchronous request will wake up sdp->sd_async_glock_wait, and all the waiters will compete with each other and waste resources. Avoid that by eliminating the per-superblock wait queue. Signed-off-by: Andreas Gruenbacher --- fs/gfs2/glock.c | 56 ++++++++++++++++++++++++++++++-------------- fs/gfs2/incore.h | 1 - fs/gfs2/ops_fstype.c | 1 - fs/gfs2/super.c | 30 ++++++++++++++++++++---- 4 files changed, 63 insertions(+), 25 deletions(-) diff --git a/fs/gfs2/glock.c b/fs/gfs2/glock.c index d0612014408e43..d22a088c66cdab 100644 --- a/fs/gfs2/glock.c +++ b/fs/gfs2/glock.c @@ -329,11 +329,6 @@ static void gfs2_holder_wake(struct gfs2_holder *gh) clear_bit(HIF_WAIT, &gh->gh_iflags); smp_mb__after_atomic(); wake_up_bit(&gh->gh_iflags, HIF_WAIT); - if (gh->gh_flags & GL_ASYNC) { - struct gfs2_sbd *sdp = glock_sbd(gh->gh_gl); - - wake_up(&sdp->sd_async_glock_wait); - } } /** @@ -512,11 +507,9 @@ static void state_change(struct gfs2_glock *gl, unsigned int new_state) static void gfs2_set_demote(int nr, struct gfs2_glock *gl) { - struct gfs2_sbd *sdp = glock_sbd(gl); - set_bit(nr, &gl->gl_flags); - smp_mb(); - wake_up(&sdp->sd_async_glock_wait); + smp_mb__after_atomic(); + wake_up_bit(&gl->gl_flags, GLF_DEMOTE); } static void gfs2_demote_wake(struct gfs2_glock *gl) @@ -1272,11 +1265,14 @@ static int glocks_pending(unsigned int num_gh, struct gfs2_holder *ghs) int gfs2_glock_async_wait(unsigned int num_gh, struct gfs2_holder *ghs, unsigned int retries) { - struct gfs2_sbd *sdp = glock_sbd(ghs[0].gh_gl); unsigned long start_time = jiffies; - int i, ret = 0; - long timeout; + struct wait_queue_head *waitq[4]; + struct wait_queue_entry wait[4]; + long ret, timeout; + int i; + BUILD_BUG_ON(ARRAY_SIZE(waitq) != ARRAY_SIZE(wait)); + BUG_ON(num_gh > ARRAY_SIZE(wait)); might_sleep(); timeout = GL_GLOCK_MIN_HOLD; @@ -1294,14 +1290,38 @@ int gfs2_glock_async_wait(unsigned int num_gh, struct gfs2_holder *ghs, timeout += (incr / 3) + get_random_long() % (incr / 3); } - if (!wait_event_interruptible_timeout(sdp->sd_async_glock_wait, - !glocks_pending(num_gh, ghs), timeout)) { - ret = -ESTALE; /* request timed out. */ - goto out; + ret = timeout; + for (i = 0; i < num_gh; i++) { + waitq[i] = bit_waitqueue(&ghs[i].gh_iflags, HIF_WAIT); + init_wait(wait + i); + } + for (;;) { + for (i = 0; i < num_gh; i++) + prepare_to_wait(waitq[i], wait + i, TASK_INTERRUPTIBLE); + if (!glocks_pending(num_gh, ghs)) + break; + if (signal_pending(current)) { + ret = -EINTR; + break; + } + ret = schedule_timeout(ret); + if (!glocks_pending(num_gh, ghs)) + break; + if (!ret) { + ret = -ESTALE; /* request timed out. */ + break; + } + if (signal_pending(current)) { + ret = -EINTR; + break; + } } - if (signal_pending(current)) - goto interrupted; + for (i = 0; i < num_gh; i++) + finish_wait(waitq[i], wait + i); + if (ret < 0) + goto out; + ret = 0; for (i = 0; i < num_gh; i++) { struct gfs2_holder *gh = &ghs[i]; int ret2; diff --git a/fs/gfs2/incore.h b/fs/gfs2/incore.h index 0a48f8e8ed6c5b..49cc232942a293 100644 --- a/fs/gfs2/incore.h +++ b/fs/gfs2/incore.h @@ -716,7 +716,6 @@ struct gfs2_sbd { struct work_struct sd_freeze_work; struct work_struct sd_withdraw_work; wait_queue_head_t sd_kill_wait; - wait_queue_head_t sd_async_glock_wait; atomic_t sd_glock_disposal; struct completion sd_locking_init; struct completion sd_withdraw_helper; diff --git a/fs/gfs2/ops_fstype.c b/fs/gfs2/ops_fstype.c index 188b3e67f2d1eb..6ec47376aecf25 100644 --- a/fs/gfs2/ops_fstype.c +++ b/fs/gfs2/ops_fstype.c @@ -90,7 +90,6 @@ static struct gfs2_sbd *init_sbd(struct super_block *sb) gfs2_tune_init(&sdp->sd_tune); init_waitqueue_head(&sdp->sd_kill_wait); - init_waitqueue_head(&sdp->sd_async_glock_wait); atomic_set(&sdp->sd_glock_disposal, 0); init_completion(&sdp->sd_locking_init); init_completion(&sdp->sd_withdraw_helper); diff --git a/fs/gfs2/super.c b/fs/gfs2/super.c index af8c715768336a..04bb4cf787d4db 100644 --- a/fs/gfs2/super.c +++ b/fs/gfs2/super.c @@ -1180,8 +1180,10 @@ static enum evict_behavior gfs2_upgrade_iopen_glock(struct inode *inode) { struct gfs2_glock *gl = gfs2_inode_glock(inode); struct gfs2_inode *ip = GFS2_I(inode); - struct gfs2_sbd *sdp = GFS2_SB(inode); struct gfs2_holder *gh = &ip->i_iopen_gh; + struct wait_queue_head *holder_waitq, *glock_waitq; + struct wait_queue_entry holder_wait, glock_wait; + long ret = 5 * HZ; int error; gh->gh_flags |= GL_NOCACHE; @@ -1212,10 +1214,28 @@ static enum evict_behavior gfs2_upgrade_iopen_glock(struct inode *inode) if (error) return EVICT_SHOULD_SKIP_DELETE; - wait_event_interruptible_timeout(sdp->sd_async_glock_wait, - !test_bit(HIF_WAIT, &gh->gh_iflags) || - glock_needs_demote(gl), - 5 * HZ); + holder_waitq = bit_waitqueue(&gh->gh_iflags, HIF_WAIT); + glock_waitq = bit_waitqueue(&gl->gl_flags, GLF_DEMOTE); + init_wait(&holder_wait); + init_wait(&glock_wait); + for (;;) { + prepare_to_wait(holder_waitq, &holder_wait, TASK_INTERRUPTIBLE); + prepare_to_wait(glock_waitq, &glock_wait, TASK_INTERRUPTIBLE); + if (gfs2_glock_poll(gh) || glock_needs_demote(gl)) + break; + if (signal_pending(current)) + break; + ret = schedule_timeout(ret); + if (gfs2_glock_poll(gh) || glock_needs_demote(gl)) + break; + if (!ret) + break; + if (signal_pending(current)) + break; + } + finish_wait(holder_waitq, &holder_wait); + finish_wait(glock_waitq, &glock_wait); + if (!test_bit(HIF_HOLDER, &gh->gh_iflags)) { gfs2_glock_dq(gh); if (glock_needs_demote(gl)) From 4a346e5c76b6bc7cd4569c0723d4d59c01d3e02c Mon Sep 17 00:00:00 2001 From: Andreas Gruenbacher Date: Tue, 1 Sep 2026 10:00:30 +0200 Subject: [PATCH 377/857] gfs2: missing whitespace in error message Add a missing whitespace in this error message in gfs2_quota_init(). Signed-off-by: Andreas Gruenbacher --- fs/gfs2/quota.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/gfs2/quota.c b/fs/gfs2/quota.c index dbfc21693dd576..52457f271f8b53 100644 --- a/fs/gfs2/quota.c +++ b/fs/gfs2/quota.c @@ -1477,7 +1477,7 @@ int gfs2_quota_init(struct gfs2_sbd *sdp) spin_lock_bucket(hash); old_qd = gfs2_qd_search_bucket_noref(hash, sdp, qc_id); if (old_qd) { - fs_err(sdp, "Corruption found in quota_change%u" + fs_err(sdp, "Corruption found in quota_change%u " "file: duplicate identifier in " "slot %u\n", sdp->sd_jdesc->jd_jid, slot); From 0b93c56833caf45c33f8260a5ab8c85cf1f72329 Mon Sep 17 00:00:00 2001 From: "Vlastimil Babka (SUSE)" Date: Mon, 31 Aug 2026 18:02:38 +0200 Subject: [PATCH 378/857] mm/slab: disallow kfree_rcu_sheaf() on PREEMPT_RT again This partially reverts commit 2a8bb29ec9b2 ("mm/slab: allow kfree_rcu_sheaf() on PREEMPT_RT"). It was based on the assumption that local_trylock() is safe on PREEMPT_RT from any context. However kvfree_rcu() is also called by set_cpus_allowed_force() with task_struct::pi_lock acquired and there it's not safe, as syzbot has reported. For the immediate fix, skip kfree_rcu_sheaf() on PREEMPT_RT again from kvfree_call_rcu(). In theory, kfree_rcu_nolock() would have the same problem when called from under pi_lock on PREEMPT_RT but that can be addressed if such a caller is proposed. Add an explanation comment, courtesy of Sebastian. Reported-by: syzbot+acf142088e0182172e58@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=acf142088e0182172e58 Reported-by: ThangNN99 Fixes: 2a8bb29ec9b2 ("mm/slab: allow kfree_rcu_sheaf() on PREEMPT_RT") Reviewed-by: Sebastian Andrzej Siewior Link: https://patch.msgid.link/20260831-b4-kfree_rcu_hotfix-v1-1-4f0fb882638b@kernel.org Signed-off-by: Vlastimil Babka (SUSE) --- mm/slab_common.c | 18 ++++++++---------- mm/slub.c | 5 +++-- 2 files changed, 11 insertions(+), 12 deletions(-) diff --git a/mm/slab_common.c b/mm/slab_common.c index b19ba1b31484ce..7223a7596dabbd 100644 --- a/mm/slab_common.c +++ b/mm/slab_common.c @@ -1667,14 +1667,6 @@ static bool kfree_rcu_sheaf(void *obj) { struct kmem_cache *s; struct slab *slab; - unsigned int free_flags = SLAB_FREE_DEFAULT; - - /* - * It is not safe to spin on PREEMPT_RT because the kernel might be - * holding a raw spinlock and slab acquires sleeping locks. - */ - if (IS_ENABLED(CONFIG_PREEMPT_RT)) - free_flags = SLAB_FREE_NOLOCK; if (is_vmalloc_addr(obj)) return false; @@ -1685,7 +1677,7 @@ static bool kfree_rcu_sheaf(void *obj) s = slab->slab_cache; if (likely(!IS_ENABLED(CONFIG_NUMA) || slab_nid(slab) == numa_mem_id())) - return __kfree_rcu_sheaf(s, obj, free_flags); + return __kfree_rcu_sheaf(s, obj, SLAB_FREE_DEFAULT); return false; } @@ -2034,7 +2026,13 @@ void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr) if (!head) might_sleep(); - if (kfree_rcu_sheaf(ptr)) + /* + * kvfree_rcu() is called by set_cpus_allowed_force() with + * task_struct::pi_lock acquired. On PREEMPT_RT the local_trylock() + * usage below will acquire the waitlock which must be avoided. + * Therefore avoid it on PREEMPT_RT. + */ + if (!IS_ENABLED(CONFIG_PREEMPT_RT) && kfree_rcu_sheaf(ptr)) return; // Queue the object but don't yet schedule the batch. diff --git a/mm/slub.c b/mm/slub.c index f9b56cb439e709..7a7e906a0e44d4 100644 --- a/mm/slub.c +++ b/mm/slub.c @@ -6088,8 +6088,9 @@ static void rcu_free_sheaf(struct rcu_head *head) /* * kvfree_call_rcu() can be called while holding a raw_spinlock_t. Since * __kfree_rcu_sheaf() may acquire a spinlock_t (sleeping lock on PREEMPT_RT), - * this would violate lock nesting rules. Therefore, kvfree_call_rcu() avoids - * this problem by passing SLAB_FREE_NOLOCK on PREEMPT_RT. + * this would violate lock nesting rules. Therefore, kfree_call_rcu_nolock() + * avoids this problem by passing SLAB_FREE_NOLOCK. kvfree_call_rcu() is + * bypassing the sheaves layer completely on PREEMPT_RT. * * However, lockdep still complains that it is invalid to acquire spinlock_t * while holding raw_spinlock_t, even on !PREEMPT_RT where spinlock_t is a From 2e918ca509ae579c61e3afd52a50c92f55b9a398 Mon Sep 17 00:00:00 2001 From: Timothy Day Date: Tue, 11 Aug 2026 12:03:35 -0400 Subject: [PATCH 379/857] ext2: annotate s_rsv_window_root as requiring s_rsv_window_lock The per-filesystem reservation window rb-tree (s_rsv_window_root) is protected by s_rsv_window_lock. Mark the s_rsv_window_root field with __guarded_by() for Clang's context analysis. The helpers (ext2_rsv_window_add, rsv_window_remove, and find_next_reservable_window) that mutate or walk the tree are all called with the s_rsv_window_lock held. Annotate these helpers with __must_hold(). The accesses in ext2_fill_super() are before the superblock is live. Since no concurrent access should be possible, s_rsv_window_lock is not taken. Convert the spinlock initialization to use scoped_guard(spinlock_init, ...) and place the writes under the guard to prevent warnings. Signed-off-by: Timothy Day Acked-by: Marco Elver Link: https://patch.msgid.link/20260811160336.782342-8-timday@thelustrecollective.com Signed-off-by: Jan Kara --- fs/ext2/balloc.c | 3 +++ fs/ext2/ext2.h | 5 +++-- fs/ext2/super.c | 26 ++++++++++++++------------ 3 files changed, 20 insertions(+), 14 deletions(-) diff --git a/fs/ext2/balloc.c b/fs/ext2/balloc.c index 53c91cb38bdee9..80acc1e1938710 100644 --- a/fs/ext2/balloc.c +++ b/fs/ext2/balloc.c @@ -334,6 +334,7 @@ search_reserve_window(struct rb_root *root, ext2_fsblk_t goal) */ void ext2_rsv_window_add(struct super_block *sb, struct ext2_reserve_window_node *rsv) + __must_hold(&EXT2_SB(sb)->s_rsv_window_lock) { struct rb_root *root = &EXT2_SB(sb)->s_rsv_window_root; struct rb_node *node = &rsv->rsv_node; @@ -373,6 +374,7 @@ void ext2_rsv_window_add(struct super_block *sb, */ static void rsv_window_remove(struct super_block *sb, struct ext2_reserve_window_node *rsv) + __must_hold(&EXT2_SB(sb)->s_rsv_window_lock) { rsv->rsv_start = EXT2_RESERVE_WINDOW_NOT_ALLOCATED; rsv->rsv_end = EXT2_RESERVE_WINDOW_NOT_ALLOCATED; @@ -759,6 +761,7 @@ static int find_next_reservable_window( struct super_block * sb, ext2_fsblk_t start_block, ext2_fsblk_t last_block) + __must_hold(&EXT2_SB(sb)->s_rsv_window_lock) { struct rb_node *next; struct ext2_reserve_window_node *rsv, *prev; diff --git a/fs/ext2/ext2.h b/fs/ext2/ext2.h index 4b1622eec2af17..7aeb7cfb0cebf3 100644 --- a/fs/ext2/ext2.h +++ b/fs/ext2/ext2.h @@ -102,7 +102,7 @@ struct ext2_sb_info { struct blockgroup_lock *s_blockgroup_lock; /* root of the per fs reservation window tree */ spinlock_t s_rsv_window_lock; - struct rb_root s_rsv_window_root; + struct rb_root s_rsv_window_root __guarded_by(&s_rsv_window_lock); struct ext2_reserve_window_node s_rsv_window_head; /* * s_lock protects against concurrent modifications of s_mount_state, @@ -712,7 +712,8 @@ extern void ext2_discard_reservation (struct inode *); extern int ext2_should_retry_alloc(struct super_block *sb, int *retries); extern void ext2_init_block_alloc_info(struct inode *inode) __must_hold(&EXT2_I(inode)->truncate_mutex); -extern void ext2_rsv_window_add(struct super_block *sb, struct ext2_reserve_window_node *rsv); +extern void ext2_rsv_window_add(struct super_block *sb, struct ext2_reserve_window_node *rsv) + __must_hold(&EXT2_SB(sb)->s_rsv_window_lock); /* dir.c */ int ext2_add_link(struct dentry *, struct inode *); diff --git a/fs/ext2/super.c b/fs/ext2/super.c index 4d9e5b43f59a69..dc8070d20b7f3d 100644 --- a/fs/ext2/super.c +++ b/fs/ext2/super.c @@ -1134,18 +1134,20 @@ static int ext2_fill_super(struct super_block *sb, struct fs_context *fc) /* per filesystem reservation list head & lock */ spin_lock_init(&sbi->s_rsv_window_lock); - sbi->s_rsv_window_root = RB_ROOT; - /* - * Add a single, static dummy reservation to the start of the - * reservation window list --- it gives us a placeholder for - * append-at-start-of-list which makes the allocation logic - * _much_ simpler. - */ - sbi->s_rsv_window_head.rsv_start = EXT2_RESERVE_WINDOW_NOT_ALLOCATED; - sbi->s_rsv_window_head.rsv_end = EXT2_RESERVE_WINDOW_NOT_ALLOCATED; - sbi->s_rsv_window_head.rsv_alloc_hit = 0; - sbi->s_rsv_window_head.rsv_goal_size = 0; - ext2_rsv_window_add(sb, &sbi->s_rsv_window_head); + scoped_guard(spinlock, &sbi->s_rsv_window_lock) { + sbi->s_rsv_window_root = RB_ROOT; + /* + * Add a single, static dummy reservation to the start of the + * reservation window list --- it gives us a placeholder for + * append-at-start-of-list which makes the allocation logic + * _much_ simpler. + */ + sbi->s_rsv_window_head.rsv_start = EXT2_RESERVE_WINDOW_NOT_ALLOCATED; + sbi->s_rsv_window_head.rsv_end = EXT2_RESERVE_WINDOW_NOT_ALLOCATED; + sbi->s_rsv_window_head.rsv_alloc_hit = 0; + sbi->s_rsv_window_head.rsv_goal_size = 0; + ext2_rsv_window_add(sb, &sbi->s_rsv_window_head); + } err = percpu_counter_init(&sbi->s_freeblocks_counter, ext2_count_free_blocks(sb), GFP_KERNEL); From a3cf61f60051ca8874f012fa8569fa971fbfbee0 Mon Sep 17 00:00:00 2001 From: Timothy Day Date: Tue, 11 Aug 2026 12:03:36 -0400 Subject: [PATCH 380/857] ext2: enable context analysis support for ext2 filesystem Update ext2 Makefile to support context analysis [1]. [1] https://docs.kernel.org/dev-tools/context-analysis.html Signed-off-by: Timothy Day Acked-by: Marco Elver Link: https://patch.msgid.link/20260811160336.782342-9-timday@thelustrecollective.com Signed-off-by: Jan Kara --- fs/ext2/Makefile | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/ext2/Makefile b/fs/ext2/Makefile index 8860948ef9ca4e..33db2e9dc90827 100644 --- a/fs/ext2/Makefile +++ b/fs/ext2/Makefile @@ -3,6 +3,8 @@ # Makefile for the linux ext2-filesystem routines. # +CONTEXT_ANALYSIS := y + obj-$(CONFIG_EXT2_FS) += ext2.o ext2-y := balloc.o dir.o file.o ialloc.o inode.o \ From a9d53db40dba3db2fa8e58c4ace5e7952ab8874d Mon Sep 17 00:00:00 2001 From: Snehal Sanghvi Date: Wed, 12 Aug 2026 03:05:31 -0700 Subject: [PATCH 381/857] RDMA/mana_ib: Enable multi-port GSI QP support for mana_ib Add MANA_IB_FEATURE_MULTI_PORT_GSI_SUPPORT so mana_ib can create a GSI QP per IB port instead of just one. When the feature is negotiated, the port's vNIC MAC is passed on GSI QP creation and each GSI QP is indexed in the QP table by port so the GSI SQ drain can reach every port. Also move MANA_SENDQ_MASK to BIT(0), freeing the top byte of the queue-id key to index GSI QPs by port and scaling the feature to the full 8-bit port range. Signed-off-by: Snehal Sanghvi Signed-off-by: Konstantin Taranov --- drivers/infiniband/hw/mana/cq.c | 26 +++++++++++++++++--------- drivers/infiniband/hw/mana/main.c | 28 ++++++++++++++++++++++++++-- drivers/infiniband/hw/mana/mana_ib.h | 15 +++++++++++++-- drivers/infiniband/hw/mana/qp.c | 17 ++++++++++++++++- 4 files changed, 72 insertions(+), 14 deletions(-) diff --git a/drivers/infiniband/hw/mana/cq.c b/drivers/infiniband/hw/mana/cq.c index d4e5e3f912689a..6f9ac8b4aac8ed 100644 --- a/drivers/infiniband/hw/mana/cq.c +++ b/drivers/infiniband/hw/mana/cq.c @@ -293,18 +293,12 @@ static int mana_process_completions(struct mana_ib_cq *cq, int nwc, struct ib_wc return wc_index; } -void mana_drain_gsi_sqs(struct mana_ib_dev *mdev) +static void mana_drain_gsi_sq(struct mana_ib_qp *qp) { - struct mana_ib_qp *qp = mana_get_qp_ref(mdev, MANA_GSI_QPN, false); + struct mana_ib_cq *cq = container_of(qp->ibqp.send_cq, struct mana_ib_cq, ibcq); struct ud_sq_shadow_wqe *shadow_wqe; - struct mana_ib_cq *cq; unsigned long flags; - if (!qp) - return; - - cq = container_of(qp->ibqp.send_cq, struct mana_ib_cq, ibcq); - spin_lock_irqsave(&cq->cq_lock, flags); while ((shadow_wqe = shadow_queue_get_next_to_complete(&qp->shadow_sq)) != NULL) { @@ -315,8 +309,22 @@ void mana_drain_gsi_sqs(struct mana_ib_dev *mdev) if (cq->ibcq.comp_handler) cq->ibcq.comp_handler(&cq->ibcq, cq->ibcq.cq_context); +} - mana_put_qp_ref(qp); +void mana_drain_gsi_sqs(struct mana_ib_dev *mdev) +{ + struct mana_ib_qp *qp; + u32 port; + + /* One GSI QP per port, indexed in the QP table by (port << 24 | MANA_GSI_QPN) */ + for (port = 1; port <= mdev->ib_dev.phys_port_cnt; port++) { + qp = mana_get_qp_ref(mdev, (port << 24) | MANA_GSI_QPN, false); + if (!qp) + continue; + + mana_drain_gsi_sq(qp); + mana_put_qp_ref(qp); + } } int mana_ib_poll_cq(struct ib_cq *ibcq, int num_entries, struct ib_wc *wc) diff --git a/drivers/infiniband/hw/mana/main.c b/drivers/infiniband/hw/mana/main.c index 57c6d9faaa81b7..83a97f1c5caabc 100644 --- a/drivers/infiniband/hw/mana/main.c +++ b/drivers/infiniband/hw/mana/main.c @@ -593,7 +593,11 @@ int mana_ib_get_port_immutable(struct ib_device *ibdev, u32 port_num, immutable->gid_tbl_len = attr.gid_tbl_len; if (mana_ib_is_rnic(dev)) { - if (port_num == 1) { + bool port_supports_cm = (port_num == 1 || + (dev->adapter_caps.feature_flags & + MANA_IB_FEATURE_MULTI_PORT_GSI_SUPPORT)); + + if (port_supports_cm) { immutable->core_cap_flags = RDMA_CORE_PORT_IBA_ROCE_UDP_ENCAP; immutable->max_mad_size = IB_MGMT_MAD_SIZE; } else { @@ -671,10 +675,14 @@ int mana_ib_query_port(struct ib_device *ibdev, u32 port, ib_get_eth_speed(ibdev, port, &props->active_speed, &props->active_width); props->pkey_tbl_len = 1; if (mana_ib_is_rnic(dev)) { + bool port_supports_cm = (port == 1 || + (dev->adapter_caps.feature_flags & + MANA_IB_FEATURE_MULTI_PORT_GSI_SUPPORT)); + props->gid_tbl_len = 16; props->ip_gids = true; props->max_msg_sz = SZ_16M; - if (port == 1) + if (port_supports_cm) props->port_cap_flags = IB_PORT_CM_SUP; } @@ -1138,6 +1146,7 @@ int mana_ib_gd_create_ud_qp(struct mana_ib_dev *mdev, struct mana_ib_qp *qp, struct gdma_context *gc = mdev_to_gc(mdev); struct mana_rnic_create_udqp_resp resp = {}; struct mana_rnic_create_udqp_req req = {}; + struct net_device *ndev; int err, i; mana_gd_init_req_hdr(&req.hdr, MANA_IB_CREATE_UD_QP, sizeof(req), sizeof(resp)); @@ -1154,6 +1163,21 @@ int mana_ib_gd_create_ud_qp(struct mana_ib_dev *mdev, struct mana_ib_qp *qp, req.max_send_sge = attr->cap.max_send_sge; req.max_recv_sge = attr->cap.max_recv_sge; req.qp_type = type; + + /* For GSI QPs, pass the vNIC MAC so the SoC can associate the QP with + * the correct port, transition to INIT, and allocate a per-port QPN. + * MAC must be in reversed byte order. + */ + if (type == IB_QPT_GSI && + (mdev->adapter_caps.feature_flags & MANA_IB_FEATURE_MULTI_PORT_GSI_SUPPORT)) { + ndev = mana_ib_get_netdev(&mdev->ib_dev, attr->port_num); + if (ndev) { + copy_in_reverse(req.mac, ndev->dev_addr, ETH_ALEN); + req.flags = MANA_UD_QP_FLAG_CREATE_IN_INIT; + req.hdr.req.msg_version = GDMA_MESSAGE_V2; + } + } + err = mana_gd_send_request(gc, sizeof(req), &req, sizeof(resp), &resp); if (err) return err; diff --git a/drivers/infiniband/hw/mana/mana_ib.h b/drivers/infiniband/hw/mana/mana_ib.h index 828ca75fd68450..1be33ed8bd3b87 100644 --- a/drivers/infiniband/hw/mana/mana_ib.h +++ b/drivers/infiniband/hw/mana/mana_ib.h @@ -24,8 +24,12 @@ /* MANA doesn't have any limit for MR size */ #define MANA_IB_MAX_MR_SIZE U64_MAX -/* Send queue ID mask */ -#define MANA_SENDQ_MASK BIT(31) +/* + * Send queue ID mask. Queue IDs are 2-bit aligned (see MANA_QID_SUBTYPE_MASK), + * so bit 0 is always free to tag send queues in the lookup table. This keeps + * the whole top byte available to index per-port GSI QPs by (port << 24). + */ +#define MANA_SENDQ_MASK BIT(0) /* Queue ID encodes type in the lower 2 bits */ #define MANA_QID_SUBTYPE_MASK 0x3 @@ -264,6 +268,7 @@ enum mana_ib_adapter_features { MANA_IB_FEATURE_DEV_COUNTERS_SUPPORT = BIT(5), MANA_IB_FEATURE_MULTI_PORTS_SUPPORT = BIT(6), MANA_IB_FEATURE_MSN_IN_WQE_SUPPORT = BIT(7), + MANA_IB_FEATURE_MULTI_PORT_GSI_SUPPORT = BIT(15), }; struct mana_ib_query_adapter_caps_resp { @@ -450,8 +455,14 @@ struct mana_rnic_create_udqp_req { u32 max_recv_wr; u32 max_send_sge; u32 max_recv_sge; + u8 mac[ETH_ALEN]; /* V2: port MAC for multi-port GSI */ + u16 flags; /* V2: MANA_UD_QP_FLAG_* */ }; /* HW Data */ +enum mana_ud_qp_flags { + MANA_UD_QP_FLAG_CREATE_IN_INIT = BIT(0), +}; + struct mana_rnic_create_udqp_resp { struct gdma_resp_hdr hdr; mana_handle_t qp_handle; diff --git a/drivers/infiniband/hw/mana/qp.c b/drivers/infiniband/hw/mana/qp.c index 29bfee9ac997f3..fac43b3a5eb7bd 100644 --- a/drivers/infiniband/hw/mana/qp.c +++ b/drivers/infiniband/hw/mana/qp.c @@ -514,8 +514,20 @@ static int mana_table_store_qp(struct mana_ib_dev *mdev, struct mana_ib_qp *qp) if (err) goto err_remove_sq; + /* GSI QPs are additionally indexed by (port << 24 | MANA_GSI_QPN) so the + * per-port GSI SQ drain can find each of them by iterating ports. + */ + if (qp->ibqp.qp_type == IB_QPT_GSI) { + err = mana_table_store_qp_qid(mdev, qp, + (qp->port << 24) | MANA_GSI_QPN, false); + if (err) + goto err_remove_rq; + } + return 0; +err_remove_rq: + mana_table_remove_qp_qid(mdev, rq->id, false); err_remove_sq: mana_table_remove_qp_qid(mdev, sq->id, true); mana_table_drain_qp_ref(qp); @@ -534,6 +546,8 @@ static void mana_table_remove_qp(struct mana_ib_dev *mdev, struct mana_ib_qp *qp mana_table_remove_qp_qid(mdev, sq->id, true); mana_table_remove_qp_qid(mdev, rq->id, false); + if (qp->ibqp.qp_type == IB_QPT_GSI) + mana_table_remove_qp_qid(mdev, (qp->port << 24) | MANA_GSI_QPN, false); mana_table_drain_qp_ref(qp); } @@ -751,7 +765,8 @@ static int mana_ib_create_ud_qp(struct ib_qp *ibqp, struct ib_pd *ibpd, ibdev_err(&mdev->ib_dev, "Failed to create ud qp %d\n", err); goto destroy_shadow_queues; } - qp->ibqp.qp_num = qp->ud_qp.queues[MANA_UD_RECV_QUEUE].id; + qp->ibqp.qp_num = (qp->ibqp.qp_type == IB_QPT_GSI) ? + MANA_GSI_QPN : qp->ud_qp.queues[MANA_UD_RECV_QUEUE].id; qp->port = attr->port_num; for (i = 0; i < MANA_UD_QUEUE_TYPE_MAX; ++i) From ed89dd98a21da7c1fd02b1d755aaead93d323a68 Mon Sep 17 00:00:00 2001 From: Jiangshan Yi Date: Tue, 1 Sep 2026 17:30:49 +0300 Subject: [PATCH 382/857] tpm: Fix heap buffer overflow in tpm_transmit_cmd() struct tpm_buf now uses an embedded flexible array for data, which is 6 bytes shorter than TPM_BUFSIZE due to the struct header. But tpm_transmit_cmd() still passes PAGE_SIZE to tpm_transmit(), and after clamping to TPM_BUFSIZE in tpm_try_transmit(), chip->ops->recv() can write past the end of data[] by 6 bytes. Pass buf->capacity to tpm_transmit() instead. Fixes: 3d9e043dab0a ("tpm-buf: Memory-safe allocations") Cc: stable@vger.kernel.org Signed-off-by: Jiangshan Yi Link: https://lore.kernel.org/r/20260901062850.379870-1-yijiangshan@kylinos.cn Reviewed-by: Jarkko Sakkinen Signed-off-by: Jarkko Sakkinen --- drivers/char/tpm/tpm-interface.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/char/tpm/tpm-interface.c b/drivers/char/tpm/tpm-interface.c index f745a098908b37..1ccdbde98b69a8 100644 --- a/drivers/char/tpm/tpm-interface.c +++ b/drivers/char/tpm/tpm-interface.c @@ -268,7 +268,7 @@ ssize_t tpm_transmit_cmd(struct tpm_chip *chip, struct tpm_buf *buf, int err; ssize_t len; - len = tpm_transmit(chip, buf->data, PAGE_SIZE); + len = tpm_transmit(chip, buf->data, buf->capacity); if (len < 0) return len; From f9926f294e4e1be9b2fe1c236e0a26635a1b189d Mon Sep 17 00:00:00 2001 From: Sam Ho Date: Fri, 14 Aug 2026 13:01:11 +0000 Subject: [PATCH 383/857] btrfs: preserve the compression property when other inode flags change Setting the compression property on an inode also sets BTRFS_INODE_COMPRESS on it, and btrfs_inode_flags_to_fsflags() reports that back as FS_COMPR_FL to FS_IOC_GETFLAGS. chattr(1), like any other FS_IOC_SETFLAGS caller, reads the current flags, flips only the bit the user asked for and writes the whole set back, so a request as unrelated as "chattr +i" reaches btrfs_fileattr_set() with FS_COMPR_FL set. btrfs_fileattr_set() takes that as a request to enable compression and overwrites the compression property with the algorithm from the mount options, falling back to zlib when the filesystem was not mounted with -o compress. The algorithm the user selected is silently replaced: # btrfs property set /mnt/foo compression zstd # btrfs property get /mnt/foo compression compression=zstd # chattr +i /mnt/foo # btrfs property get /mnt/foo compression compression=zlib Every chattr operation triggers this, not just +i, and directories are affected as well, so files created afterwards inherit the wrong algorithm too. On a filesystem mounted with -o compress=lzo the property is replaced with lzo instead. Recovering needs a chattr -i first, because the immutable flag rejects the setxattr that "btrfs property set" issues. Prefer the algorithm recorded in the compression property and only fall back to the mount default when there is no property, so that unrelated flag changes no longer overwrite the user's choice. Inodes that have the compress flag set but no property still get the default, so they behave as before. Reviewed-by: Qu Wenruo Signed-off-by: Sam Ho Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 21 ++++++++++++++++++--- 1 file changed, 18 insertions(+), 3 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 72bc9d4f77087b..e4b2da31a0d5de 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -384,6 +384,7 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap, inode_flags &= ~BTRFS_INODE_COMPRESS; inode_flags |= BTRFS_INODE_NOCOMPRESS; } else if (fsflags & FS_COMPR_FL) { + enum btrfs_compression_type comp_type; if (IS_SWAPFILE(&inode->vfs_inode)) return -ETXTBSY; @@ -391,9 +392,23 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap, inode_flags |= BTRFS_INODE_COMPRESS; inode_flags &= ~BTRFS_INODE_NOCOMPRESS; - comp = btrfs_compress_type2str(fs_info->compress_type); - if (!comp || comp[0] == 0) - comp = btrfs_compress_type2str(BTRFS_COMPRESS_ZLIB); + /* + * Keep the algorithm recorded in the compression property, + * otherwise changing an unrelated attribute would reset it to + * the mount default, since FS_IOC_SETFLAGS callers write back + * the whole flag set they got from FS_IOC_GETFLAGS and that + * includes FS_COMPR_FL for any inode carrying the property. + * + * Inodes with the compress flag set but no property keep using + * the mount default, so they behave as before. + */ + if (inode->prop_compress) + comp_type = inode->prop_compress; + else if (fs_info->compress_type) + comp_type = fs_info->compress_type; + else + comp_type = BTRFS_COMPRESS_ZLIB; + comp = btrfs_compress_type2str(comp_type); } else { inode_flags &= ~(BTRFS_INODE_COMPRESS | BTRFS_INODE_NOCOMPRESS); } From 034bc75d32e90df344aef4fca60c51cc40cb6080 Mon Sep 17 00:00:00 2001 From: Ayman Chaudhry Date: Wed, 26 Aug 2026 10:13:12 +0000 Subject: [PATCH 384/857] pmdomain: renesas: rcar-sysc: Update description of chan_offs, chan_bit, isr_bit The descriptions of `rcar_sysc_area.chan_offs`, `rcar_sysc_area.chan_bits`, and `rcar_sysc_area.isr_bit` do not clearly document how these fields are used. When `rcar_sysc_area.flags` is set to `PD_ALWAYS_ON` (i.e. `PD_NO_CR`), these fields are ignored, therefore improve the description of `rcar_sysc_area.chan_offs`, `rcar_sysc_area.chan_bit`, and `rcar_sysc_area.isr_bit` to make it clear that the field is set to 0 if power is always on. Signed-off-by: Ayman Chaudhry Signed-off-by: Ulf Hansson --- drivers/pmdomain/renesas/rcar-sysc.h | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/drivers/pmdomain/renesas/rcar-sysc.h b/drivers/pmdomain/renesas/rcar-sysc.h index 07ffce310686b2..ef598ecc710a26 100644 --- a/drivers/pmdomain/renesas/rcar-sysc.h +++ b/drivers/pmdomain/renesas/rcar-sysc.h @@ -29,9 +29,15 @@ struct rcar_sysc_area { const char *name; - u16 chan_offs; /* Offset of PWRSR register for this area */ - u8 chan_bit; /* Bit in PWR* (except for PWRUP in PWRSR) */ - u8 isr_bit; /* Bit in SYSCI*R */ + u16 chan_offs; /* + * PWRSR register offset; or 0 if area is + * always on + */ + u8 chan_bit; /* + * Bit in PWR* (except for PWRUP in PWRSR); + * 0 if area is always on + */ + u8 isr_bit; /* Bit in SYSCI*R; 0 if area is always on */ s8 parent; /* -1 if none */ u8 flags; /* See PD_* */ }; From d48265fe57be78f212a4eeb58588575ea0df4e1a Mon Sep 17 00:00:00 2001 From: Hardeep Sharma Date: Thu, 27 Aug 2026 23:21:06 +0530 Subject: [PATCH 385/857] pmdomain: qcom: rpmhpd: Add power domains for Kuno Add the RPMh power-domain data for the Qualcomm Kuno SoC to enable the new qcom,kuno-rpmhpd compatible to expose RPMh power domains. Kuno exposes CX, MX and MXC domains (including always-on variants). Reviewed-by: Konrad Dybcio Reviewed-by: Abel Vesa Signed-off-by: Hardeep Sharma Signed-off-by: Ulf Hansson --- drivers/pmdomain/qcom/rpmhpd.c | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/drivers/pmdomain/qcom/rpmhpd.c b/drivers/pmdomain/qcom/rpmhpd.c index 90743275942db2..725ce802f68317 100644 --- a/drivers/pmdomain/qcom/rpmhpd.c +++ b/drivers/pmdomain/qcom/rpmhpd.c @@ -681,6 +681,21 @@ static const struct rpmhpd_desc kaanapali_desc = { .num_pds = ARRAY_SIZE(kaanapali_rpmhpds), }; +/* Kuno RPMH power domains */ +static struct rpmhpd *kuno_rpmhpds[] = { + [RPMHPD_CX] = &cx, + [RPMHPD_CX_AO] = &cx_ao, + [RPMHPD_MX] = &mx, + [RPMHPD_MX_AO] = &mx_ao, + [RPMHPD_MXC] = &mxc, + [RPMHPD_MXC_AO] = &mxc_ao, +}; + +static const struct rpmhpd_desc kuno_desc = { + .rpmhpds = kuno_rpmhpds, + .num_pds = ARRAY_SIZE(kuno_rpmhpds), +}; + /* Hawi RPMH powerdomains */ static struct rpmhpd *hawi_rpmhpds[] = { [RPMHPD_CX] = &cx, @@ -885,6 +900,7 @@ static const struct of_device_id rpmhpd_match_table[] = { { .compatible = "qcom,glymur-rpmhpd", .data = &glymur_desc }, { .compatible = "qcom,hawi-rpmhpd", .data = &hawi_desc }, { .compatible = "qcom,kaanapali-rpmhpd", .data = &kaanapali_desc }, + { .compatible = "qcom,kuno-rpmhpd", .data = &kuno_desc }, { .compatible = "qcom,milos-rpmhpd", .data = &milos_desc }, { .compatible = "qcom,nord-rpmhpd", .data = &nord_desc }, { .compatible = "qcom,qcs615-rpmhpd", .data = &qcs615_desc }, From 425ca686928dda312a11ab470a2655b9ccb73e65 Mon Sep 17 00:00:00 2001 From: Anurag Pateriya Date: Wed, 26 Aug 2026 21:54:54 +0800 Subject: [PATCH 386/857] pmdomain: qcom: rpmhpd: Add NMXC power domain for Nord Add the nmxc.lvl RPMh resource and register it in the Nord power domain table. Nord supplies the NSP memory rail from this dedicated resource rather than from the shared MX rail, so consumers need it exposed as its own power domain. Signed-off-by: Anurag Pateriya Reviewed-by: Abel Vesa Reviewed-by: Konrad Dybcio Signed-off-by: Shawn Guo Signed-off-by: Ulf Hansson --- drivers/pmdomain/qcom/rpmhpd.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/drivers/pmdomain/qcom/rpmhpd.c b/drivers/pmdomain/qcom/rpmhpd.c index 725ce802f68317..321a4e2cf1dba3 100644 --- a/drivers/pmdomain/qcom/rpmhpd.c +++ b/drivers/pmdomain/qcom/rpmhpd.c @@ -198,6 +198,11 @@ static struct rpmhpd mxc_ao = { .res_name = "mxc.lvl", }; +static struct rpmhpd nmxc = { + .pd = { .name = "nmxc", }, + .res_name = "nmxc.lvl", +}; + static struct rpmhpd nsp = { .pd = { .name = "nsp", }, .res_name = "nsp.lvl", @@ -327,6 +332,7 @@ static struct rpmhpd *nord_rpmhpds[] = { [RPMHPD_MX_AO] = &mx_ao, [RPMHPD_MXC] = &mxc, [RPMHPD_MXC_AO] = &mxc_ao, + [RPMHPD_NMXC] = &nmxc, [RPMHPD_NSP0] = &nsp0, [RPMHPD_NSP1] = &nsp1, [RPMHPD_NSP2] = &nsp2, From ef69f75643c37dcd79e874dbf19ef94dbc75fb64 Mon Sep 17 00:00:00 2001 From: Sreeshankar K Date: Tue, 1 Sep 2026 09:24:43 +0000 Subject: [PATCH 387/857] pmdomain: qcom: rpmhpd: Add RPMh power domains for SM7250 Add RPMh power domains for SM7250 SoC. Signed-off-by: Sreeshankar K Reviewed-by: Abel Vesa Reviewed-by: Konrad Dybcio Signed-off-by: Ulf Hansson --- drivers/pmdomain/qcom/rpmhpd.c | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/drivers/pmdomain/qcom/rpmhpd.c b/drivers/pmdomain/qcom/rpmhpd.c index 321a4e2cf1dba3..fa8fed5bd3e585 100644 --- a/drivers/pmdomain/qcom/rpmhpd.c +++ b/drivers/pmdomain/qcom/rpmhpd.c @@ -492,6 +492,24 @@ static const struct rpmhpd_desc sm7150_desc = { .num_pds = ARRAY_SIZE(sm7150_rpmhpds), }; +/* SM7250 RPMH powerdomains */ +static struct rpmhpd *sm7250_rpmhpds[] = { + [RPMHPD_CX] = &cx_w_mx_parent, + [RPMHPD_CX_AO] = &cx_ao_w_mx_parent, + [RPMHPD_EBI] = &ebi, + [RPMHPD_GFX] = &gfx, + [RPMHPD_LCX] = &lcx, + [RPMHPD_LMX] = &lmx, + [RPMHPD_MSS] = &mss, + [RPMHPD_MX] = &mx, + [RPMHPD_MX_AO] = &mx_ao, +}; + +static const struct rpmhpd_desc sm7250_desc = { + .rpmhpds = sm7250_rpmhpds, + .num_pds = ARRAY_SIZE(sm7250_rpmhpds), +}; + /* SM8150 RPMH powerdomains */ static struct rpmhpd *sm8150_rpmhpds[] = { [SM8150_CX] = &cx_w_mx_parent, @@ -928,6 +946,7 @@ static const struct of_device_id rpmhpd_match_table[] = { { .compatible = "qcom,sm4450-rpmhpd", .data = &sm4450_desc }, { .compatible = "qcom,sm6350-rpmhpd", .data = &sm6350_desc }, { .compatible = "qcom,sm7150-rpmhpd", .data = &sm7150_desc }, + { .compatible = "qcom,sm7250-rpmhpd", .data = &sm7250_desc }, { .compatible = "qcom,sm8150-rpmhpd", .data = &sm8150_desc }, { .compatible = "qcom,sm8250-rpmhpd", .data = &sm8250_desc }, { .compatible = "qcom,sm8350-rpmhpd", .data = &sm8350_desc }, From ca52f4764c8754d006e53cd7be3f2cb1a2b98fa4 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:19:56 +0300 Subject: [PATCH 388/857] Bluetooth: L2CAP: take chan->lock for l2cap_chan_add/ready/del chan->lock must be held for __l2cap_chan_add as eg. calls to l2cap_chan_close assume chan->conn writes are guarded by it. It must be held for l2cap_chan_del() due to l2cap_sock.c:l2cap_chan_conn, l2cap_monitor_timeout, etc. Similarly it should be held for l2cap_ops::ready (assumed in 6lowpan.c). Also teardown usually has chan->lock held, it should always have it held to have the same locking context. The lock is not correctly held by l2cap_core in several places. Add the missing locks for l2cap_chan_del/add/ready(), except in l2cap_ecred_rsp_defer() which needs separate fix as it needs lock nesting. Fixes: 6fef032af009 ("Bluetooth: L2CAP: Fix use-after-free in l2cap_sock_new_connection_cb()") Signed-off-by: Pauli Virtanen Reported-by: Eulgyu Kim Reported-by: Jaeyoung Chung Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 3 ++- net/bluetooth/l2cap_core.c | 15 +++++++++++++++ 2 files changed, 17 insertions(+), 1 deletion(-) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index 69d193fee351a6..43a67562b23879 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -973,7 +973,8 @@ int l2cap_chan_check_security(struct l2cap_chan *chan, bool initiator); void l2cap_chan_set_defaults(struct l2cap_chan *chan, struct l2cap_chan *pchan); int l2cap_ertm_init(struct l2cap_chan *chan); void l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan); -void __l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan); +void __l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan) + __must_hold(&chan->lock); typedef void (*l2cap_chan_func_t)(struct l2cap_chan *chan, void *data); void l2cap_chan_list(struct l2cap_conn *conn, l2cap_chan_func_t func, void *data); diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 2410e8f6d58787..e1430c183a8eb2 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -665,7 +665,9 @@ void __l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan) void l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan) { mutex_lock(&conn->lock); + l2cap_chan_lock(chan); __l2cap_chan_add(conn, chan); + l2cap_chan_unlock(chan); mutex_unlock(&conn->lock); } @@ -4079,6 +4081,8 @@ static struct l2cap_chan *l2cap_new_connection(struct l2cap_conn *conn, if (!chan) return NULL; + l2cap_chan_lock(chan); + l2cap_chan_set_defaults(chan, pchan); chan->ops = pchan->ops; @@ -4087,10 +4091,13 @@ static struct l2cap_chan *l2cap_new_connection(struct l2cap_conn *conn, if (pchan->ops->new_connection && pchan->ops->new_connection(pchan, chan) < 0) { l2cap_chan_del(chan, 0); + l2cap_chan_unlock(chan); l2cap_chan_put(chan); return NULL; } + l2cap_chan_unlock(chan); + return chan; } @@ -5061,6 +5068,8 @@ static int l2cap_le_connect_req(struct l2cap_conn *conn, goto response_unlock; } + l2cap_chan_lock(chan); + bacpy(&chan->src, &conn->hcon->src); bacpy(&chan->dst, &conn->hcon->dst); chan->src_type = bdaddr_src_type(conn->hcon); @@ -5094,6 +5103,8 @@ static int l2cap_le_connect_req(struct l2cap_conn *conn, result = L2CAP_CR_LE_SUCCESS; } + l2cap_chan_unlock(chan); + response_unlock: l2cap_chan_unlock(pchan); l2cap_chan_put(pchan); @@ -5286,6 +5297,8 @@ static inline int l2cap_ecred_conn_req(struct l2cap_conn *conn, continue; } + l2cap_chan_lock(chan); + bacpy(&chan->src, &conn->hcon->src); bacpy(&chan->dst, &conn->hcon->dst); chan->src_type = bdaddr_src_type(conn->hcon); @@ -5318,6 +5331,8 @@ static inline int l2cap_ecred_conn_req(struct l2cap_conn *conn, } else { l2cap_chan_ready(chan); } + + l2cap_chan_unlock(chan); } unlock: From 368dc7fcaced7635f5c444240e24bd882963e8ad Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:19:57 +0300 Subject: [PATCH 389/857] Bluetooth: L2CAP: add l2cap_chan_close_unlocked() and locking helpers l2cap_chan_close() requires holding chan->lock and chan->conn->lock if associated chan->conn exists, to guard eg. conn->chan_l. Taking the locks with right ordering requires handling a race condition. Add helper function l2cap_chan_(un)lock_conn that do the locking right. Add l2cap_chan_close_unlocked() that does not require locks to be held, as all callsites do this lock -> close -> unlock pattern. Link: https://syzkaller.appspot.com/bug?extid=0e4ebcc970728e056324 Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 17 ++++++++++ net/bluetooth/l2cap_core.c | 61 ++++++++++++++++++++++++++++++++--- 2 files changed, 73 insertions(+), 5 deletions(-) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index 43a67562b23879..84f557d354cae3 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -962,6 +962,8 @@ int l2cap_add_scid(struct l2cap_chan *chan, __u16 scid); struct l2cap_chan *l2cap_chan_create(void); void l2cap_chan_close(struct l2cap_chan *chan, int reason); +void l2cap_chan_close_unlocked(struct l2cap_chan *chan, int reason) + __must_not_hold(&chan->lock); int l2cap_chan_connect(struct l2cap_chan *chan, __le16 psm, u16 cid, bdaddr_t *dst, u8 dst_type, u16 timeout); int l2cap_chan_reconfigure(struct l2cap_chan *chan, __u16 mtu); @@ -988,4 +990,19 @@ void l2cap_conn_put(struct l2cap_conn *conn); int l2cap_register_user(struct l2cap_conn *conn, struct l2cap_user *user); void l2cap_unregister_user(struct l2cap_conn *conn, struct l2cap_user *user); +bool l2cap_chan_lock_conn(struct l2cap_chan *chan) + __acquires(&chan->lock) __cond_acquires(true, &chan->conn->lock); + +/* Release macro for l2cap_chan_lock_conn, so context analysis understands it */ +#define l2cap_chan_unlock_conn(chan, conn_locked) \ + ({ \ + struct l2cap_chan *__chan = (chan); \ + struct l2cap_conn *__conn = __chan->conn; \ + l2cap_chan_unlock(__chan); \ + if (conn_locked) { \ + mutex_unlock(&__conn->lock); \ + l2cap_conn_put(__conn); \ + } \ + }) + #endif /* __L2CAP_H */ diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index e1430c183a8eb2..3bf877be081d24 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -59,6 +59,7 @@ static void l2cap_tx(struct l2cap_chan *chan, struct l2cap_ctrl *control, static void l2cap_retrans_timeout(struct work_struct *work); static void l2cap_monitor_timeout(struct work_struct *work); static void l2cap_ack_timeout(struct work_struct *work); +static void __l2cap_chan_close(struct l2cap_chan *chan, int reason); static inline u8 bdaddr_type(u8 link_type, u8 bdaddr_type) { @@ -422,7 +423,7 @@ static void l2cap_chan_timeout(struct work_struct *work) else reason = ETIMEDOUT; - l2cap_chan_close(chan, reason); + __l2cap_chan_close(chan, reason); chan->ops->close(chan); @@ -829,7 +830,7 @@ static void l2cap_chan_connect_reject(struct l2cap_chan *chan) l2cap_send_cmd(conn, chan->ident, L2CAP_CONN_RSP, sizeof(rsp), &rsp); } -void l2cap_chan_close(struct l2cap_chan *chan, int reason) +static void __l2cap_chan_close(struct l2cap_chan *chan, int reason) { struct l2cap_conn *conn = chan->conn; @@ -878,8 +879,58 @@ void l2cap_chan_close(struct l2cap_chan *chan, int reason) break; } } + +void l2cap_chan_close(struct l2cap_chan *chan, int reason) +{ + __l2cap_chan_close(chan, reason); +} EXPORT_SYMBOL(l2cap_chan_close); +/* Take chan->lock. If chan->conn is non-NULL, take new reference on it, take + * chan->conn->lock, and return true. Otherwise return false. + */ +bool l2cap_chan_lock_conn(struct l2cap_chan *chan) + __context_unsafe(/* conditional locking */) +{ + /* Handle conn->lock > chan->lock ordering + race on chan->conn */ + for (;;) { + struct l2cap_conn *conn; + + l2cap_chan_lock(chan); + conn = chan->conn; + if (conn) + l2cap_conn_get(conn); + l2cap_chan_unlock(chan); + + if (conn) + mutex_lock(&conn->lock); + + l2cap_chan_lock(chan); + + if (chan->conn != conn) { + l2cap_chan_unlock(chan); + if (conn) { + mutex_unlock(&conn->lock); + l2cap_conn_put(conn); + } + schedule(); + continue; + } + + return chan->conn; + } +} + +void l2cap_chan_close_unlocked(struct l2cap_chan *chan, int reason) +{ + bool have_conn; + + have_conn = l2cap_chan_lock_conn(chan); + __l2cap_chan_close(chan, reason); + l2cap_chan_unlock_conn(chan, have_conn); +} +EXPORT_SYMBOL(l2cap_chan_close_unlocked); + static inline u8 l2cap_get_auth_type(struct l2cap_chan *chan) { switch (chan->chan_type) { @@ -1570,7 +1621,7 @@ static void l2cap_conn_start(struct l2cap_conn *conn) if (!l2cap_mode_supported(chan->mode, conn->feat_mask) && test_bit(CONF_STATE2_DEVICE, &chan->conf_state)) { - l2cap_chan_close(chan, ECONNRESET); + __l2cap_chan_close(chan, ECONNRESET); l2cap_chan_unlock(chan); continue; } @@ -1578,7 +1629,7 @@ static void l2cap_conn_start(struct l2cap_conn *conn) if (l2cap_check_enc_key_size(conn->hcon, chan)) l2cap_start_connection(chan); else - l2cap_chan_close(chan, ECONNREFUSED); + __l2cap_chan_close(chan, ECONNREFUSED); } else if (chan->state == BT_CONNECT2) { struct l2cap_conn_rsp rsp; @@ -7667,7 +7718,7 @@ static inline void l2cap_check_encryption(struct l2cap_chan *chan, u8 encrypt) __set_chan_timer(chan, L2CAP_ENC_TIMEOUT); } else if (chan->sec_level == BT_SECURITY_HIGH || chan->sec_level == BT_SECURITY_FIPS) - l2cap_chan_close(chan, ECONNREFUSED); + __l2cap_chan_close(chan, ECONNREFUSED); } else { if (chan->sec_level == BT_SECURITY_MEDIUM) __clear_chan_timer(chan); From 86b3773598263417b00fd6563c01561a6cebaf64 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:19:58 +0300 Subject: [PATCH 390/857] Bluetooth: L2CAP: fix race condition in l2cap_sock_shutdown() l2cap_sock_shutdown() has the race condition [Task 1] [Task 2] l2cap_sock_shutdown l2cap_sock_connect l2cap_chan_lock l2cap_chan_connect conn = ... /* == NULL*/ l2cap_chan_unlock ------------> l2cap_chan_lock if (conn) /* false */ __l2cap_chan_add(conn, chan) l2cap_chan_lock <-------------- l2cap_chan_unlock l2cap_chan_close /* chan->conn->lock not held! */ conn->lock protects conn->chan_l and is not properly held here. Use the l2cap_chan_close_unlocked() helper that ensures conn->lock is held for l2cap_chan_close(). Fixes: ab4eedb790ca ("Bluetooth: L2CAP: Fix corrupted list in hci_chan_del") Reported-by: Eulgyu Kim Reported-by: Jaeyoung Chung Link: https://lore.kernel.org/linux-bluetooth/20260824153908.2327306-1-jjy600901@snu.ac.kr/ Link: https://syzkaller.appspot.com/bug?extid=0e4ebcc970728e056324 Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_sock.c | 19 +------------------ 1 file changed, 1 insertion(+), 18 deletions(-) diff --git a/net/bluetooth/l2cap_sock.c b/net/bluetooth/l2cap_sock.c index b553b6356af81c..0265b650868216 100644 --- a/net/bluetooth/l2cap_sock.c +++ b/net/bluetooth/l2cap_sock.c @@ -1406,7 +1406,6 @@ static int l2cap_sock_shutdown(struct socket *sock, int how) { struct sock *sk = sock->sk; struct l2cap_chan *chan; - struct l2cap_conn *conn; int err = 0; BT_DBG("sock %p, sk %p, how %d", sock, sk, how); @@ -1463,23 +1462,7 @@ static int l2cap_sock_shutdown(struct socket *sock, int how) sk->sk_shutdown |= SEND_SHUTDOWN; release_sock(sk); - l2cap_chan_lock(chan); - /* prevent conn structure from being freed */ - conn = l2cap_conn_hold_unless_zero(chan->conn); - l2cap_chan_unlock(chan); - - if (conn) - /* mutex lock must be taken before l2cap_chan_lock() */ - mutex_lock(&conn->lock); - - l2cap_chan_lock(chan); - l2cap_chan_close(chan, 0); - l2cap_chan_unlock(chan); - - if (conn) { - mutex_unlock(&conn->lock); - l2cap_conn_put(conn); - } + l2cap_chan_close_unlocked(chan, 0); lock_sock(sk); From 4e64bd587b64561deb4ce014af521d8ba9db5898 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:19:59 +0300 Subject: [PATCH 391/857] Bluetooth: 6lowpan: use l2cap_chan_close_unlocked() 6lowpan.c is using l2cap_chan_close() without taking chan->conn->lock, so it may modify conn->chan_l without holding the guarding lock. Fix the locking by using the l2cap_chan_close_unlocked() helper that acquires the necessary locks. Fixes: 15f32cabf426 ("Bluetooth: 6lowpan: add missing l2cap_chan_lock()") Link: https://syzkaller.appspot.com/bug?extid=0e4ebcc970728e056324 Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/6lowpan.c | 27 +++++++++------------------ 1 file changed, 9 insertions(+), 18 deletions(-) diff --git a/net/bluetooth/6lowpan.c b/net/bluetooth/6lowpan.c index 30f4afa18bc8f2..4ea55950e599f0 100644 --- a/net/bluetooth/6lowpan.c +++ b/net/bluetooth/6lowpan.c @@ -921,9 +921,7 @@ static int bt_6lowpan_disconnect(struct l2cap_conn *conn, u8 dst_type) BT_DBG("peer %p chan %p", peer, peer->chan); - l2cap_chan_lock(peer->chan); - l2cap_chan_close(peer->chan, ENOENT); - l2cap_chan_unlock(peer->chan); + l2cap_chan_close_unlocked(peer->chan, ENOENT); return 0; } @@ -1025,9 +1023,9 @@ static void disconnect_all_peers(void) struct lowpan_peer *peer; int nchans; - /* l2cap_chan_close() cannot be called from RCU, and lock ordering - * chan->lock > devices_lock prevents taking write side lock, so copy - * then close. + /* l2cap_chan_close_unlocked() cannot be called from RCU, and lock + * ordering chan->lock > devices_lock prevents taking write side lock, + * so copy then close. */ rcu_read_lock(); @@ -1062,9 +1060,7 @@ static void disconnect_all_peers(void) spin_unlock(&devices_lock); for (i = 0; i < nchans; ++i) { - l2cap_chan_lock(chans[i]); - l2cap_chan_close(chans[i], ENOENT); - l2cap_chan_unlock(chans[i]); + l2cap_chan_close_unlocked(chans[i], ENOENT); l2cap_chan_put(chans[i]); } } while (nchans); @@ -1082,9 +1078,7 @@ static void do_enable_set(bool flag) mutex_lock(&set_lock); if (listen_chan) { - l2cap_chan_lock(listen_chan); - l2cap_chan_close(listen_chan, 0); - l2cap_chan_unlock(listen_chan); + l2cap_chan_close_unlocked(listen_chan, 0); l2cap_chan_put(listen_chan); } @@ -1132,9 +1126,7 @@ static ssize_t lowpan_control_write(struct file *fp, mutex_lock(&set_lock); if (listen_chan) { - l2cap_chan_lock(listen_chan); - l2cap_chan_close(listen_chan, 0); - l2cap_chan_unlock(listen_chan); + l2cap_chan_close_unlocked(listen_chan, 0); l2cap_chan_put(listen_chan); listen_chan = NULL; } @@ -1303,10 +1295,9 @@ static void __exit bt_6lowpan_exit(void) debugfs_remove(lowpan_control_debugfs); if (listen_chan) { - l2cap_chan_lock(listen_chan); - l2cap_chan_close(listen_chan, 0); - l2cap_chan_unlock(listen_chan); + l2cap_chan_close_unlocked(listen_chan, 0); l2cap_chan_put(listen_chan); + listen_chan = NULL; } disconnect_devices(); From c5123fddfef12f97a2a897fee339130944c95b38 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:20:00 +0300 Subject: [PATCH 392/857] Bluetooth: L2CAP: remove unused l2cap_chan_close() l2cap_chan_close() is now unused, and l2cap_chan_close_unlocked() should be used instead. Remove l2cap_chan_close(). Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 1 - net/bluetooth/l2cap_core.c | 6 ------ 2 files changed, 7 deletions(-) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index 84f557d354cae3..e395ab5493f70d 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -961,7 +961,6 @@ int l2cap_add_psm(struct l2cap_chan *chan, bdaddr_t *src, __le16 psm); int l2cap_add_scid(struct l2cap_chan *chan, __u16 scid); struct l2cap_chan *l2cap_chan_create(void); -void l2cap_chan_close(struct l2cap_chan *chan, int reason); void l2cap_chan_close_unlocked(struct l2cap_chan *chan, int reason) __must_not_hold(&chan->lock); int l2cap_chan_connect(struct l2cap_chan *chan, __le16 psm, u16 cid, diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 3bf877be081d24..348803067c3836 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -880,12 +880,6 @@ static void __l2cap_chan_close(struct l2cap_chan *chan, int reason) } } -void l2cap_chan_close(struct l2cap_chan *chan, int reason) -{ - __l2cap_chan_close(chan, reason); -} -EXPORT_SYMBOL(l2cap_chan_close); - /* Take chan->lock. If chan->conn is non-NULL, take new reference on it, take * chan->conn->lock, and return true. Otherwise return false. */ From 02122fd8002984b963c38a7c743934748d25214a Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:20:01 +0300 Subject: [PATCH 393/857] Bluetooth: 6lowpan: avoid concurrent peer_del() in bt_6lowpan_disconnect bt_6lowpan_disconnect() looks up and accesses peer->chan, without holding locks guaranteeing peer_del() cannot free the peer concurrently. Take devices_lock to ensure peer can be dereferenced safely. Fixes: 15f32cabf426 ("Bluetooth: 6lowpan: add missing l2cap_chan_lock()") Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/6lowpan.c | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/net/bluetooth/6lowpan.c b/net/bluetooth/6lowpan.c index 4ea55950e599f0..ddcdd2aff91fe2 100644 --- a/net/bluetooth/6lowpan.c +++ b/net/bluetooth/6lowpan.c @@ -912,16 +912,27 @@ static int bt_6lowpan_connect(bdaddr_t *addr, u8 dst_type) static int bt_6lowpan_disconnect(struct l2cap_conn *conn, u8 dst_type) { struct lowpan_peer *peer; + struct l2cap_chan *chan; BT_DBG("conn %p dst type %u", conn, dst_type); + spin_lock(&devices_lock); + peer = lookup_peer(conn); - if (!peer) + if (!peer) { + spin_unlock(&devices_lock); return -ENOENT; + } + + chan = peer->chan; + l2cap_chan_hold(chan); + + spin_unlock(&devices_lock); - BT_DBG("peer %p chan %p", peer, peer->chan); + BT_DBG("peer %p chan %p", peer, chan); - l2cap_chan_close_unlocked(peer->chan, ENOENT); + l2cap_chan_close_unlocked(chan, ENOENT); + l2cap_chan_put(chan); return 0; } From 62df59925278d0c09424ffa3ef7044922e40ddcc Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:20:02 +0300 Subject: [PATCH 394/857] Bluetooth: L2CAP: hold conn->lock for __l2cap_ecred_conn_rsp_defer __l2cap_ecred_conn_rsp_defer() > __l2cap_chan_list_id() accesses conn->chan_l which is guarded by conn->lock. The lock fails to be held when calling from l2cap_sock.c. Fix by using l2cap_chan_conn_lock(), and taking the locks in required order l2cap_conn::lock > l2cap_chan::lock > sk. Leave fast path with sk->sk_state precheck. Move the L2CAP defer handling to l2cap_sock_defer(). The code should also take l2cap_chan_lock() for sibling channels, but that needs separate fix due to lock nesting. Fixes: ab4eedb790ca ("Bluetooth: L2CAP: Fix corrupted list in hci_chan_del") Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_sock.c | 68 +++++++++++++++++++++++++++----------- 1 file changed, 48 insertions(+), 20 deletions(-) diff --git a/net/bluetooth/l2cap_sock.c b/net/bluetooth/l2cap_sock.c index 0265b650868216..dee3025f0ec24f 100644 --- a/net/bluetooth/l2cap_sock.c +++ b/net/bluetooth/l2cap_sock.c @@ -1243,40 +1243,68 @@ static void l2cap_publish_rx_avail(struct l2cap_chan *chan) l2cap_chan_rx_avail(chan, -1); } -static int l2cap_sock_recvmsg(struct socket *sock, struct msghdr *msg, - size_t len, int flags) +static int l2cap_sock_defer(struct sock *sk) { - struct sock *sk = sock->sk; - struct l2cap_pinfo *pi = l2cap_pi(sk); - int err; + struct l2cap_chan *chan = l2cap_pi(sk)->chan; + bool have_conn; + int err = 0; - if (unlikely(flags & MSG_ERRQUEUE)) - return sock_recv_errqueue(sk, msg, len, SOL_BLUETOOTH, - BT_SCM_ERROR); + /* Fast path check */ + lock_sock(sk); + if (sk->sk_state != BT_CONNECT2) { + release_sock(sk); + return 0; + } + release_sock(sk); + have_conn = l2cap_chan_lock_conn(chan); lock_sock(sk); if (sk->sk_state == BT_CONNECT2 && test_bit(BT_SK_DEFER_SETUP, &bt_sk(sk)->flags)) { - if (pi->chan->mode == L2CAP_MODE_EXT_FLOWCTL) { + err = 1; + + if (!have_conn) { + release_sock(sk); + err = -ENOTCONN; + } else if (chan->mode == L2CAP_MODE_EXT_FLOWCTL) { sk->sk_state = BT_CONNECTED; - pi->chan->state = BT_CONNECTED; - __l2cap_ecred_conn_rsp_defer(pi->chan); - } else if (bdaddr_type_is_le(pi->chan->src_type)) { + chan->state = BT_CONNECTED; + release_sock(sk); + __l2cap_ecred_conn_rsp_defer(chan); + } else if (bdaddr_type_is_le(chan->src_type)) { sk->sk_state = BT_CONNECTED; - pi->chan->state = BT_CONNECTED; - __l2cap_le_connect_rsp_defer(pi->chan); + chan->state = BT_CONNECTED; + release_sock(sk); + __l2cap_le_connect_rsp_defer(chan); } else { sk->sk_state = BT_CONFIG; - pi->chan->state = BT_CONFIG; - __l2cap_connect_rsp_defer(pi->chan); + chan->state = BT_CONFIG; + release_sock(sk); + __l2cap_connect_rsp_defer(chan); } - - err = 0; - goto done; + } else { + release_sock(sk); } - release_sock(sk); + l2cap_chan_unlock_conn(chan, have_conn); + return err; +} + +static int l2cap_sock_recvmsg(struct socket *sock, struct msghdr *msg, + size_t len, int flags) +{ + struct sock *sk = sock->sk; + struct l2cap_pinfo *pi = l2cap_pi(sk); + int err; + + if (unlikely(flags & MSG_ERRQUEUE)) + return sock_recv_errqueue(sk, msg, len, SOL_BLUETOOTH, + BT_SCM_ERROR); + + err = l2cap_sock_defer(sk); + if (err) + return err < 0 ? err : 0; if (sock->type == SOCK_STREAM) err = bt_sock_stream_recvmsg(sock, msg, len, flags); From 604e2f6d1e9588dc7b1d34409895fdb04b44b558 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:20:03 +0300 Subject: [PATCH 395/857] Bluetooth: L2CAP: hold l2cap_conn::lock in l2cap_connect_cfm() l2cap_new_connection() -> __l2cap_chan_add() modifies l2cap_conn::chan_l, which is guarded by l2cap_conn::lock. The lock is not held in l2cap_connect_cfm(). Fix by holding conn->lock in l2cap_connect_cfm() to make the locking systematic. Fixes: ab4eedb790ca ("Bluetooth: L2CAP: Fix corrupted list in hci_chan_del") Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_core.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 348803067c3836..116f809ab757d4 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -7648,6 +7648,8 @@ static void l2cap_connect_cfm(struct hci_conn *hcon, u8 status) * we left off, because the list lock would prevent calling the * potentially sleeping l2cap_chan_lock() function. */ + mutex_lock(&conn->lock); + pchan = l2cap_global_fixed_chan(NULL, hcon); while (pchan) { struct l2cap_chan *chan, *next; @@ -7672,6 +7674,8 @@ static void l2cap_connect_cfm(struct hci_conn *hcon, u8 status) pchan = next; } + mutex_unlock(&conn->lock); + l2cap_conn_ready(conn); } From 886931f0c7e6fb6b3f596fe2397eb3a309da2755 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:20:04 +0300 Subject: [PATCH 396/857] Bluetooth: L2CAP: add annotations for l2cap_chan list locking Add context analysis annotations for l2cap_conn::chan_l and chan_list locking. Add corresponding required annotations to accessors and callers. This is not complete chan_l annotation, l2cap_chan::list and l2cap_chan_del() locking is currently not fully correct, and needs separate fix + annotations. Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 7 ++-- net/bluetooth/l2cap_core.c | 61 ++++++++++++++++++++++++++++++++--- 2 files changed, 61 insertions(+), 7 deletions(-) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index e395ab5493f70d..a991fc07515cd1 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -668,7 +668,7 @@ struct l2cap_conn { struct l2cap_chan *smp; - struct list_head chan_l; + struct list_head chan_l __guarded_by(&lock); struct mutex lock; struct kref ref; struct list_head users; @@ -954,7 +954,8 @@ void l2cap_cleanup_sockets(void); bool l2cap_is_socket(struct socket *sock); void __l2cap_le_connect_rsp_defer(struct l2cap_chan *chan); -void __l2cap_ecred_conn_rsp_defer(struct l2cap_chan *chan); +void __l2cap_ecred_conn_rsp_defer(struct l2cap_chan *chan) + __must_hold(&chan->lock) __must_hold(&chan->conn->lock); void __l2cap_connect_rsp_defer(struct l2cap_chan *chan); int l2cap_add_psm(struct l2cap_chan *chan, bdaddr_t *src, __le16 psm); @@ -975,7 +976,7 @@ void l2cap_chan_set_defaults(struct l2cap_chan *chan, struct l2cap_chan *pchan); int l2cap_ertm_init(struct l2cap_chan *chan); void l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan); void __l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan) - __must_hold(&chan->lock); + __must_hold(&conn->lock) __must_hold(&chan->lock); typedef void (*l2cap_chan_func_t)(struct l2cap_chan *chan, void *data); void l2cap_chan_list(struct l2cap_conn *conn, l2cap_chan_func_t func, void *data); diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 116f809ab757d4..5af7da8d171f6a 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -44,8 +44,8 @@ bool enable_ecred = IS_ENABLED(CONFIG_BT_LE_L2CAP_ECRED); static u32 l2cap_feat_mask = L2CAP_FEAT_FIXED_CHAN | L2CAP_FEAT_UCD; -static LIST_HEAD(chan_list); static DEFINE_RWLOCK(chan_list_lock); +static __guarded_by(&chan_list_lock) LIST_HEAD(chan_list); static struct sk_buff *l2cap_build_cmd(struct l2cap_conn *conn, u8 code, u8 ident, u16 dlen, void *data); @@ -87,6 +87,7 @@ static inline u8 bdaddr_dst_type(struct hci_conn *hcon) static struct l2cap_chan *__l2cap_get_chan_by_dcid(struct l2cap_conn *conn, u16 cid) + __must_hold(&conn->lock) { struct l2cap_chan *c; @@ -99,6 +100,7 @@ static struct l2cap_chan *__l2cap_get_chan_by_dcid(struct l2cap_conn *conn, static struct l2cap_chan *__l2cap_get_chan_by_scid(struct l2cap_conn *conn, u16 cid) + __must_hold(&conn->lock) { struct l2cap_chan *c; @@ -114,6 +116,7 @@ static struct l2cap_chan *__l2cap_get_chan_by_scid(struct l2cap_conn *conn, */ static struct l2cap_chan *l2cap_get_chan_by_scid(struct l2cap_conn *conn, u16 cid) + __must_hold(&conn->lock) { struct l2cap_chan *c; @@ -129,6 +132,7 @@ static struct l2cap_chan *l2cap_get_chan_by_scid(struct l2cap_conn *conn, */ static struct l2cap_chan *l2cap_get_chan_by_dcid(struct l2cap_conn *conn, u16 cid) + __must_hold(&conn->lock) { struct l2cap_chan *c; @@ -141,6 +145,7 @@ static struct l2cap_chan *l2cap_get_chan_by_dcid(struct l2cap_conn *conn, static struct l2cap_chan *__l2cap_get_chan_by_ident(struct l2cap_conn *conn, u8 ident) + __must_hold(&conn->lock) { struct l2cap_chan *c; @@ -153,6 +158,7 @@ static struct l2cap_chan *__l2cap_get_chan_by_ident(struct l2cap_conn *conn, static struct l2cap_chan *__l2cap_global_chan_by_addr(__le16 psm, bdaddr_t *src, u8 src_type) + __must_hold_shared(&chan_list_lock) { struct l2cap_chan *c; @@ -230,6 +236,7 @@ int l2cap_add_scid(struct l2cap_chan *chan, __u16 scid) } static u16 l2cap_alloc_cid(struct l2cap_conn *conn) + __must_hold(&conn->lock) { u16 cid, dyn_end; @@ -728,6 +735,7 @@ EXPORT_SYMBOL_GPL(l2cap_chan_del); static void __l2cap_chan_list_id(struct l2cap_conn *conn, u16 id, l2cap_chan_func_t func, void *data) + __must_hold(&conn->lock) { struct l2cap_chan *chan, *l; @@ -739,6 +747,7 @@ static void __l2cap_chan_list_id(struct l2cap_conn *conn, u16 id, static void __l2cap_chan_list(struct l2cap_conn *conn, l2cap_chan_func_t func, void *data) + __must_hold(&conn->lock) { struct l2cap_chan *chan; @@ -806,6 +815,9 @@ static void l2cap_chan_ecred_connect_reject(struct l2cap_chan *chan) { l2cap_state_change(chan, BT_DISCONN); + lockdep_assert_held(&chan->lock); + lockdep_assert_held(&chan->conn->lock); + __l2cap_ecred_conn_rsp_defer(chan); } @@ -1423,6 +1435,7 @@ static void l2cap_ecred_defer_connect(struct l2cap_chan *chan, void *data) } static void l2cap_ecred_connect(struct l2cap_chan *chan) + __must_hold(&chan->conn->lock) { struct l2cap_conn *conn = chan->conn; struct l2cap_ecred_conn_data data; @@ -1456,6 +1469,7 @@ static void l2cap_ecred_connect(struct l2cap_chan *chan) } static void l2cap_le_start(struct l2cap_chan *chan) + __must_hold(&chan->conn->lock) { struct l2cap_conn *conn = chan->conn; @@ -1476,6 +1490,7 @@ static void l2cap_le_start(struct l2cap_chan *chan) } static void l2cap_start_connection(struct l2cap_chan *chan) + __must_hold(&chan->conn->lock) { if (chan->conn->hcon->type == LE_LINK) { l2cap_le_start(chan); @@ -1525,6 +1540,7 @@ static bool l2cap_check_enc_key_size(struct hci_conn *hcon, } static void l2cap_do_start(struct l2cap_chan *chan) + __must_hold(&chan->conn->lock) { struct l2cap_conn *conn = chan->conn; @@ -1591,6 +1607,7 @@ static void l2cap_send_disconn_req(struct l2cap_chan *chan, int err) /* ---- L2CAP connections ---- */ static void l2cap_conn_start(struct l2cap_conn *conn) + __must_hold(&conn->lock) { struct l2cap_chan *chan, *tmp; @@ -1599,6 +1616,8 @@ static void l2cap_conn_start(struct l2cap_conn *conn) list_for_each_entry_safe(chan, tmp, &conn->chan_l, list) { l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + if (chan->chan_type != L2CAP_CHAN_CONN_ORIENTED) { l2cap_chan_ready(chan); l2cap_chan_unlock(chan); @@ -1715,6 +1734,8 @@ static void l2cap_conn_ready(struct l2cap_conn *conn) l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + if (hcon->type == LE_LINK) { l2cap_le_start(chan); } else if (chan->chan_type != L2CAP_CHAN_CONN_ORIENTED) { @@ -1737,6 +1758,7 @@ static void l2cap_conn_ready(struct l2cap_conn *conn) /* Notify sockets that we cannot guaranty reliability anymore */ static void l2cap_conn_unreliable(struct l2cap_conn *conn, int err) + __must_hold(&conn->lock) { struct l2cap_chan *chan; @@ -3034,6 +3056,7 @@ static void l2cap_pass_to_tx_fbit(struct l2cap_chan *chan, /* Copy frame to all raw sockets on that connection */ static void l2cap_raw_recv(struct l2cap_conn *conn, struct sk_buff *skb) + __must_hold(&conn->lock) { struct sk_buff *nskb; struct l2cap_chan *chan; @@ -4087,6 +4110,7 @@ static void l2cap_conf_rfc_get(struct l2cap_chan *chan, void *rsp, int len) static inline int l2cap_command_rej(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_cmd_rej_unk *rej = (struct l2cap_cmd_rej_unk *) data; @@ -4119,6 +4143,7 @@ static inline int l2cap_command_rej(struct l2cap_conn *conn, */ static struct l2cap_chan *l2cap_new_connection(struct l2cap_conn *conn, struct l2cap_chan *pchan) + __must_hold(&conn->lock) { struct l2cap_chan *chan; @@ -4148,6 +4173,7 @@ static struct l2cap_chan *l2cap_new_connection(struct l2cap_conn *conn, static void l2cap_connect(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u8 *data, u8 rsp_code) + __must_hold(&conn->lock) __context_unsafe(/* conditional locking */) { struct l2cap_conn_req *req = (struct l2cap_conn_req *) data; @@ -4278,6 +4304,7 @@ static void l2cap_connect(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, static int l2cap_connect_req(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { if (cmd_len < sizeof(struct l2cap_conn_req)) return -EPROTO; @@ -4289,6 +4316,7 @@ static int l2cap_connect_req(struct l2cap_conn *conn, static int l2cap_connect_create_rsp(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_conn_rsp *rsp = (struct l2cap_conn_rsp *) data; u16 scid, dcid, result, status; @@ -4406,6 +4434,7 @@ static void cmd_reject_invalid_cid(struct l2cap_conn *conn, u8 ident, static inline int l2cap_config_req(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_conf_req *req = (struct l2cap_conf_req *) data; u16 dcid, flags; @@ -4519,6 +4548,7 @@ static inline int l2cap_config_req(struct l2cap_conn *conn, static inline int l2cap_config_rsp(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_conf_rsp *rsp = (struct l2cap_conf_rsp *)data; u16 scid, flags, result; @@ -4628,6 +4658,7 @@ static inline int l2cap_config_rsp(struct l2cap_conn *conn, static inline int l2cap_disconnect_req(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_disconn_req *req = (struct l2cap_disconn_req *) data; struct l2cap_disconn_rsp rsp; @@ -4669,6 +4700,7 @@ static inline int l2cap_disconnect_req(struct l2cap_conn *conn, static inline int l2cap_disconnect_rsp(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_disconn_rsp *rsp = (struct l2cap_disconn_rsp *) data; u16 dcid, scid; @@ -4756,6 +4788,7 @@ static inline int l2cap_information_req(struct l2cap_conn *conn, static inline int l2cap_information_rsp(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_info_rsp *rsp = (struct l2cap_info_rsp *) data; u16 type, result; @@ -4863,6 +4896,7 @@ static inline int l2cap_conn_param_update_req(struct l2cap_conn *conn, static int l2cap_le_connect_rsp(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_le_conn_rsp *rsp = (struct l2cap_le_conn_rsp *) data; struct hci_conn *hcon = conn->hcon; @@ -4969,6 +5003,7 @@ static void l2cap_put_ident(struct l2cap_conn *conn, u8 code, u8 id) static inline int l2cap_bredr_sig_cmd(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { int err = 0; @@ -5030,6 +5065,7 @@ static inline int l2cap_bredr_sig_cmd(struct l2cap_conn *conn, static int l2cap_le_connect_req(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_le_conn_req *req = (struct l2cap_le_conn_req *) data; struct l2cap_le_conn_rsp rsp; @@ -5178,6 +5214,7 @@ static int l2cap_le_connect_req(struct l2cap_conn *conn, static inline int l2cap_le_credits(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_le_credits *pkt; struct l2cap_chan *chan; @@ -5227,6 +5264,7 @@ static inline int l2cap_le_credits(struct l2cap_conn *conn, static inline int l2cap_ecred_conn_req(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_ecred_conn_req *req = (void *) data; DEFINE_RAW_FLEX(struct l2cap_ecred_conn_rsp, pdu, dcid, L2CAP_ECRED_MAX_CID); @@ -5399,6 +5437,7 @@ static inline int l2cap_ecred_conn_req(struct l2cap_conn *conn, static inline int l2cap_ecred_conn_rsp(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_ecred_conn_rsp *rsp = (void *) data; struct hci_conn *hcon = conn->hcon; @@ -5526,6 +5565,7 @@ static inline int l2cap_ecred_conn_rsp(struct l2cap_conn *conn, static inline int l2cap_ecred_reconf_req(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_ecred_reconf_req *req = (void *) data; struct l2cap_ecred_reconf_rsp rsp; @@ -5624,6 +5664,7 @@ static inline int l2cap_ecred_reconf_req(struct l2cap_conn *conn, static inline int l2cap_ecred_reconf_rsp(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_chan *chan, *tmp; struct l2cap_ecred_reconf_rsp *rsp = (void *)data; @@ -5664,6 +5705,7 @@ static inline int l2cap_ecred_reconf_rsp(struct l2cap_conn *conn, static inline int l2cap_le_command_rej(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { struct l2cap_cmd_rej_unk *rej = (struct l2cap_cmd_rej_unk *) data; struct l2cap_chan *chan; @@ -5691,6 +5733,7 @@ static inline int l2cap_le_command_rej(struct l2cap_conn *conn, static inline int l2cap_le_sig_cmd(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u16 cmd_len, u8 *data) + __must_hold(&conn->lock) { int err = 0; @@ -5755,6 +5798,7 @@ static inline int l2cap_le_sig_cmd(struct l2cap_conn *conn, static inline void l2cap_le_sig_channel(struct l2cap_conn *conn, struct sk_buff *skb) + __must_hold(&conn->lock) { struct hci_conn *hcon = conn->hcon; struct l2cap_cmd_hdr *cmd; @@ -5813,6 +5857,7 @@ static inline void l2cap_sig_send_mtu_rej(struct l2cap_conn *conn, u8 ident) static inline void l2cap_sig_channel(struct l2cap_conn *conn, struct sk_buff *skb) + __must_hold(&conn->lock) { struct hci_conn *hcon = conn->hcon; struct l2cap_cmd_hdr *cmd; @@ -7050,6 +7095,7 @@ static int l2cap_ecred_data_rcv(struct l2cap_chan *chan, struct sk_buff *skb) static void l2cap_data_channel(struct l2cap_conn *conn, u16 cid, struct sk_buff *skb) + __must_hold(&conn->lock) { struct l2cap_chan *chan; @@ -7158,6 +7204,7 @@ static void l2cap_conless_channel(struct l2cap_conn *conn, __le16 psm, } static void l2cap_recv_frame(struct l2cap_conn *conn, struct sk_buff *skb) + __must_hold(&conn->lock) { struct l2cap_hdr *lh = (void *) skb->data; struct hci_conn *hcon = conn->hcon; @@ -7267,9 +7314,9 @@ static struct l2cap_conn *l2cap_conn_add(struct hci_conn *hcon) hci_dev_test_flag(hcon->hdev, HCI_FORCE_BREDR_SMP))) conn->local_fixed_chan |= L2CAP_FC_SMP_BREDR; - mutex_init(&conn->lock); - - INIT_LIST_HEAD(&conn->chan_l); + scoped_guard(mutex_init, &conn->lock) { + INIT_LIST_HEAD(&conn->chan_l); + } INIT_LIST_HEAD(&conn->users); INIT_DELAYED_WORK(&conn->info_timer, l2cap_info_timeout); @@ -7485,6 +7532,8 @@ int l2cap_chan_connect(struct l2cap_chan *chan, __le16 psm, u16 cid, __l2cap_chan_add(conn, chan); + lockdep_assert_held(&chan->conn->lock); + /* l2cap_chan_add takes its own ref so we can drop this one */ hci_conn_drop(hcon); @@ -7707,6 +7756,8 @@ static void l2cap_disconn_cfm(struct hci_conn *hcon, u8 reason) } static inline void l2cap_check_encryption(struct l2cap_chan *chan, u8 encrypt) + __must_hold(&chan->lock) + __must_hold(&chan->conn->lock) { if (chan->chan_type != L2CAP_CHAN_CONN_ORIENTED) return; @@ -7739,6 +7790,8 @@ static void l2cap_security_cfm(struct hci_conn *hcon, u8 status, u8 encrypt) list_for_each_entry(chan, &conn->chan_l, list) { l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + BT_DBG("chan %p scid 0x%4.4x state %s", chan, chan->scid, state_to_string(chan->state)); From 26836086682ac9017a53b6bd3be1bb7e37dcec2c Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:20:06 +0300 Subject: [PATCH 397/857] Bluetooth: L2CAP: hold chan in l2cap_ecred_conn_rsp() l2cap_chan_del() calls l2cap_chan_put() to drop the conn->chan_l reference. If this was the last reference, UAF follows. l2cap_ecred_conn_rsp() iterates chan_l list and calls l2cap_chan_del() on some members, without holding chan reference. Fix by holding refcount while using chan after l2cap_chan_del(). Since orig is looked up by dcid provided by remote, it's also possible orig == chan, so reference needs to be held also after orig use. Fixes: 41c2713b204e ("Bluetooth: L2CAP: Fix possible crash on l2cap_ecred_conn_rsp") Assisted-by: deepseek-v4-flash Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_core.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 5af7da8d171f6a..876c0639c35c5a 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -5468,12 +5468,14 @@ static inline int l2cap_ecred_conn_rsp(struct l2cap_conn *conn, chan->state == BT_CONNECTED) continue; + l2cap_chan_hold(chan); l2cap_chan_lock(chan); /* Check that there is a dcid for each pending channel */ if (cmd_len < sizeof(dcid)) { l2cap_chan_del(chan, ECONNREFUSED); l2cap_chan_unlock(chan); + l2cap_chan_put(chan); continue; } @@ -5512,6 +5514,8 @@ static inline int l2cap_ecred_conn_rsp(struct l2cap_conn *conn, __set_chan_timer(orig, 0); l2cap_chan_unlock(orig); } + + l2cap_chan_put(chan); continue; } @@ -5557,6 +5561,7 @@ static inline int l2cap_ecred_conn_rsp(struct l2cap_conn *conn, } l2cap_chan_unlock(chan); + l2cap_chan_put(chan); } return err; From 761224d13f8a5d84a9c06b9948a37b1b1dbc7837 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:20:09 +0300 Subject: [PATCH 398/857] Bluetooth: L2CAP: make concurrent l2cap_set_timer() refcounting safe Since l2cap_set_timer() does not check return value of schedule_delayed_work(), two concurrent calls may result to l2cap_chan refcount leak. Change the refcounting by using mod_delayed_work() and checking its return value. Code paths aside from l2cap_chan_busy() hold chan->lock, so this has little correctness impact. Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index a991fc07515cd1..2315a3993c7bdb 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -847,12 +847,11 @@ static inline void l2cap_set_timer(struct l2cap_chan *chan, BT_DBG("chan %p state %s timeout %ld", chan, state_to_string(chan->state), timeout); - /* If delayed work cancelled do not hold(chan) - since it is already done with previous set_timer */ - if (!cancel_delayed_work(work)) - l2cap_chan_hold(chan); + l2cap_chan_hold(chan); - schedule_delayed_work(work, timeout); + /* put(chan) if timer was already queued so it already has a ref */ + if (mod_delayed_work(system_percpu_wq, work, timeout)) + l2cap_chan_put(chan); } static inline bool l2cap_clear_timer(struct l2cap_chan *chan, From 2c127ee79a0b260fa4a4042c36542e772f7c20f5 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Sat, 29 Aug 2026 17:20:10 +0300 Subject: [PATCH 399/857] Bluetooth: L2CAP: remove conditional locking from l2cap_connect() Context analysis does not understand conditional locking. Restructure l2cap_connect() by removing conditional locking at the cost of some code duplication, so that static analysis can see its content. Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_core.c | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 876c0639c35c5a..9da689f0a50a9f 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -4174,7 +4174,6 @@ static struct l2cap_chan *l2cap_new_connection(struct l2cap_conn *conn, static void l2cap_connect(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, u8 *data, u8 rsp_code) __must_hold(&conn->lock) - __context_unsafe(/* conditional locking */) { struct l2cap_conn_req *req = (struct l2cap_conn_req *) data; struct l2cap_conn_rsp rsp; @@ -4191,7 +4190,13 @@ static void l2cap_connect(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, &conn->hcon->dst, ACL_LINK); if (!pchan) { result = L2CAP_CR_BAD_PSM; - goto response; + + rsp.scid = cpu_to_le16(scid); + rsp.dcid = cpu_to_le16(dcid); + rsp.result = cpu_to_le16(result); + rsp.status = cpu_to_le16(status); + l2cap_send_cmd(conn, cmd->ident, rsp_code, sizeof(rsp), &rsp); + return; } l2cap_chan_lock(pchan); @@ -4273,9 +4278,6 @@ static void l2cap_connect(struct l2cap_conn *conn, struct l2cap_cmd_hdr *cmd, rsp.status = cpu_to_le16(status); l2cap_send_cmd(conn, cmd->ident, rsp_code, sizeof(rsp), &rsp); - if (!pchan) - return; - if (result == L2CAP_CR_PEND && status == L2CAP_CS_NO_INFO) { struct l2cap_info_req info; info.type = cpu_to_le16(L2CAP_IT_FEAT_MASK); From 7a749cbd4b0234cd6fbb6086a6c406d76878e941 Mon Sep 17 00:00:00 2001 From: Cong Nguyen Date: Tue, 1 Sep 2026 22:54:04 +0700 Subject: [PATCH 400/857] hwmon: (gpio-fan) take fan_data->lock in gpio_fan_shutdown() set_fan_speed() writes the control GPIOs one bit at a time. Every other caller locks around it; gpio_fan_shutdown() doesn't. If it races a locked caller, the GPIO writes can interleave and leave the fan at a speed neither caller asked for. Fixes: b95579cd8795 ("hwmon: (gpio-fan) Add a shutdown handler to poweroff the fans") Reported-by: Sashiko AI review Link: https://lore.kernel.org/r/20260830152150.27F5F1F000E9@smtp.kernel.org Assisted-by: Claude:claude-opus-4 Signed-off-by: Cong Nguyen Link: https://patch.msgid.link/20260901155404.1532092-1-congnt264@gmail.com Signed-off-by: Guenter Roeck --- drivers/hwmon/gpio-fan.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/drivers/hwmon/gpio-fan.c b/drivers/hwmon/gpio-fan.c index 7f36e5f6f22308..df8bd970760505 100644 --- a/drivers/hwmon/gpio-fan.c +++ b/drivers/hwmon/gpio-fan.c @@ -612,8 +612,11 @@ static void gpio_fan_shutdown(struct platform_device *pdev) { struct gpio_fan_data *fan_data = platform_get_drvdata(pdev); - if (fan_data->gpios) + if (fan_data->gpios) { + mutex_lock(&fan_data->lock); set_fan_speed(fan_data, 0); + mutex_unlock(&fan_data->lock); + } } static int gpio_fan_runtime_suspend(struct device *dev) From 512bf44847667f5b11276747d578990e3033e585 Mon Sep 17 00:00:00 2001 From: Rong Zhang Date: Wed, 2 Sep 2026 02:19:18 +0800 Subject: [PATCH 401/857] Bluetooth: Properly disable remote wakeup for MT7922/MT7925 on Ryzen platform It is reported that a remote wakeup could cause MT7922/MT7925's btusb interface completely unresponsive. Resetting the xHCI root hub doesn't help at all, and recovering from such a state needs a power cycle. All reports seen to be relevant to Ryzen-based laptops. These NICs are usually used as OEM components thanks to some sort of reference designs. Their popularity on other platforms is unclear. While there is still a chance that the quirk may exist on other platforms, be cautious and only apply the quirk to direct children of Ryzen platforms's root hubs for the time being. In most cases the root hub is on the SoC or PCH, which needs the quirk. Unfortunately, this can't distinguish root hubs on PCIe add-in cards. Such roughness should be acceptable, as PCIe USB controller add-in cards are less commonly used nowadays. On the other hand, applying the quirk doesn't hurt any functionalities either, as the device can still be used as a wakeup source if desired. Theoretically, we could retrieve the root hub's PCI vendor ID with some hierarchy magic, but that's too intrusive... Meanwhile, though device_set_wakeup_capable(false) is the correct fix for other NICs with fake remote wakeup capabilities, doing so for MT7922/MT7925 effectively prevents it from being used as wakeup sources as per userspace requests. Hence, return -EBUSY on runtime suspend to prevent the interface from being autosuspended while it's still opened, which has the same effect as device_set_wakeup_capable(false), since disabling remote wakeup simply causes the USB core to gate runtime autosuspend as well due to needs_remote_wakeup == 1. The interface can be safely autosuspended as long as remote wakeup is disabled, i.e., after closing the HCI device. Specifically, the interface may still take the advantage of remote wakeup in order to wake up the system from sleep if userspace has enabled it as a wakeup source. Fixes: e31d761628ad ("Bluetooth: btmtk: Disable remote wakeup for MT7922/MT7925") Tested-by: Rafael Passos Signed-off-by: Rong Zhang Signed-off-by: Luiz Augusto von Dentz --- drivers/bluetooth/btmtk.c | 10 ------ drivers/bluetooth/btusb.c | 73 ++++++++++++++++++++++++++++++++++++--- 2 files changed, 69 insertions(+), 14 deletions(-) diff --git a/drivers/bluetooth/btmtk.c b/drivers/bluetooth/btmtk.c index 73ba4a029e9ccf..ea0c320f982708 100644 --- a/drivers/bluetooth/btmtk.c +++ b/drivers/bluetooth/btmtk.c @@ -1374,16 +1374,6 @@ int btmtk_usb_setup(struct hci_dev *hdev) break; case 0x7922: case 0x7925: - /* - * A remote wakeup could cause the device completely unresponsive, and - * recovering from such a state needs a power cycle. - * - * Since the remote wakeup capability is super broken, just disable it - * to get rid of the troubles. The device can still be autosuspended - * when the bluetooth interface is closed. - */ - device_set_wakeup_capable(&btmtk_data->udev->dev, false); - fallthrough; case 0x7961: case 0x7902: case 0x6639: diff --git a/drivers/bluetooth/btusb.c b/drivers/bluetooth/btusb.c index 84614e60d142f1..03039ceaa77d33 100644 --- a/drivers/bluetooth/btusb.c +++ b/drivers/bluetooth/btusb.c @@ -6,6 +6,7 @@ * Copyright (C) 2005-2008 Marcel Holtmann */ +#include #include #include #include @@ -982,6 +983,7 @@ struct btqca_data { #define BTUSB_USE_ALT3_FOR_WBS 15 #define BTUSB_ALT6_CONTINUOUS_TX 16 #define BTUSB_HW_SSR_ACTIVE 17 +#define BTUSB_WAKEUP_BROKEN 18 struct btusb_data { struct hci_dev *hdev; @@ -2971,10 +2973,25 @@ static int btusb_send_frame_mtk(struct hci_dev *hdev, struct sk_buff *skb) } } +static inline bool platform_is_ryzen(void) +{ +#ifdef CONFIG_X86 + return boot_cpu_has(X86_FEATURE_ZEN); +#else + return false; +#endif +} + +static inline bool is_direct_child_of_root_hub(struct usb_device *udev) +{ + return udev->parent == udev->bus->root_hub; +} + static int btusb_mtk_setup(struct hci_dev *hdev) { struct btusb_data *data = hci_get_drvdata(hdev); struct btmtk_data *btmtk_data = hci_get_priv(hdev); + int err; /* MediaTek WMT vendor cmd requiring below USB resources to * complete the handshake. @@ -2991,7 +3008,40 @@ static int btusb_mtk_setup(struct hci_dev *hdev) btusb_mtk_claim_iso_intf(data); } - return btmtk_usb_setup(hdev); + err = btmtk_usb_setup(hdev); + if (err) + return err; + + switch (btmtk_data->dev_id) { + case 0x7922: + case 0x7925: + /* + * All reports seen to be relevant to Ryzen-based laptops. These + * NICs are usually used as OEM components thanks to some sort + * of reference designs. + * + * Their popularity on other platforms is unclear. While there + * is still a chance that the quirk may exist on other + * platforms, be cautious and only apply the quirk to direct + * children of Ryzen platforms's root hubs for the time being. + * + * In most cases the root hub is on the SoC or PCH, which needs + * the quirk. Unfortunately, this can't distinguish root hubs on + * PCIe add-in cards. Such roughness should be acceptable, as + * PCIe USB controller add-in cards are less commonly used + * nowadays. On the other hand, applying the quirk doesn't hurt + * any functionalities either, as the device can still be used + * as a wakeup source if desired. + * + * Theoretically, we could retrieve the root hub's PCI vendor ID + * with some hierarchy magic, but that's too intrusive... + */ + if (platform_is_ryzen() && is_direct_child_of_root_hub(data->udev)) + set_bit(BTUSB_WAKEUP_BROKEN, &data->flags); + break; + } + + return 0; } static int btusb_mtk_shutdown(struct hci_dev *hdev) @@ -4567,11 +4617,26 @@ static int btusb_suspend(struct usb_interface *intf, pm_message_t message) BT_DBG("intf %p", intf); - /* Don't auto-suspend if there are connections or discovery in - * progress; external suspend calls shall never fail. + /* + * It is reported that remote wakeup events could sometimes cause some + * adapters completely unresponsive. Resetting the xHCI root hub doesn't + * help at all, and recovering from such a state needs a power cycle. + * Since disabling remote wakeup simply causes the USB core to gate + * runtime autosuspend as well due to needs_remote_wakeup == 1, let's do + * this ourselves to make our life easier. The interface can be safely + * autosuspended as long as remote wakeup is disabled, i.e., after + * closing the HCI device. + * + * Don't auto-suspend if there are connections or discovery in progress. + * + * External suspend calls shall never fail. Specifically, a device with + * broken remote wakeup may still take the advantage of remote wakeup in + * order to wake up the system from sleep if userspace has enabled it as + * a wakeup source. */ if (PMSG_IS_AUTO(message) && - (hci_conn_count(data->hdev) || hci_discovery_active(data->hdev))) + ((test_bit(BTUSB_WAKEUP_BROKEN, &data->flags) && data->intf->needs_remote_wakeup) || + hci_conn_count(data->hdev) || hci_discovery_active(data->hdev))) return -EBUSY; if (data->suspend_count++) From 22a50c3745c5ca2927300a26ca2d09dd6ce71b0e Mon Sep 17 00:00:00 2001 From: Mario Limonciello Date: Tue, 1 Sep 2026 16:37:22 -0500 Subject: [PATCH 402/857] tpm: Call cmd_ready/go_idle for each command transmission Some TPM implementations, particularly fTPM using the CRB interface, require the TPM to transition through idle and ready states for each command rather than once per session. The current implementation calls cmd_ready once during tpm_chip_start() and go_idle once during tpm_chip_stop(). For fTPM, when multiple commands are sent without per-command idle transitions, subsequent commands timeout as the TPM is waiting for the transition. Fix this by calling cmd_ready before each command and go_idle on all exit paths in tpm_try_transmit(). For TPM implementations that don't require per-command transitions, the callbacks return immediately based on the start method. Remove the now-redundant per-session calls from tpm_chip_start() and tpm_chip_stop() along with their helpers. This resolves timeout errors during TPM initialization on systems where BIOS has already performed TPM startup. Signed-off-by: Mario Limonciello Reviewed-by: Jarkko Sakkinen Link: https://lore.kernel.org/r/20260901213723.3017371-1-mario.limonciello@amd.com Signed-off-by: Jarkko Sakkinen --- drivers/char/tpm/tpm-chip.c | 24 ------------------------ drivers/char/tpm/tpm-interface.c | 21 +++++++++++++++++++++ 2 files changed, 21 insertions(+), 24 deletions(-) diff --git a/drivers/char/tpm/tpm-chip.c b/drivers/char/tpm/tpm-chip.c index 12b7394b34bdce..0be5dbfaa72ebf 100644 --- a/drivers/char/tpm/tpm-chip.c +++ b/drivers/char/tpm/tpm-chip.c @@ -66,22 +66,6 @@ static void tpm_relinquish_locality(struct tpm_chip *chip) chip->locality = -1; } -static int tpm_cmd_ready(struct tpm_chip *chip) -{ - if (!chip->ops->cmd_ready) - return 0; - - return chip->ops->cmd_ready(chip); -} - -static int tpm_go_idle(struct tpm_chip *chip) -{ - if (!chip->ops->go_idle) - return 0; - - return chip->ops->go_idle(chip); -} - static void tpm_clk_enable(struct tpm_chip *chip) { if (chip->ops->clk_enable) @@ -116,13 +100,6 @@ int tpm_chip_start(struct tpm_chip *chip) } } - ret = tpm_cmd_ready(chip); - if (ret) { - tpm_relinquish_locality(chip); - tpm_clk_disable(chip); - return ret; - } - return 0; } EXPORT_SYMBOL_GPL(tpm_chip_start); @@ -137,7 +114,6 @@ EXPORT_SYMBOL_GPL(tpm_chip_start); */ void tpm_chip_stop(struct tpm_chip *chip) { - tpm_go_idle(chip); tpm_relinquish_locality(chip); tpm_clk_disable(chip); } diff --git a/drivers/char/tpm/tpm-interface.c b/drivers/char/tpm/tpm-interface.c index 1ccdbde98b69a8..b4e749e70b02e7 100644 --- a/drivers/char/tpm/tpm-interface.c +++ b/drivers/char/tpm/tpm-interface.c @@ -19,6 +19,7 @@ * calls to msleep. */ +#include #include #include #include @@ -89,8 +90,16 @@ static bool tpm_transmit_completed(u8 status, struct tpm_chip *chip) return status_masked == chip->ops->req_complete_val; } +static void tpm_go_idle(struct tpm_chip *chip) +{ + if (chip->ops->go_idle) + chip->ops->go_idle(chip); +} +DEFINE_FREE(tpm_go_idle, struct tpm_chip *, if (_T) tpm_go_idle(_T)) + static ssize_t tpm_try_transmit(struct tpm_chip *chip, void *buf, size_t bufsiz) { + struct tpm_chip *chip_idle __free(tpm_go_idle) = NULL; struct tpm_header *header = buf; int rc; ssize_t len = 0; @@ -113,6 +122,18 @@ static ssize_t tpm_try_transmit(struct tpm_chip *chip, void *buf, size_t bufsiz) return -E2BIG; } + if (chip->ops->cmd_ready) { + rc = chip->ops->cmd_ready(chip); + if (rc) { + dev_err(&chip->dev, + "%s: cmd_ready(): error %d\n", __func__, rc); + return rc; + } + } + + /* Ensure go_idle() is called on every exit path from here on. */ + chip_idle = chip; + rc = chip->ops->send(chip, buf, bufsiz, count); if (rc < 0) { if (rc != -EPIPE) From b85ed7f7259b49077955e835ecc32b32b75053e8 Mon Sep 17 00:00:00 2001 From: Charles Haithcock Date: Tue, 1 Sep 2026 15:42:49 -0600 Subject: [PATCH 403/857] watchdog: Differentiate scenarios when watchdog is closed Presently, when a watchdog device is closed, we print "watchdog did not stop" in a few different scenarios; 1. When nowayout is set 2. When the watchdog is able to close, has received the magic character to stop, but fails to close in device-specific code paths 3. When userspace deliberately closes it without stopping it For 1, we explicitly print we can not close because of nowayout. Nothing differentiates the other two however. This change adds a print to indicate the watchdog was closed while still running. Suggested-by: Guenter Roeck Signed-off-by: Charles Haithcock Link: https://patch.msgid.link/20260901214251.760184-1-chaithco@redhat.com [groeck: Fixed multi-line alignment] Signed-off-by: Guenter Roeck --- drivers/watchdog/watchdog_dev.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/drivers/watchdog/watchdog_dev.c b/drivers/watchdog/watchdog_dev.c index d7895009a2de39..d8ed77d8a220ab 100644 --- a/drivers/watchdog/watchdog_dev.c +++ b/drivers/watchdog/watchdog_dev.c @@ -305,6 +305,10 @@ static int watchdog_stop(struct watchdog_device *wdd) if (wdd->ops->stop) { clear_bit(WDOG_HW_RUNNING, &wdd->status); err = wdd->ops->stop(wdd); + if (err < 0) { + pr_err("watchdog%d: Failed to stop watchdog: %pe\n", + wdd->id, ERR_PTR(err)); + } trace_watchdog_stop(wdd, err); } else { set_bit(WDOG_HW_RUNNING, &wdd->status); From b74dc82a3b289b872eb4ceaa2a3528e9d9b53931 Mon Sep 17 00:00:00 2001 From: Mahad Ibrahim Date: Wed, 22 Jul 2026 23:02:46 +0000 Subject: [PATCH 404/857] lkdtm: use kmalloc() instead of __get_free_page lkdtm_debugfs_entry and direct_entry use __get_free_page to allocate a temporary buffer, perform copy_from_user to get the crashtype name, strim() to strip whitespace and find_crashtype to find the corresponding crashtype that is being requested. The lkdtm_debugfs_read uses __get_free_page to allocate a temporary buffer to store all the available crashtypes, and then copy it to userspace. The buffers that are allocated can be allocated with kmalloc as there is nothing special that requires a struct page, or the page allocator. kmalloc() additionally provides a better API that doesn't require ugly casts which obfuscate the code and kfree does not need to know the size of the freed object. Replace use of __get_free_page() with kmalloc(). Signed-off-by: Mahad Ibrahim Acked-by: Mike Rapoport (Microsoft) Link: https://patch.msgid.link/20260722230246.2869-1-mahad.ibrahim.dev@gmail.com Signed-off-by: Kees Cook --- drivers/misc/lkdtm/core.c | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/drivers/misc/lkdtm/core.c b/drivers/misc/lkdtm/core.c index ededa32d674405..01bebcb33bd47e 100644 --- a/drivers/misc/lkdtm/core.c +++ b/drivers/misc/lkdtm/core.c @@ -236,11 +236,11 @@ static ssize_t lkdtm_debugfs_entry(struct file *f, if (count >= PAGE_SIZE) return -EINVAL; - buf = (char *)__get_free_page(GFP_KERNEL); + buf = kmalloc(PAGE_SIZE, GFP_KERNEL); if (!buf) return -ENOMEM; if (copy_from_user(buf, user_buf, count)) { - free_page((unsigned long) buf); + kfree(buf); return -EFAULT; } /* NULL-terminate and remove enter */ @@ -248,7 +248,7 @@ static ssize_t lkdtm_debugfs_entry(struct file *f, strim(buf); crashtype = find_crashtype(buf); - free_page((unsigned long)buf); + kfree(buf); if (!crashtype) return -EINVAL; @@ -271,7 +271,7 @@ static ssize_t lkdtm_debugfs_read(struct file *f, char __user *user_buf, ssize_t out; char *buf; - buf = (char *)__get_free_page(GFP_KERNEL); + buf = kmalloc(PAGE_SIZE, GFP_KERNEL); if (buf == NULL) return -ENOMEM; @@ -290,7 +290,7 @@ static ssize_t lkdtm_debugfs_read(struct file *f, char __user *user_buf, out = simple_read_from_buffer(user_buf, count, off, buf, n); - free_page((unsigned long) buf); + kfree(buf); return out; } @@ -313,11 +313,11 @@ static ssize_t direct_entry(struct file *f, const char __user *user_buf, if (count < 1) return -EINVAL; - buf = (char *)__get_free_page(GFP_KERNEL); + buf = kmalloc(PAGE_SIZE, GFP_KERNEL); if (!buf) return -ENOMEM; if (copy_from_user(buf, user_buf, count)) { - free_page((unsigned long) buf); + kfree(buf); return -EFAULT; } /* NULL-terminate and remove enter */ @@ -325,7 +325,7 @@ static ssize_t direct_entry(struct file *f, const char __user *user_buf, strim(buf); crashtype = find_crashtype(buf); - free_page((unsigned long) buf); + kfree(buf); if (!crashtype) return -EINVAL; From 1a2657b31fcf212fc704861ce8649cd176679c3d Mon Sep 17 00:00:00 2001 From: Pengpeng Hou Date: Sun, 30 Aug 2026 20:50:44 +0800 Subject: [PATCH 405/857] hwmon: (aspeed-pwm-tacho) Propagate reset deassert errors aspeed_pwm_tacho_probe() installs its reset cleanup action and configures the controller after an unchecked reset deassertion. Stop probing when the reset controller rejects the transition, before the hwmon device becomes visible. Fixes: 18c514cc0e02 ("hwmon: (aspeed-pwm-tacho) Deassert reset in probe") Signed-off-by: Pengpeng Hou Link: https://patch.msgid.link/20260830125044.97718-1-pengpeng@iscas.ac.cn Signed-off-by: Guenter Roeck --- drivers/hwmon/aspeed-pwm-tacho.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/drivers/hwmon/aspeed-pwm-tacho.c b/drivers/hwmon/aspeed-pwm-tacho.c index 1c5945d4ba3777..bfce589c3fb1f1 100644 --- a/drivers/hwmon/aspeed-pwm-tacho.c +++ b/drivers/hwmon/aspeed-pwm-tacho.c @@ -934,7 +934,9 @@ static int aspeed_pwm_tacho_probe(struct platform_device *pdev) "missing or invalid reset controller device tree entry"); return PTR_ERR(priv->rst); } - reset_control_deassert(priv->rst); + ret = reset_control_deassert(priv->rst); + if (ret) + return ret; ret = devm_add_action_or_reset(dev, aspeed_pwm_tacho_remove, priv); if (ret) From d53f7aff1880bf1514bbd1ae13d0c3ee5a96a937 Mon Sep 17 00:00:00 2001 From: ChenXiaoSong Date: Tue, 25 Aug 2026 13:51:06 +0000 Subject: [PATCH 406/857] smb/server: support compound fid in notify requests A Windows client can send a compound request containing: Create Request, File: ; Notify Request The Notify Request uses FFFF...FFFF as the compound FID. Signed-off-by: ChenXiaoSong Signed-off-by: Namjae Jeon --- fs/smb/server/smb2pdu.c | 41 ----------------------------------------- 1 file changed, 41 deletions(-) diff --git a/fs/smb/server/smb2pdu.c b/fs/smb/server/smb2pdu.c index 0ecc52fde69ca0..ae15902a8fee1f 100644 --- a/fs/smb/server/smb2pdu.c +++ b/fs/smb/server/smb2pdu.c @@ -11870,47 +11870,6 @@ int smb2_notify(struct ksmbd_work *work) return -EIO; } - /* - * macOS backupd sends CHANGE_NOTIFY with FileId=FFFF...FFFF (share-root - * sentinel) to watch for changes on the share root without holding an - * open handle. Respond STATUS_PENDING + STATUS_NOTIFY_CLEANUP immediately; - * without this, backupd aborts Time Machine setup on STATUS_FILE_CLOSED. - */ - if (req->VolatileFileId == SMB2_NO_FID && - req->PersistentFileId == SMB2_NO_FID) { - in_work = ksmbd_alloc_work_struct(); - if (!in_work || allocate_interim_rsp_buf(in_work)) { - if (in_work) - ksmbd_free_work_struct(in_work); - rsp->hdr.Status = STATUS_INSUFFICIENT_RESOURCES; - smb2_set_err_rsp(work); - return 0; - } - if (setup_async_work(work, NULL, NULL)) { - ksmbd_free_work_struct(in_work); - rsp->hdr.Status = STATUS_INSUFFICIENT_RESOURCES; - smb2_set_err_rsp(work); - return 0; - } - smb2_send_interim_resp(work, STATUS_PENDING); - in_work->conn = work->conn; - in_hdr = smb_get_msg(in_work->response_buf); - memcpy(in_hdr, ksmbd_resp_buf_next(work), - __SMB2_HEADER_STRUCTURE_SIZE); - in_hdr->Flags |= SMB2_FLAGS_ASYNC_COMMAND; - in_hdr->Id.AsyncId = cpu_to_le64(work->async_id); - smb2_set_err_rsp(in_work); - in_hdr->Status = STATUS_NOTIFY_CLEANUP; - in_work->async_id = work->async_id; - work->async_id = 0; - release_async_work(work); - if (smb2_send_interim_work(in_work, work, false)) - ksmbd_debug(SMB, "failed to send notify cleanup\n"); - ksmbd_free_work_struct(in_work); - work->send_no_response = 1; - return 0; - } - /* * KSMBD does not implement a real change-notification backend. * Genuine SMB2 servers (and macOS smbfs) never complete a From e6bd36a5e24b595f8abe92c3df071b801df0d0e7 Mon Sep 17 00:00:00 2001 From: Gael Blivet Date: Sat, 29 Aug 2026 00:00:00 +0900 Subject: [PATCH 407/857] ksmbd: refactor smb2_notify() to a blocking wait The previous design registered a synthetic work struct (in_work) directly on conn->async_requests and deferred the response to a workqueue -- a bespoke async/cancel implementation duplicating what setup_async_work(), release_async_work(), and smb2_send_interim_resp() already provide for smb2_lock()'s pending byte-range lock. Replace it with that same pattern: setup_async_work() on the calling work itself, registered on fp->blocked_works, woken by cancel or by the handle closing via the existing set_close_state_blocked_works(). This removes the synthetic work struct, the notify_pendings list and its close-time drain, and the deferred workqueue send, leaving smb2_notify() sharing the same async/cancel machinery as smb2_lock() instead of its own separate copy. The worker now blocks on ksmbd_wq for as long as the watch stays open, instead of returning immediately. This also makes the skeleton ready for a future event-delivery implementation on the same blocking wait. Suggested-by: ChenXiaoSong Signed-off-by: Gael Blivet Assisted-by: Claude:claude-sonnet-5 Tested-by: ChenXiaoSong Reviewed-by: ChenXiaoSong Signed-off-by: Namjae Jeon --- fs/smb/server/ksmbd_work.c | 3 - fs/smb/server/ksmbd_work.h | 5 +- fs/smb/server/smb2pdu.c | 341 +++++++++++-------------------------- fs/smb/server/vfs_cache.c | 48 ------ fs/smb/server/vfs_cache.h | 6 - 5 files changed, 98 insertions(+), 305 deletions(-) diff --git a/fs/smb/server/ksmbd_work.c b/fs/smb/server/ksmbd_work.c index f35335307670fe..097de59f807cc7 100644 --- a/fs/smb/server/ksmbd_work.c +++ b/fs/smb/server/ksmbd_work.c @@ -57,7 +57,6 @@ struct ksmbd_work *ksmbd_alloc_work_struct(void) INIT_LIST_HEAD(&work->request_entry); INIT_LIST_HEAD(&work->async_request_entry); INIT_LIST_HEAD(&work->fp_entry); - INIT_LIST_HEAD(&work->notify_entry); INIT_LIST_HEAD(&work->aux_read_list); work->iov_alloc_cnt = ARRAY_SIZE(work->iov_inline); work->iov = work->iov_inline; @@ -87,8 +86,6 @@ void ksmbd_free_work_struct(struct ksmbd_work *work) if (work->async_id) ksmbd_release_id(&work->conn->async_ida, work->async_id); - if (work->owns_conn_ref) - ksmbd_conn_put(work->conn); ksmbd_fd_put(work, work->request_open); kmem_cache_free(work_cache, work); } diff --git a/fs/smb/server/ksmbd_work.h b/fs/smb/server/ksmbd_work.h index 0844aa929f55da..3a14e4d69aa1ef 100644 --- a/fs/smb/server/ksmbd_work.h +++ b/fs/smb/server/ksmbd_work.h @@ -91,8 +91,6 @@ struct ksmbd_work { bool compress_response:1; /* Is this SYNC or ASYNC ksmbd_work */ bool asynchronous:1; - /* Work owns a reference to @conn. */ - bool owns_conn_ref:1; bool need_invalidate_rkey:1; bool request_open_chseq_tracked:1; bool session_setup_reauth:1; @@ -115,9 +113,8 @@ struct ksmbd_work { struct list_head request_entry; /* List head at conn->async_requests */ struct list_head async_request_entry; + /* List head at ksmbd_file->blocked_works */ struct list_head fp_entry; - /* List head at ksmbd_file->notify_pendings */ - struct list_head notify_entry; }; /** diff --git a/fs/smb/server/smb2pdu.c b/fs/smb/server/smb2pdu.c index ae15902a8fee1f..e7c965e7b06849 100644 --- a/fs/smb/server/smb2pdu.c +++ b/fs/smb/server/smb2pdu.c @@ -58,10 +58,6 @@ static void __wbuf(struct ksmbd_work *work, void **req, void **rsp) } } -static struct ksmbd_work *smb2_notify_cancel_claim(void **argv); -static void smb2_notify_cancel_fn(void **argv); -static void smb2_complete_notify_cancel(struct ksmbd_work *in_work); - #define WORK_BUFFERS(w, rq, rs) __wbuf((w), (void **)&(rq), (void **)&(rs)) #define SMB2_CREATE_FILE_ATTRIBUTE_MASK \ @@ -1245,6 +1241,7 @@ void smb2_send_interim_resp(struct ksmbd_work *work, __le32 status) { struct smb2_hdr *rsp_hdr; struct ksmbd_work *in_work = ksmbd_alloc_work_struct(); + u16 command; if (!in_work) return; @@ -1268,6 +1265,23 @@ void smb2_send_interim_resp(struct ksmbd_work *work, __le32 status) smb2_set_err_rsp(in_work); rsp_hdr->Status = status; + /* + * Async interim responses are unsigned, but final responses must + * follow the normal signing rules. The synthetic work has no + * request buffer, so use the original work for request signing + * checks and the response header for SMB3 command selection. + */ + command = work->conn->ops->get_cmd_val(work); + if (status != STATUS_PENDING && !work->encrypted && work->sess && + work->conn->ops->set_sign_rsp && + (work->sess->sign || + (work->conn->ops->is_sign_req && + work->conn->ops->is_sign_req(work, command)))) { + in_work->sess = work->sess; + work->conn->ops->set_sign_rsp(in_work); + in_work->sess = NULL; + } + if (smb2_send_interim_work(in_work, work, true)) ksmbd_debug(SMB, "failed to send interim response\n"); ksmbd_free_work_struct(in_work); @@ -9699,7 +9713,6 @@ int smb2_cancel(struct ksmbd_work *work) struct smb2_hdr *hdr = smb_get_msg(work->request_buf); struct smb2_hdr *chdr; struct ksmbd_work *iter; - struct ksmbd_work *cancelled_notify = NULL; struct list_head *command_list; if (work->next_smb2_rcv_hdr_off) @@ -9737,23 +9750,11 @@ int smb2_cancel(struct ksmbd_work *work) "smb2 with AsyncId %llu cancelled command = 0x%x\n", le64_to_cpu(hdr->Id.AsyncId), le16_to_cpu(chdr->Command)); - if (iter->cancel_fn == smb2_notify_cancel_fn) - cancelled_notify = - smb2_notify_cancel_claim(iter->cancel_argv); - else if (iter->cancel_fn) + if (iter->cancel_fn) iter->cancel_fn(iter->cancel_argv); break; } spin_unlock(&conn->request_lock); - - /* - * Complete a cancelled notify before this CANCEL handler returns. - * Deferring it to the system workqueue lets a following request and - * its response overtake STATUS_CANCELLED, leaving clients waiting - * for the original notify even though the cancellation was accepted. - */ - if (cancelled_notify) - smb2_complete_notify_cancel(cancelled_notify); } else { command_list = &conn->requests; @@ -11710,137 +11711,26 @@ int smb2_oplock_break(struct ksmbd_work *work) return 0; } -/* - * Cancel handler for a deferred CHANGE_NOTIFY. Races against - * __ksmbd_close_fd()'s notify_pendings drain (vfs_cache.c), which can run - * concurrently on a different connection closing the same handle -- only - * one of the two may claim and free in_work, so both sides check - * list_empty() under fp->f_lock before touching it (list_del_init() - * leaves a node empty, so whichever side removes it first is the owner; - * the loser must not touch in_work again, since the winner may already be - * freeing it). - * - * smb2_cancel() holds conn->request_lock (a spinlock) for the entire - * time it walks conn->async_requests and calls this function -- so this - * runs with preemption disabled and must not sleep or re-acquire that - * same lock. release_async_work() does both (it takes conn->request_lock - * itself, and frees things that can involve sleeping paths), so calling - * it from here would self-deadlock the very thread processing the - * client's CANCEL command. ksmbd_conn_write() can also sleep (it takes - * conn's write mutex). So: do only the non-sleeping, no-relock cleanup - * inline here. smb2_cancel() sends and frees the claimed notify after it - * drops request_lock, preserving response order for a client CANCEL. The - * connection teardown caller has no such post-unlock path, so its wrapper - * defers the send and free to a workqueue. - */ -struct notify_cancel_ctx { - struct work_struct work; - struct ksmbd_work *in_work; +struct ksmbd_notify_req { + wait_queue_head_t wait; }; -static void smb2_send_notify_cancelled(struct ksmbd_work *work) -{ - struct smb2_hdr *hdr = smb_get_msg(work->response_buf); - struct ksmbd_conn *conn = work->conn; - struct ksmbd_session *sess; - - sess = ksmbd_session_lookup(conn, le64_to_cpu(hdr->SessionId)); - if (sess) { - work->sess = sess; - if (work->encrypted && sess->enc && conn->ops->encrypt_resp) { - conn->ops->encrypt_resp(work); - } else if (conn->ops->is_sign_req && conn->ops->set_sign_rsp && - conn->ops->is_sign_req(work, - conn->ops->get_cmd_val(work))) { - conn->ops->set_sign_rsp(work); - } - } - - ksmbd_conn_write(work); - if (sess) { - ksmbd_user_session_put(sess); - work->sess = NULL; - } -} - -static void smb2_notify_cancel_deferred(struct work_struct *w) -{ - struct notify_cancel_ctx *ctx = - container_of(w, struct notify_cancel_ctx, work); - struct ksmbd_conn *conn = ctx->in_work->conn; - - smb2_complete_notify_cancel(ctx->in_work); - kfree(ctx); - /* - * The connection teardown waits for r_count before destroying - * connection sessions and their proc entries. - */ - ksmbd_conn_r_count_dec(conn); -} - -static struct ksmbd_work *smb2_notify_cancel_claim(void **argv) -{ - struct ksmbd_work *in_work = (struct ksmbd_work *)argv[0]; - struct ksmbd_file *fp = (struct ksmbd_file *)argv[1]; - bool claimed; - - spin_lock(&fp->f_lock); - claimed = !list_empty(&in_work->notify_entry); - if (claimed) - list_del_init(&in_work->notify_entry); - spin_unlock(&fp->f_lock); - - if (!claimed) - return NULL; - - /* conn->request_lock is held by smb2_cancel() or connection teardown. */ - in_work->cancel_fn = NULL; - kfree(in_work->cancel_argv); - in_work->cancel_argv = NULL; - return in_work; -} - -static void smb2_complete_notify_cancel(struct ksmbd_work *in_work) -{ - struct smb2_hdr *in_hdr = smb_get_msg(in_work->response_buf); - - in_hdr->Status = STATUS_CANCELLED; - smb2_send_notify_cancelled(in_work); - release_async_work(in_work); - ksmbd_free_work_struct(in_work); -} - -static void smb2_notify_cancel_fn(void **argv) +/* + * Cancel handler for a pending CHANGE_NOTIFY. Called either by + * smb2_cancel() (conn->request_lock held, work->state already set to + * KSMBD_WORK_CANCELLED by the caller) or by + * set_close_state_blocked_works() (vfs_cache.c, fp->f_lock held, + * work->state already set to KSMBD_WORK_CLOSED by the caller) -- both + * callers hold a spinlock across this call, so it must not sleep. + * wake_up() only wakes the waiter in smb2_notify(); it does not touch + * fp->blocked_works itself, matching smb2_remove_blocked_lock()'s same + * non-mutating style for the equivalent byte-range-lock wait. + */ +static void smb2_notify_cancel(void **argv) { - struct ksmbd_work *in_work = smb2_notify_cancel_claim(argv); - struct ksmbd_conn *conn; - struct notify_cancel_ctx *ctx; - - if (!in_work) - return; - conn = in_work->conn; + struct ksmbd_notify_req *notify_req = argv[0]; - ctx = kmalloc(sizeof(*ctx), GFP_ATOMIC); - if (!ctx) { - /* Can't defer the response -- free without sending one. */ - list_del_init(&in_work->async_request_entry); - in_work->asynchronous = false; - if (in_work->async_id) { - ksmbd_release_id(&conn->async_ida, in_work->async_id); - in_work->async_id = 0; - } - ksmbd_free_work_struct(in_work); - return; - } - ctx->in_work = in_work; - INIT_WORK(&ctx->work, smb2_notify_cancel_deferred); - /* - * This deferred work can outlive the connection handler's receive loop. - * Keep teardown from destroying the connection's sessions until the - * deferred response has finished using them. - */ - ksmbd_conn_r_count_inc(conn); - schedule_work(&ctx->work); + wake_up(¬ify_req->wait); } /** @@ -11853,9 +11743,11 @@ int smb2_notify(struct ksmbd_work *work) { struct smb2_change_notify_req *req; struct smb2_change_notify_rsp *rsp; - struct ksmbd_work *in_work; - struct smb2_hdr *in_hdr; - struct ksmbd_file *fp; + struct ksmbd_notify_req notify_req; + struct ksmbd_file *fp = NULL; + void **argv = NULL; + bool async_work = false; + int err = 0; ksmbd_debug(SMB, "Received smb2 notify\n"); @@ -11866,123 +11758,83 @@ int smb2_notify(struct ksmbd_work *work) if (work->next_smb2_rcv_hdr_off && req->hdr.NextCommand) { rsp->hdr.Status = STATUS_INTERNAL_ERROR; - smb2_set_err_rsp(work); - return -EIO; + err = -EIO; + goto out; } - /* - * KSMBD does not implement a real change-notification backend. - * Genuine SMB2 servers (and macOS smbfs) never complete a - * CHANGE_NOTIFY spontaneously: it is satisfied only by a real - * directory change, or with STATUS_NOTIFY_CLEANUP when the watched - * handle is closed. Completing it early (e.g. on a timer) makes - * Finder treat the cleanup as "directory changed" and re-enumerate - * the directory forever, leaving items unopenable. Returning - * STATUS_NOT_IMPLEMENTED here (like stock ksmbd) makes macOS smbfs - * hard-freeze on unmount, so this must stay deferred. - */ fp = ksmbd_lookup_fd_slow(work, req->VolatileFileId, req->PersistentFileId); if (!fp) { rsp->hdr.Status = STATUS_FILE_CLOSED; - smb2_set_err_rsp(work); - return 0; + err = -ENOENT; + goto out; } - in_work = ksmbd_alloc_work_struct(); - if (!in_work || allocate_interim_rsp_buf(in_work)) { - if (in_work) - ksmbd_free_work_struct(in_work); - ksmbd_fd_put(work, fp); - rsp->hdr.Status = STATUS_INSUFFICIENT_RESOURCES; - smb2_set_err_rsp(work); - return 0; - } - /* - * in_work is synthetic (not from the normal request-receiving - * pipeline), so it has no request_buf of its own. It gets registered - * into conn->async_requests below, and smb2_cancel() unconditionally - * computes smb_get_msg(iter->request_buf) for every entry in that - * list while searching for a match -- give it its own small buffer - * (not an alias of response_buf: ksmbd_free_work_struct() kvfree()s - * both separately, so aliasing them would double-free) so that stays - * a harmless read instead of a near-NULL dereference. - */ - in_work->request_buf = kzalloc(MAX_CIFS_SMALL_BUFFER_SIZE, KSMBD_DEFAULT_GFP); - if (!in_work->request_buf) { - ksmbd_free_work_struct(in_work); - ksmbd_fd_put(work, fp); + argv = kmalloc(sizeof(void *), KSMBD_DEFAULT_GFP); + if (!argv) { rsp->hdr.Status = STATUS_INSUFFICIENT_RESOURCES; - smb2_set_err_rsp(work); - return 0; + err = -ENOMEM; + goto out; } - memcpy(smb_get_msg(in_work->request_buf), req, - __SMB2_HEADER_STRUCTURE_SIZE); + init_waitqueue_head(¬ify_req.wait); + argv[0] = ¬ify_req; - if (setup_async_work(work, NULL, NULL)) { - ksmbd_free_work_struct(in_work); - ksmbd_fd_put(work, fp); + err = setup_async_work(work, smb2_notify_cancel, argv); + if (err) { rsp->hdr.Status = STATUS_INSUFFICIENT_RESOURCES; - smb2_set_err_rsp(work); - return 0; + goto out; } - - smb2_send_interim_resp(work, STATUS_PENDING); - - /* Keep the async IDA alive until the deferred work is released. */ - in_work->conn = ksmbd_conn_get(work->conn); - in_work->owns_conn_ref = true; - in_work->encrypted = work->encrypted; - in_hdr = smb_get_msg(in_work->response_buf); - memcpy(in_hdr, ksmbd_resp_buf_next(work), __SMB2_HEADER_STRUCTURE_SIZE); - in_hdr->Flags |= SMB2_FLAGS_ASYNC_COMMAND; - in_hdr->Id.AsyncId = cpu_to_le64(work->async_id); - smb2_set_err_rsp(in_work); - in_hdr->Status = STATUS_NOTIFY_CLEANUP; + async_work = true; /* - * Transfer ownership of the async id to in_work; it stays reserved - * until in_work is freed after the deferred response is sent on - * close, so it can't be reused for an unrelated async response. + * Handle close holds the file-table write lock while it marks the + * handle closed and walks blocked_works. Hold the matching read lock + * across the state check and registration so close cannot finish its + * walk between the lookup above and this list insertion. */ - in_work->async_id = work->async_id; - work->async_id = 0; - release_async_work(work); + read_lock(&work->sess->file_table.lock); + if (fp->f_state != FP_INITED) { + read_unlock(&work->sess->file_table.lock); + rsp->hdr.Status = STATUS_NOTIFY_CLEANUP; + err = -ENOENT; + goto out; + } + spin_lock(&fp->f_lock); + list_add_tail(&work->fp_entry, &fp->blocked_works); + spin_unlock(&fp->f_lock); + read_unlock(&work->sess->file_table.lock); - /* - * work itself is about to be recycled by the normal request-processing - * pipeline, so it can't stay the target of a future CANCEL -- register - * in_work instead, reusing the same async_id, so a client-sent CANCEL - * for this notify actually finds something to cancel instead of - * silently doing nothing until the handle eventually closes. - */ - in_work->asynchronous = true; - in_work->cancel_argv = kmalloc_array(2, sizeof(void *), KSMBD_DEFAULT_GFP); - if (in_work->cancel_argv) { - in_work->cancel_argv[0] = in_work; - in_work->cancel_argv[1] = fp; - in_work->cancel_fn = smb2_notify_cancel_fn; - } - - if (!ksmbd_conn_link_async_request(work->conn, in_work)) { - kfree(in_work->cancel_argv); - in_work->cancel_argv = NULL; - in_work->cancel_fn = NULL; - in_work->asynchronous = false; - ksmbd_fd_put(work, fp); - if (smb2_send_interim_work(in_work, work, false)) - ksmbd_debug(SMB, "failed to send notify cleanup\n"); - ksmbd_free_work_struct(in_work); - work->send_no_response = 1; - return 0; + smb2_send_interim_resp(work, STATUS_PENDING); + + err = wait_event_interruptible(notify_req.wait, + READ_ONCE(work->state) != KSMBD_WORK_ACTIVE); + if (err && READ_ONCE(work->state) == KSMBD_WORK_ACTIVE) { + /* + * Woken by a signal, not a real cancel/close. There is no + * notification backend yet to report anything else against, + * so treat this the same as a client-side cancel. + */ + WRITE_ONCE(work->state, KSMBD_WORK_CANCELLED); } spin_lock(&fp->f_lock); - list_add_tail(&in_work->notify_entry, &fp->notify_pendings); + list_del_init(&work->fp_entry); spin_unlock(&fp->f_lock); - ksmbd_fd_put(work, fp); + rsp->hdr.Status = work->state == KSMBD_WORK_CLOSED ? + STATUS_NOTIFY_CLEANUP : STATUS_CANCELLED; + smb2_send_interim_resp(work, rsp->hdr.Status); work->send_no_response = 1; - return 0; + +out: + if (rsp->hdr.Status != STATUS_SUCCESS && !work->send_no_response) + smb2_set_err_rsp(work); + if (async_work) + release_async_work(work); + else + kfree(argv); + if (fp) + ksmbd_fd_put(work, fp); + return err; } /** @@ -12177,11 +12029,12 @@ void smb3_set_sign_rsp(struct ksmbd_work *work) struct channel *chann; char signature[SMB2_CMACAES_SIZE]; struct kvec *iov; - u16 command = conn->ops->get_cmd_val(work); + u16 command; int n_vec; char *signing_key; hdr = ksmbd_resp_buf_curr(work); + command = le16_to_cpu(hdr->Command); if (command == SMB2_SESSION_SETUP_HE && (!conn->binding || hdr->Status != STATUS_SUCCESS)) { diff --git a/fs/smb/server/vfs_cache.c b/fs/smb/server/vfs_cache.c index fd2c595f048688..a96b764c4db59e 100644 --- a/fs/smb/server/vfs_cache.c +++ b/fs/smb/server/vfs_cache.c @@ -617,7 +617,6 @@ static void __ksmbd_close_fd(struct ksmbd_file_table *ft, struct ksmbd_file *fp) { struct file *filp; struct ksmbd_lock *smb_lock, *tmp_lock; - struct ksmbd_work *cn_work; fd_limit_close(); ksmbd_remove_durable_fd(fp); @@ -652,52 +651,6 @@ static void __ksmbd_close_fd(struct ksmbd_file_table *ft, struct ksmbd_file *fp) kfree(smb_lock); } - /* - * Complete any CHANGE_NOTIFY left pending on this handle now that - * it is closed. KSMBD never completes CHANGE_NOTIFY spontaneously - * (no real change-notification backend), only on close -- matching - * genuine SMB2/macOS smbfs semantics and avoiding the Finder - * "directory changed, re-enumerate everything" loop. - * - * smb2_notify() on another connection can be adding to - * notify_pendings under fp->f_lock at the same time this handle is - * closed, and a client-sent CANCEL can concurrently be racing to - * claim the same entry via smb2_notify_cancel_fn() (smb2pdu.c). - * Pop one entry at a time under the lock via list_del_init() rather - * than a bulk list_splice_init(): list_del_init() leaves the node - * self-linked ("empty"), which is what the cancel path checks under - * the same lock to tell whether it lost the race -- a bulk splice - * would instead relink every entry into a shared local list, so an - * entry claimed here would still read as "not empty" to a racing - * cancel_fn, and both sides could end up freeing the same work. - * ksmbd_conn_write() can sleep (it takes conn's write mutex), so it - * must not be called while fp->f_lock is held -- release the lock - * before processing each popped entry, then reacquire it for the - * next. - */ - for (;;) { - spin_lock(&fp->f_lock); - if (list_empty(&fp->notify_pendings)) { - spin_unlock(&fp->f_lock); - break; - } - cn_work = list_first_entry(&fp->notify_pendings, - struct ksmbd_work, notify_entry); - list_del_init(&cn_work->notify_entry); - spin_unlock(&fp->f_lock); - - ksmbd_conn_write(cn_work); - /* - * release_async_work() removes cn_work from - * conn->async_requests, frees cancel_argv, and releases+zeroes - * async_id -- all needed before ksmbd_free_work_struct(), which - * only releases async_id itself if still nonzero (i.e. if this - * hadn't already been done). - */ - release_async_work(cn_work); - ksmbd_free_work_struct(cn_work); - } - /* * Drop fp's strong reference on conn (taken in ksmbd_open_fd() / * ksmbd_reopen_durable_fd()). Durable fps that reached the @@ -1266,7 +1219,6 @@ struct ksmbd_file *ksmbd_open_fd(struct ksmbd_work *work, struct file *filp) INIT_LIST_HEAD(&fp->blocked_works); INIT_LIST_HEAD(&fp->node); INIT_LIST_HEAD(&fp->lock_list); - INIT_LIST_HEAD(&fp->notify_pendings); spin_lock_init(&fp->f_lock); mutex_init(&fp->readdir_lock); atomic_set(&fp->refcount, 1); diff --git a/fs/smb/server/vfs_cache.h b/fs/smb/server/vfs_cache.h index 1884f6deb9d0e4..732ae26dd6a7db 100644 --- a/fs/smb/server/vfs_cache.h +++ b/fs/smb/server/vfs_cache.h @@ -162,12 +162,6 @@ struct ksmbd_file { unsigned int outstanding_requests; unsigned int outstanding_pre_requests; struct ksmbd_lock_sequence lock_seq[KSMBD_LOCK_SEQ_ARRAY_SIZE]; - - /* - * Pending CHANGE_NOTIFY completions for this handle, sent with - * STATUS_NOTIFY_CLEANUP when the handle is closed. - */ - struct list_head notify_pendings; }; static inline void set_ctx_actor(struct dir_context *ctx, From 8e685c0a546db7c616f68c9467932c45c0da746a Mon Sep 17 00:00:00 2001 From: Namjae Jeon Date: Sun, 23 Aug 2026 18:11:36 +0900 Subject: [PATCH 408/857] ksmbd: doc: update SMB3 multichannel support status SMB3 request replay support is now available. Update the ksmbd documentation to reflect that SMB3 multichannel is supported. Signed-off-by: Namjae Jeon --- Documentation/filesystems/smb/ksmbd.rst | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/Documentation/filesystems/smb/ksmbd.rst b/Documentation/filesystems/smb/ksmbd.rst index 672c5d3892ff91..050af25d5339be 100644 --- a/Documentation/filesystems/smb/ksmbd.rst +++ b/Documentation/filesystems/smb/ksmbd.rst @@ -82,8 +82,7 @@ Signing Update Supported. Pre-authentication integrity Supported. SMB3 encryption(CCM, GCM) Supported. (CCM/GCM128 and CCM/GCM256 supported) SMB direct(RDMA) Supported. -SMB3 Multi-channel Partially Supported. Planned to implement - replay/retry mechanisms for future. +SMB3 Multi-channel Supported. Receive Side Scaling mode Supported. SMB3.1.1 POSIX extension Supported. ACLs Partially Supported. only DACLs available, SACLs From 274f189e497e1588c6beb47d126f725385c5eae3 Mon Sep 17 00:00:00 2001 From: Namjae Jeon Date: Sun, 23 Aug 2026 18:11:37 +0900 Subject: [PATCH 409/857] ksmbd: doc: update RDMA feature support status ksmbd supports SMB3 encryption over RDMA, while signing over RDMA is still under development. Update the feature status table accordingly. Signed-off-by: Namjae Jeon --- Documentation/filesystems/smb/ksmbd.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Documentation/filesystems/smb/ksmbd.rst b/Documentation/filesystems/smb/ksmbd.rst index 050af25d5339be..2425f321f5b7c7 100644 --- a/Documentation/filesystems/smb/ksmbd.rst +++ b/Documentation/filesystems/smb/ksmbd.rst @@ -112,7 +112,8 @@ ksmbd/nfsd interoperability Planned for future. The features that ksmbd support are Leases, Notify, ACLs and Share modes. SMB3.1.1 Compression Supported. SMB3.1.1 over QUIC Planned for future. -Signing/Encryption over RDMA Planned for future. +Signing over RDMA Under development. +Encryption over RDMA Supported. SMB3.1.1 GMAC signing support Planned for future. ============================== ================================================= From 9833f1fce6b6598ed355625b92bf99c9d346a0d9 Mon Sep 17 00:00:00 2001 From: Hang Nan Date: Wed, 19 Aug 2026 11:30:11 +0800 Subject: [PATCH 410/857] ksmbd: add KUnit test for the DACL walk boundary smb_check_perm_dacl() must stop walking ACEs at the DACL declared size instead of using the enclosing security descriptor length. Add the ksmbd KUnit test configuration and a semantic harness that verifies a crafted access-granting ACE beyond the declared DACL size is ignored. Suggested-by: ChenXiaoSong Suggested-by: Namjae Jeon Signed-off-by: Hang Nan Reviewed-by: ChenXiaoSong Signed-off-by: Namjae Jeon --- fs/smb/server/Kconfig | 2 + fs/smb/server/Makefile | 1 + fs/smb/server/tests/Kconfig | 15 +++ fs/smb/server/tests/Makefile | 4 + fs/smb/server/tests/smbacl_kunit.c | 170 +++++++++++++++++++++++++++++ 5 files changed, 192 insertions(+) create mode 100644 fs/smb/server/tests/Kconfig create mode 100644 fs/smb/server/tests/Makefile create mode 100644 fs/smb/server/tests/smbacl_kunit.c diff --git a/fs/smb/server/Kconfig b/fs/smb/server/Kconfig index 221ec9717a8341..b7665e0e49423c 100644 --- a/fs/smb/server/Kconfig +++ b/fs/smb/server/Kconfig @@ -71,3 +71,5 @@ config SMB_SERVER_KERBEROS5 bool "Support for Kerberos 5" depends on SMB_SERVER default y + +source "fs/smb/server/tests/Kconfig" diff --git a/fs/smb/server/Makefile b/fs/smb/server/Makefile index a3e9306055e8bd..9bc87695a53c76 100644 --- a/fs/smb/server/Makefile +++ b/fs/smb/server/Makefile @@ -19,3 +19,4 @@ $(obj)/ksmbd_spnego_negtokentarg.asn1.o: $(obj)/ksmbd_spnego_negtokentarg.asn1.c ksmbd-$(CONFIG_SMB_SERVER_SMBDIRECT) += transport_rdma.o ksmbd-$(CONFIG_PROC_FS) += proc.o +obj-$(CONFIG_SMB_SERVER_KUNIT_TESTS) += tests/ diff --git a/fs/smb/server/tests/Kconfig b/fs/smb/server/tests/Kconfig new file mode 100644 index 00000000000000..ad7a4e94ceaa88 --- /dev/null +++ b/fs/smb/server/tests/Kconfig @@ -0,0 +1,15 @@ +# SPDX-License-Identifier: GPL-2.0-or-later +# Copyright (C) 2026 Hang Nan + +config SMB_SERVER_KUNIT_TESTS + tristate "KUnit tests for SMB3 server helpers" if !KUNIT_ALL_TESTS + depends on SMB_SERVER && SMB_KUNIT_TESTS && TMPFS_XATTR + default SMB_KUNIT_TESTS + help + This builds the KUnit tests for ksmbd server helpers. The tests + exercise internal server functionality and help detect regressions + in server-side behavior. They are intended for kernel developers + and are not suitable for production systems. + + For more information on KUnit and unit tests in the kernel, + please read Documentation/dev-tools/kunit/index.rst. diff --git a/fs/smb/server/tests/Makefile b/fs/smb/server/tests/Makefile new file mode 100644 index 00000000000000..8738ab0b0667ba --- /dev/null +++ b/fs/smb/server/tests/Makefile @@ -0,0 +1,4 @@ +# SPDX-License-Identifier: GPL-2.0-or-later +# Copyright (C) 2026 Hang Nan + +obj-$(CONFIG_SMB_SERVER_KUNIT_TESTS) += smbacl_kunit.o diff --git a/fs/smb/server/tests/smbacl_kunit.c b/fs/smb/server/tests/smbacl_kunit.c new file mode 100644 index 00000000000000..733c2fa92030d8 --- /dev/null +++ b/fs/smb/server/tests/smbacl_kunit.c @@ -0,0 +1,170 @@ +// SPDX-License-Identifier: GPL-2.0-or-later +/* + * KUnit tests for ksmbd security descriptor (DACL) handling. + * + * Copyright (C) 2026 Hang Nan + * + * The tests pin the DACL declared-size boundary in smb_check_perm_dacl(): + * + * - ksmbd_dacl_walk_must_stop_at_declared_size: a pure semantic harness + * that models the ACE walk. Walking to the end of the enclosing + * security descriptor (the pre-fix behaviour) selects an ACE that + * sits beyond struct smb_acl::size; stopping at the declared DACL + * size (the fixed behaviour) rejects it. + */ + +#include +#include + +#include "../smbacl.h" +#include "../smb_common.h" + +struct ksmbd_acl_walk_result { + bool found; + bool allowed; + const struct smb_ace *selected; +}; + +static const struct smb_sid test_nonmatching_sid = { + 1, 5, {0, 0, 0, 0, 0, 5}, + { cpu_to_le32(21), cpu_to_le32(1), cpu_to_le32(2), + cpu_to_le32(3), cpu_to_le32(9999) } +}; + +/* + * S-1-22-1-0: the SID id_to_sid(0, SIDUNIX_USER) resolves to, i.e. what + * smb_check_perm_dacl() looks for when called with uid == 0. + */ +static const struct smb_sid test_owner_sid = { + 1, 2, {0, 0, 0, 0, 0, 22}, + { cpu_to_le32(1), cpu_to_le32(0) } +}; + +static int test_compare_sids(const struct smb_sid *a, const struct smb_sid *b) +{ + int i; + + if (a->revision != b->revision || a->num_subauth != b->num_subauth) + return 1; + for (i = 0; i < NUM_AUTHS; i++) { + if (a->authority[i] != b->authority[i]) + return 1; + } + for (i = 0; i < a->num_subauth; i++) { + if (a->sub_auth[i] != b->sub_auth[i]) + return 1; + } + return 0; +} + +static u16 test_ace_size(const struct smb_sid *sid) +{ + return offsetof(struct smb_ace, sid) + CIFS_SID_BASE_SIZE + + sid->num_subauth * sizeof(__le32); +} + +static u16 fill_test_ace(struct smb_ace *ace, const struct smb_sid *sid, + u32 access_req) +{ + u16 size = test_ace_size(sid); + + ace->type = ACCESS_ALLOWED_ACE_TYPE; + ace->flags = 0; + ace->size = cpu_to_le16(size); + ace->access_req = cpu_to_le32(access_req); + memcpy(&ace->sid, sid, size - offsetof(struct smb_ace, sid)); + return size; +} + +static struct ksmbd_acl_walk_result test_walk_dacl(struct smb_acl *pdacl, + int walk_boundary, + const struct smb_sid *target, + u32 requested) +{ + struct ksmbd_acl_walk_result result = {}; + struct smb_ace *ace; + int aces_size; + int i; + + ace = (struct smb_ace *)((char *)pdacl + sizeof(struct smb_acl)); + aces_size = walk_boundary - sizeof(struct smb_acl); + for (i = 0; i < le16_to_cpu(pdacl->num_aces); i++) { + u16 ace_size; + + if (aces_size < offsetof(struct smb_ace, sid) + CIFS_SID_BASE_SIZE) + break; + ace_size = le16_to_cpu(ace->size); + if (ace_size > aces_size || + ace_size < offsetof(struct smb_ace, sid) + CIFS_SID_BASE_SIZE) + break; + aces_size -= ace_size; + + if (ace->sid.num_subauth > SID_MAX_SUB_AUTHORITIES || + ace_size < offsetof(struct smb_ace, sid) + CIFS_SID_BASE_SIZE + + sizeof(__le32) * ace->sid.num_subauth) + break; + + if (!test_compare_sids(target, &ace->sid)) { + result.found = true; + result.selected = ace; + result.allowed = !(requested & ~le32_to_cpu(ace->access_req)); + return result; + } + + ace = (struct smb_ace *)((char *)ace + ace_size); + } + + return result; +} + +static void ksmbd_dacl_walk_must_stop_at_declared_size(struct kunit *test) +{ + struct ksmbd_acl_walk_result declared, enclosing; + struct smb_acl *acl; + struct smb_ace *ace1, *fake; + u16 ace1_size, fake_size; + u16 pdacl_size; + u16 acl_size; + + acl = kunit_kzalloc(test, 128, GFP_KERNEL); + KUNIT_ASSERT_NOT_NULL(test, acl); + + acl->revision = cpu_to_le16(2); + acl->num_aces = cpu_to_le16(2); + + ace1 = (struct smb_ace *)((char *)acl + sizeof(*acl)); + ace1_size = fill_test_ace(ace1, &test_nonmatching_sid, 0); + fake = (struct smb_ace *)((char *)ace1 + ace1_size); + fake_size = fill_test_ace(fake, &test_owner_sid, FILE_READ_DATA); + + pdacl_size = sizeof(*acl) + ace1_size; + acl_size = pdacl_size + fake_size; + acl->size = cpu_to_le16(pdacl_size); + + declared = test_walk_dacl(acl, pdacl_size, &test_owner_sid, + FILE_READ_DATA); + enclosing = test_walk_dacl(acl, acl_size, &test_owner_sid, + FILE_READ_DATA); + + KUNIT_EXPECT_FALSE(test, declared.found); + KUNIT_EXPECT_FALSE(test, declared.allowed); + + /* Demonstrates that the buggy acl_size boundary selects fake ACE #2. */ + KUNIT_EXPECT_TRUE(test, enclosing.found); + KUNIT_EXPECT_TRUE(test, enclosing.allowed); +} + +static struct kunit_case ksmbd_smbacl_test_cases[] = { + KUNIT_CASE(ksmbd_dacl_walk_must_stop_at_declared_size), + {} +}; + +static struct kunit_suite ksmbd_smbacl_test_suite = { + .name = "ksmbd-smbacl", + .test_cases = ksmbd_smbacl_test_cases, +}; + +kunit_test_suite(ksmbd_smbacl_test_suite); + +MODULE_DESCRIPTION("KUnit tests for ksmbd smbacl helpers"); +MODULE_LICENSE("GPL"); From 99aa5bca61be1b0fc17e82dd3d52a7d0d274092c Mon Sep 17 00:00:00 2001 From: Hang Nan Date: Mon, 24 Aug 2026 12:29:32 +0900 Subject: [PATCH 411/857] ksmbd: test smb_check_perm_dacl() DACL walk boundary Drive smb_check_perm_dacl() through ksmbd's NTACL xattr path with a crafted descriptor whose second ACE is beyond the declared DACL size. Verify that the out-of-boundary ACE is not selected and access remains denied. Suggested-by: ChenXiaoSong Signed-off-by: Hang Nan Reviewed-by: ChenXiaoSong Signed-off-by: Namjae Jeon --- fs/smb/server/smbacl.c | 2 + fs/smb/server/tests/smbacl_kunit.c | 90 ++++++++++++++++++++++++++++++ fs/smb/server/vfs.c | 2 + 3 files changed, 94 insertions(+) diff --git a/fs/smb/server/smbacl.c b/fs/smb/server/smbacl.c index 1fad6ccf3a72ab..7c60520f3b63b0 100644 --- a/fs/smb/server/smbacl.c +++ b/fs/smb/server/smbacl.c @@ -7,6 +7,7 @@ */ #include +#include #include #include #include @@ -1665,6 +1666,7 @@ int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path, kfree(pntsd); return rc; } +EXPORT_SYMBOL_IF_KUNIT(smb_check_perm_dacl); int set_info_sec(struct ksmbd_conn *conn, struct ksmbd_tree_connect *tcon, const struct path *path, struct smb_ntsd *pntsd, int ntsd_len, diff --git a/fs/smb/server/tests/smbacl_kunit.c b/fs/smb/server/tests/smbacl_kunit.c index 733c2fa92030d8..391b1f5d181cd6 100644 --- a/fs/smb/server/tests/smbacl_kunit.c +++ b/fs/smb/server/tests/smbacl_kunit.c @@ -11,13 +11,22 @@ * security descriptor (the pre-fix behaviour) selects an ACE that * sits beyond struct smb_acl::size; stopping at the declared DACL * size (the fixed behaviour) rejects it. + * + * - ksmbd_smb_check_perm_dacl_boundary: drives the real + * smb_check_perm_dacl() with a descriptor stored through ksmbd's own + * NTACL xattr path on a tmpfs file, and asserts that a post-boundary + * ACE is not selected for a regular access check. */ #include +#include +#include +#include #include #include "../smbacl.h" #include "../smb_common.h" +#include "../vfs.h" struct ksmbd_acl_walk_result { bool found; @@ -154,8 +163,88 @@ static void ksmbd_dacl_walk_must_stop_at_declared_size(struct kunit *test) KUNIT_EXPECT_TRUE(test, enclosing.allowed); } +/* + * Build an NTSD whose DACL declares one ACE (pdacl->size) but actually + * contains two: the second ACE sits beyond the declared DACL boundary + * yet inside the enclosing security descriptor. The trailing ACE applies + * to S-1-22-1-0, which smb_check_perm_dacl() looks for when uid is zero. + */ +static struct smb_ntsd *build_boundary_ntsd(struct kunit *test, + const struct smb_sid *first_sid, + u32 first_access, + u32 trailing_access, + int *ntsd_size) +{ + struct smb_ntsd *pntsd; + struct smb_acl *pdacl; + struct smb_ace *ace; + u16 first_size = test_ace_size(first_sid); + u16 trailing_size = test_ace_size(&test_owner_sid); + + *ntsd_size = sizeof(struct smb_ntsd) + sizeof(struct smb_acl) + + first_size + trailing_size; + pntsd = kunit_kzalloc(test, *ntsd_size, GFP_KERNEL); + if (!pntsd) + return NULL; + + pntsd->revision = cpu_to_le16(SD_REVISION); + pntsd->type = cpu_to_le16(DACL_PRESENT); + pntsd->dacloffset = cpu_to_le32(sizeof(struct smb_ntsd)); + + pdacl = (struct smb_acl *)((char *)pntsd + sizeof(struct smb_ntsd)); + pdacl->revision = cpu_to_le16(2); + pdacl->num_aces = cpu_to_le16(2); + pdacl->size = cpu_to_le16(sizeof(struct smb_acl) + first_size); + + ace = (struct smb_ace *)((char *)pdacl + sizeof(struct smb_acl)); + fill_test_ace(ace, first_sid, first_access); + + ace = (struct smb_ace *)((char *)ace + first_size); + fill_test_ace(ace, &test_owner_sid, trailing_access); + + return pntsd; +} + +static void ksmbd_smb_check_perm_dacl_boundary_test(struct kunit *test) +{ + struct file *file; + struct smb_ntsd *pntsd; + __le32 daccess = cpu_to_le32(FILE_READ_DATA); + int ntsd_size, rc; + + pntsd = build_boundary_ntsd(test, &test_nonmatching_sid, 0, + FILE_READ_DATA, &ntsd_size); + KUNIT_ASSERT_NOT_NULL(test, pntsd); + + file = shmem_file_setup("ksmbd-kunit-dacl", 0, + mk_vma_flags(VMA_NORESERVE_BIT)); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, file); + + rc = ksmbd_vfs_set_sd_xattr(NULL, mnt_idmap(file->f_path.mnt), + &file->f_path, pntsd, ntsd_size, + false); + KUNIT_EXPECT_EQ(test, 0, rc); + if (rc) + goto out; + + rc = smb_check_perm_dacl(NULL, &file->f_path, &daccess, + cpu_to_le32(FILE_READ_DATA), 0, false); + + /* + * The post-boundary ACE (ACE #2, beyond pdacl->size) grants + * FILE_READ_DATA to the caller's SID, but it must not be + * selected: the walk stops at the declared DACL size and access + * is denied. Before the fix the walk used the enclosing + * descriptor length, selected ACE #2 and returned 0. + */ + KUNIT_EXPECT_EQ(test, -EACCES, rc); +out: + fput(file); +} + static struct kunit_case ksmbd_smbacl_test_cases[] = { KUNIT_CASE(ksmbd_dacl_walk_must_stop_at_declared_size), + KUNIT_CASE(ksmbd_smb_check_perm_dacl_boundary_test), {} }; @@ -168,3 +257,4 @@ kunit_test_suite(ksmbd_smbacl_test_suite); MODULE_DESCRIPTION("KUnit tests for ksmbd smbacl helpers"); MODULE_LICENSE("GPL"); +MODULE_IMPORT_NS("EXPORTED_FOR_KUNIT_TESTING"); diff --git a/fs/smb/server/vfs.c b/fs/smb/server/vfs.c index c2c9aaa5de1b58..3a6f3139c6f526 100644 --- a/fs/smb/server/vfs.c +++ b/fs/smb/server/vfs.c @@ -5,6 +5,7 @@ */ #include +#include #include #include #include @@ -1671,6 +1672,7 @@ int ksmbd_vfs_set_sd_xattr(struct ksmbd_conn *conn, kfree(def_smb_acl); return rc; } +EXPORT_SYMBOL_IF_KUNIT(ksmbd_vfs_set_sd_xattr); int ksmbd_vfs_get_sd_xattr(struct ksmbd_conn *conn, struct mnt_idmap *idmap, From eec7358b94b2d2b488f7ff34ad5a5a95f7ed5735 Mon Sep 17 00:00:00 2001 From: Hang Nan Date: Wed, 19 Aug 2026 11:30:13 +0800 Subject: [PATCH 412/857] ksmbd: test maximal-access DACL walk boundary Add a maximal-access variant of the smb_check_perm_dacl() boundary test. The in-boundary ACE grants read access, while a trailing ACE beyond the declared DACL size grants write access. Verify that maximal-access calculation includes the in-boundary permission and ignores the trailing permission. Suggested-by: ChenXiaoSong Signed-off-by: Hang Nan Reviewed-by: ChenXiaoSong Signed-off-by: Namjae Jeon --- fs/smb/server/tests/smbacl_kunit.c | 47 ++++++++++++++++++++++++++++-- 1 file changed, 44 insertions(+), 3 deletions(-) diff --git a/fs/smb/server/tests/smbacl_kunit.c b/fs/smb/server/tests/smbacl_kunit.c index 391b1f5d181cd6..33496b4d31a3ec 100644 --- a/fs/smb/server/tests/smbacl_kunit.c +++ b/fs/smb/server/tests/smbacl_kunit.c @@ -12,10 +12,11 @@ * sits beyond struct smb_acl::size; stopping at the declared DACL * size (the fixed behaviour) rejects it. * - * - ksmbd_smb_check_perm_dacl_boundary: drives the real + * - ksmbd_smb_check_perm_dacl_boundary and + * ksmbd_smb_check_perm_dacl_maximal_boundary: drive the real * smb_check_perm_dacl() with a descriptor stored through ksmbd's own - * NTACL xattr path on a tmpfs file, and asserts that a post-boundary - * ACE is not selected for a regular access check. + * NTACL xattr path on a tmpfs file, and assert that a post-boundary + * ACE is not selected for either a regular or maximal access check. */ #include @@ -242,9 +243,49 @@ static void ksmbd_smb_check_perm_dacl_boundary_test(struct kunit *test) fput(file); } +static void +ksmbd_smb_check_perm_dacl_maximal_boundary_test(struct kunit *test) +{ + struct file *file; + struct smb_ntsd *pntsd; + __le32 daccess = FILE_MAXIMAL_ACCESS_LE; + int ntsd_size, rc; + + /* + * The in-boundary ACE grants read access. The trailing ACE grants + * write access, which must not be included in the maximal access mask. + */ + pntsd = build_boundary_ntsd(test, &test_owner_sid, FILE_READ_DATA, + FILE_WRITE_DATA, &ntsd_size); + KUNIT_ASSERT_NOT_NULL(test, pntsd); + + file = shmem_file_setup("ksmbd-kunit-dacl-maximal", 0, + mk_vma_flags(VMA_NORESERVE_BIT)); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, file); + + rc = ksmbd_vfs_set_sd_xattr(NULL, mnt_idmap(file->f_path.mnt), + &file->f_path, pntsd, ntsd_size, + false); + KUNIT_EXPECT_EQ(test, 0, rc); + if (rc) + goto out; + + rc = smb_check_perm_dacl(NULL, &file->f_path, &daccess, + FILE_MAXIMAL_ACCESS_LE, 0, false); + KUNIT_EXPECT_EQ(test, 0, rc); + if (rc) + goto out; + + KUNIT_EXPECT_TRUE(test, le32_to_cpu(daccess) & FILE_READ_DATA); + KUNIT_EXPECT_FALSE(test, le32_to_cpu(daccess) & FILE_WRITE_DATA); +out: + fput(file); +} + static struct kunit_case ksmbd_smbacl_test_cases[] = { KUNIT_CASE(ksmbd_dacl_walk_must_stop_at_declared_size), KUNIT_CASE(ksmbd_smb_check_perm_dacl_boundary_test), + KUNIT_CASE(ksmbd_smb_check_perm_dacl_maximal_boundary_test), {} }; From da6066cf54a9c3f54b8f1d75a9352a6588326a17 Mon Sep 17 00:00:00 2001 From: Namjae Jeon Date: Fri, 28 Aug 2026 16:04:45 +0900 Subject: [PATCH 413/857] ksmbd: doc: update feature status Add MSDFS, Continuous Availability (CA), and AD/DC to the ksmbd feature status list, along with resilient handle. Mark Continuous Availability, AD/DC, persistent handle, and SMB2 notify as under development. Signed-off-by: Namjae Jeon --- Documentation/filesystems/smb/ksmbd.rst | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/Documentation/filesystems/smb/ksmbd.rst b/Documentation/filesystems/smb/ksmbd.rst index 2425f321f5b7c7..728cfca1320561 100644 --- a/Documentation/filesystems/smb/ksmbd.rst +++ b/Documentation/filesystems/smb/ksmbd.rst @@ -85,6 +85,9 @@ SMB direct(RDMA) Supported. SMB3 Multi-channel Supported. Receive Side Scaling mode Supported. SMB3.1.1 POSIX extension Supported. +MSDFS Planned for future. +Continuous Availability (CA) Under development. +AD/DC Under development. ACLs Partially Supported. only DACLs available, SACLs (auditing) is planned for the future. For ownership (SIDs) ksmbd generates random subauth @@ -97,8 +100,9 @@ ACLs Partially Supported. only DACLs available, SACLs member. Kerberos Supported. Durable handle v1,v2 Supported. -Persistent handle Planned for future. -SMB2 notify Planned for future. +Persistent handle Under development. +Resilient handle Planned for future. +SMB2 notify Under development. Sparse file support Supported. DCE/RPC support Partially Supported. a few calls(NetShareEnumAll, NetServerGetInfo, SAMR, LSARPC) that are needed From e1ab109df034ee83166a53ec86cd448559214014 Mon Sep 17 00:00:00 2001 From: Liu Zhenlong Date: Wed, 19 Aug 2026 00:35:13 +0800 Subject: [PATCH 414/857] RDMA/rtrs-clt: use find_next_zero_bit() for permit allocation __rtrs_get_permit() scans permits_map with find_first_zero_bit() and claims the bit with test_and_set_bit_lock(), restarting from bit 0 on a lost race. Use find_next_zero_bit() to resume from the last position so a lost race does not rescan the already-set low bits; on reaching the end, wrap to the beginning to exhaust the map. Compile-tested: arm64 defconfig + INFINIBAND_RTRS_CLIENT=m, rtrs-clt.o Reviewed-by: Jack Wang Signed-off-by: Liu Zhenlong Link: https://patch.msgid.link/20260818163513.53875-1-dragonliu2018@gmail.com Signed-off-by: Leon Romanovsky --- drivers/infiniband/ulp/rtrs/rtrs-clt.c | 23 ++++++++++++++--------- 1 file changed, 14 insertions(+), 9 deletions(-) diff --git a/drivers/infiniband/ulp/rtrs/rtrs-clt.c b/drivers/infiniband/ulp/rtrs/rtrs-clt.c index 7b2c51ae614f81..f48872bb002442 100644 --- a/drivers/infiniband/ulp/rtrs/rtrs-clt.c +++ b/drivers/infiniband/ulp/rtrs/rtrs-clt.c @@ -70,19 +70,24 @@ __rtrs_get_permit(struct rtrs_clt_sess *clt, enum rtrs_clt_con_type con_type) { size_t max_depth = clt->queue_depth; struct rtrs_permit *permit; - int bit; + unsigned long bit = 0; /* - * Adapted from null_blk get_tag(). Callers from different cpus may - * grab the same bit, since find_first_zero_bit is not atomic. - * But then the test_and_set_bit_lock will fail for all the - * callers but one, so that they will loop again. - * This way an explicit spinlock is not required. + * Callers from different CPUs may grab the same bit, since the bitmap + * scan is not atomic. But then the test_and_set_bit_lock() will fail + * for all the callers but one, so that they loop again. This way an + * explicit spinlock is not required. find_next_zero_bit() resumes + * from the last position so that a lost race does not rescan the + * already-set low bits; if it reaches the end, wrap to the beginning + * to exhaust the map and still find a permit freed below the cursor. */ do { - bit = find_first_zero_bit(clt->permits_map, max_depth); - if (bit >= max_depth) - return NULL; + bit = find_next_zero_bit(clt->permits_map, max_depth, bit); + if (bit >= max_depth) { + bit = find_first_zero_bit(clt->permits_map, max_depth); + if (bit >= max_depth) + return NULL; + } } while (test_and_set_bit_lock(bit, clt->permits_map)); permit = get_permit(clt, bit); From 746933bc47ca2b966d5a86cbdb617789a0bb439b Mon Sep 17 00:00:00 2001 From: Biju Das Date: Thu, 4 Jun 2026 16:18:50 +0100 Subject: [PATCH 415/857] arm64: dts: renesas: r9a08g046: Add Mali-G31 GPU node Add the Mali-G31 GPU node to the SoC DTSI. Signed-off-by: Biju Das Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260604151855.307772-3-biju.das.jz@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r9a08g046.dtsi | 126 +++++++++++++++++++++ 1 file changed, 126 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/r9a08g046.dtsi b/arch/arm64/boot/dts/renesas/r9a08g046.dtsi index c63a857f0e5b01..608ba1f9e2f8c2 100644 --- a/arch/arm64/boot/dts/renesas/r9a08g046.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a08g046.dtsi @@ -64,6 +64,110 @@ }; }; + gpu_opp_table: opp-table-1 { + compatible = "operating-points-v2"; + + opp-600000000 { + opp-hz = /bits/ 64 <600000000>; + opp-microvolt = <1000000>; + }; + + opp-533330000 { + opp-hz = /bits/ 64 <533330000>; + opp-microvolt = <1000000>; + }; + + opp-500000000 { + opp-hz = /bits/ 64 <500000000>; + opp-microvolt = <1000000>; + }; + + opp-400000000 { + opp-hz = /bits/ 64 <400000000>; + opp-microvolt = <1000000>; + }; + + opp-300000000 { + opp-hz = /bits/ 64 <300000000>; + opp-microvolt = <1000000>; + }; + + opp-266667000 { + opp-hz = /bits/ 64 <266667000>; + opp-microvolt = <1000000>; + }; + + opp-250000000 { + opp-hz = /bits/ 64 <250000000>; + opp-microvolt = <1000000>; + }; + + opp-200000000 { + opp-hz = /bits/ 64 <200000000>; + opp-microvolt = <1000000>; + }; + + opp-150000000 { + opp-hz = /bits/ 64 <150000000>; + opp-microvolt = <1000000>; + }; + + opp-133333000 { + opp-hz = /bits/ 64 <133333000>; + opp-microvolt = <1000000>; + }; + + opp-125000000 { + opp-hz = /bits/ 64 <125000000>; + opp-microvolt = <1000000>; + }; + + opp-100000000 { + opp-hz = /bits/ 64 <100000000>; + opp-microvolt = <1000000>; + }; + + opp-75000000 { + opp-hz = /bits/ 64 <75000000>; + opp-microvolt = <1000000>; + }; + + opp-66667000 { + opp-hz = /bits/ 64 <66667000>; + opp-microvolt = <1000000>; + }; + + opp-62500000 { + opp-hz = /bits/ 64 <62500000>; + opp-microvolt = <1000000>; + }; + + opp-50000000 { + opp-hz = /bits/ 64 <50000000>; + opp-microvolt = <1000000>; + }; + + opp-18750000 { + opp-hz = /bits/ 64 <18750000>; + opp-microvolt = <1000000>; + }; + + opp-16667000 { + opp-hz = /bits/ 64 <16667000>; + opp-microvolt = <1000000>; + }; + + opp-15625000 { + opp-hz = /bits/ 64 <15625000>; + opp-microvolt = <1000000>; + }; + + opp-12500000 { + opp-hz = /bits/ 64 <12500000>; + opp-microvolt = <1000000>; + }; + }; + cpus { #address-cells = <1>; #size-cells = <0>; @@ -592,6 +696,28 @@ status = "disabled"; }; + gpu: gpu@108b0000 { + compatible = "renesas,r9a08g046-mali", + "arm,mali-bifrost"; + reg = <0x0 0x108b0000 0x0 0x10000>; + interrupts = , + , + , + ; + interrupt-names = "job", "mmu", "gpu", "event"; + clocks = <&cpg CPG_MOD R9A08G046_GE3D_CLK>, + <&cpg CPG_MOD R9A08G046_GE3D_AXI_CLK>, + <&cpg CPG_MOD R9A08G046_GE3D_ACE_CLK>; + clock-names = "gpu", "bus", "bus_ace"; + power-domains = <&cpg>; + resets = <&cpg R9A08G046_GE3D_RESETN>, + <&cpg R9A08G046_GE3D_AXI_RESETN>, + <&cpg R9A08G046_GE3D_ACE_RESETN>; + reset-names = "rst", "axi_rst", "ace_rst"; + operating-points-v2 = <&gpu_opp_table>; + status = "disabled"; + }; + cpg: clock-controller@11010000 { compatible = "renesas,r9a08g046-cpg"; reg = <0 0x11010000 0 0x10000>; From d09ffb8db2c72fc6bde483afd735d5565adaa283 Mon Sep 17 00:00:00 2001 From: Biju Das Date: Thu, 4 Jun 2026 16:18:51 +0100 Subject: [PATCH 416/857] arm64: dts: renesas: rzg3l-smarc-som: Enable Mali-G31 Enable the Mali-G31 (GPU) node on the RZ/G3L SMARC SoM board. Signed-off-by: Biju Das Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260604151855.307772-4-biju.das.jz@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/rzg3l-smarc-som.dtsi | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/rzg3l-smarc-som.dtsi b/arch/arm64/boot/dts/renesas/rzg3l-smarc-som.dtsi index 091a227233cbaa..98d9cf323cb884 100644 --- a/arch/arm64/boot/dts/renesas/rzg3l-smarc-som.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg3l-smarc-som.dtsi @@ -45,6 +45,15 @@ reg = <0x0 0x48000000 0x0 0x78000000>; }; + reg_1p0v: regulator-1p0v { + compatible = "regulator-fixed"; + regulator-name = "fixed-1.0V"; + regulator-min-microvolt = <1000000>; + regulator-max-microvolt = <1000000>; + regulator-boot-on; + regulator-always-on; + }; + reg_1p8v: regulator-1p8v { compatible = "regulator-fixed"; regulator-name = "fixed-1.8V"; @@ -100,6 +109,11 @@ clock-frequency = <24000000>; }; +&gpu { + status = "okay"; + mali-supply = <®_1p0v>; +}; + &i2c0 { pinctrl-0 = <&i2c0_pins>; pinctrl-names = "default"; From 59cb3ce5c6eaa57456b9a40ff2e0b3919d0f9f28 Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Mon, 15 Jun 2026 12:54:51 +0100 Subject: [PATCH 417/857] arm64: dts: renesas: r9a09g077: Add VSPD and FCPVD nodes Add VSPD and FCPVD nodes to RZ/T2H SoC DTSI. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260615115455.1412098-2-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r9a09g077.dtsi | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi index bac39390ead74d..4f4fffaffed4c0 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi @@ -1403,6 +1403,28 @@ status = "disabled"; }; }; + + fcpvd: fcp@920d0000 { + compatible = "renesas,r9a09g077-fcpvd", "renesas,fcpv"; + reg = <0 0x920d0000 0 0x10000>; + clocks = <&cpg CPG_CORE R9A09G077_CLK_PCLKAH>, + <&cpg CPG_MOD 1204>, + <&cpg CPG_CORE R9A09G077_LCDC_CLKD>; + clock-names = "aclk", "pclk", "vclk"; + power-domains = <&cpg>; + }; + + vspd: vsp@920e0000 { + compatible = "renesas,r9a09g077-vsp2", "renesas,r9a07g044-vsp2"; + reg = <0 0x920e0000 0 0x8000>; + interrupts = ; + clocks = <&cpg CPG_CORE R9A09G077_CLK_PCLKAH>, + <&cpg CPG_MOD 1204>, + <&cpg CPG_CORE R9A09G077_LCDC_CLKD>; + clock-names = "aclk", "pclk", "vclk"; + power-domains = <&cpg>; + renesas,fcp = <&fcpvd>; + }; }; stmmac_axi_setup: stmmac-axi-config { From 8e67675a64776fb947e4701933c9ac812488e289 Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Mon, 15 Jun 2026 12:54:52 +0100 Subject: [PATCH 418/857] arm64: dts: renesas: r9a09g077: Add DU node Add Display Unit (DU) node to SoC DTSI. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260615115455.1412098-3-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r9a09g077.dtsi | 24 ++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi index 4f4fffaffed4c0..73cfb62952c4b6 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi @@ -1404,6 +1404,30 @@ }; }; + du: display@920c0000 { + compatible = "renesas,r9a09g077-du"; + reg = <0 0x920c0000 0 0x10000>; + interrupts = ; + clocks = <&cpg CPG_CORE R9A09G077_CLK_PCLKAH>, + <&cpg CPG_MOD 1204>, + <&cpg CPG_CORE R9A09G077_LCDC_CLKD>; + clock-names = "aclk", "pclk", "vclk"; + power-domains = <&cpg>; + renesas,vsps = <&vspd 0>; + status = "disabled"; + + ports { + #address-cells = <1>; + #size-cells = <0>; + + port@0 { + reg = <0>; + du_out_rgb: endpoint { + }; + }; + }; + }; + fcpvd: fcp@920d0000 { compatible = "renesas,r9a09g077-fcpvd", "renesas,fcpv"; reg = <0 0x920d0000 0 0x10000>; From 15267ff79d79b356802fbdcd160c631c20468782 Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Mon, 15 Jun 2026 12:54:53 +0100 Subject: [PATCH 419/857] arm64: dts: renesas: r9a09g087: Add VSPD and FCPVD nodes Add VSPD and FCPVD nodes to RZ/N2H SoC DTSI. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260615115455.1412098-4-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r9a09g087.dtsi | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi index 03b976d93e1058..b673ea129fc374 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi @@ -1406,6 +1406,28 @@ status = "disabled"; }; }; + + fcpvd: fcp@920d0000 { + compatible = "renesas,r9a09g087-fcpvd", "renesas,fcpv"; + reg = <0 0x920d0000 0 0x10000>; + clocks = <&cpg CPG_CORE R9A09G087_CLK_PCLKAH>, + <&cpg CPG_MOD 1204>, + <&cpg CPG_CORE R9A09G087_LCDC_CLKD>; + clock-names = "aclk", "pclk", "vclk"; + power-domains = <&cpg>; + }; + + vspd: vsp@920e0000 { + compatible = "renesas,r9a09g087-vsp2", "renesas,r9a07g044-vsp2"; + reg = <0 0x920e0000 0 0x8000>; + interrupts = ; + clocks = <&cpg CPG_CORE R9A09G087_CLK_PCLKAH>, + <&cpg CPG_MOD 1204>, + <&cpg CPG_CORE R9A09G087_LCDC_CLKD>; + clock-names = "aclk", "pclk", "vclk"; + power-domains = <&cpg>; + renesas,fcp = <&fcpvd>; + }; }; stmmac_axi_setup: stmmac-axi-config { From f212f078072017e32776dbab3f792b36c339b1f1 Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Mon, 15 Jun 2026 12:54:54 +0100 Subject: [PATCH 420/857] arm64: dts: renesas: r9a09g087: Add DU node Add Display Unit (DU) node to SoC DTSI. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260615115455.1412098-5-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r9a09g087.dtsi | 24 ++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi index b673ea129fc374..2be6d39a37ad25 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi @@ -1407,6 +1407,30 @@ }; }; + du: display@920c0000 { + compatible = "renesas,r9a09g087-du", "renesas,r9a09g077-du"; + reg = <0 0x920c0000 0 0x10000>; + interrupts = ; + clocks = <&cpg CPG_CORE R9A09G087_CLK_PCLKAH>, + <&cpg CPG_MOD 1204>, + <&cpg CPG_CORE R9A09G087_LCDC_CLKD>; + clock-names = "aclk", "pclk", "vclk"; + power-domains = <&cpg>; + renesas,vsps = <&vspd 0>; + status = "disabled"; + + ports { + #address-cells = <1>; + #size-cells = <0>; + + port@0 { + reg = <0>; + du_out_rgb: endpoint { + }; + }; + }; + }; + fcpvd: fcp@920d0000 { compatible = "renesas,r9a09g087-fcpvd", "renesas,fcpv"; reg = <0 0x920d0000 0 0x10000>; From e8a135f03b4371c9f1314490364e99034e5d9932 Mon Sep 17 00:00:00 2001 From: Biju Das Date: Fri, 19 Jun 2026 08:56:19 +0100 Subject: [PATCH 421/857] arm64: dts: renesas: r9a08g045: Move max-frequency to SoC dtsi Move the max-frequency property for SDHI0/1/2 from the board-level SMARC dtsi files into the r9a08g045.dtsi SoC file, since these values reflect controller/SoC limitations rather than board-specific. This removes the duplicated max-frequency = <125000000> entries for SDHI0 (both SD and eMMC variants) and SDHI1 in rzg3s-smarc-som.dtsi and rzg3s-smarc.dtsi, and the max-frequency = <50000000> entry for SDHI2, consolidating them as defaults in r9a08g045.dtsi instead. Boards needing a different limit can still override max-frequency locally. Signed-off-by: Biju Das Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260619075621.126961-1-biju.das.jz@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r9a08g045.dtsi | 3 +++ arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi | 3 --- arch/arm64/boot/dts/renesas/rzg3s-smarc.dtsi | 1 - 3 files changed, 3 insertions(+), 4 deletions(-) diff --git a/arch/arm64/boot/dts/renesas/r9a08g045.dtsi b/arch/arm64/boot/dts/renesas/r9a08g045.dtsi index 3a69bb246babd0..ae92d45ede387b 100644 --- a/arch/arm64/boot/dts/renesas/r9a08g045.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a08g045.dtsi @@ -655,6 +655,7 @@ <&cpg CPG_MOD R9A08G045_SDHI0_IMCLK2>, <&cpg CPG_MOD R9A08G045_SDHI0_ACLK>; clock-names = "core", "clkh", "cd", "aclk"; + max-frequency = <125000000>; resets = <&cpg R9A08G045_SDHI0_IXRST>; power-domains = <&cpg>; status = "disabled"; @@ -670,6 +671,7 @@ <&cpg CPG_MOD R9A08G045_SDHI1_IMCLK2>, <&cpg CPG_MOD R9A08G045_SDHI1_ACLK>; clock-names = "core", "clkh", "cd", "aclk"; + max-frequency = <125000000>; resets = <&cpg R9A08G045_SDHI1_IXRST>; power-domains = <&cpg>; status = "disabled"; @@ -685,6 +687,7 @@ <&cpg CPG_MOD R9A08G045_SDHI2_IMCLK2>, <&cpg CPG_MOD R9A08G045_SDHI2_ACLK>; clock-names = "core", "clkh", "cd", "aclk"; + max-frequency = <50000000>; resets = <&cpg R9A08G045_SDHI2_IXRST>; power-domains = <&cpg>; status = "disabled"; diff --git a/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi b/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi index b45acfe6288a7c..9039a927bc46e3 100644 --- a/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi @@ -184,7 +184,6 @@ bus-width = <4>; sd-uhs-sdr50; sd-uhs-sdr104; - max-frequency = <125000000>; status = "okay"; }; #else @@ -199,7 +198,6 @@ mmc-hs200-1_8v; non-removable; fixed-emmc-driver-type = <1>; - max-frequency = <125000000>; status = "okay"; }; #endif @@ -210,7 +208,6 @@ pinctrl-names = "default"; vmmc-supply = <&vcc_sdhi2>; bus-width = <4>; - max-frequency = <50000000>; status = "okay"; }; #endif diff --git a/arch/arm64/boot/dts/renesas/rzg3s-smarc.dtsi b/arch/arm64/boot/dts/renesas/rzg3s-smarc.dtsi index 70af605168b07c..e3821d8c01e30a 100644 --- a/arch/arm64/boot/dts/renesas/rzg3s-smarc.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg3s-smarc.dtsi @@ -285,7 +285,6 @@ bus-width = <4>; sd-uhs-sdr50; sd-uhs-sdr104; - max-frequency = <125000000>; status = "okay"; }; From 789cf2dd39cb45eca6cd2c9d8603f23d501c39eb Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Sun, 5 Jul 2026 14:52:46 +0200 Subject: [PATCH 422/857] arm64: dts: renesas: sparrow-hawk: Always enable edge connector I2C busses The interfaces on the edge connector of Retronix R-Car V4H Sparrow Hawk board may be controlled from userspace using matching userspace tooling that includes e.g. i2c-tools. Enable the edge connector I2C busses I2C3 and I2C4 to allow userspace applications to use those busses and access peripherals attached to those busses. Co-developed-by: Yuya Hamamachi Signed-off-by: Yuya Hamamachi Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260705125324.13519-1-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- .../boot/dts/renesas/r8a779g3-sparrow-hawk-fan-argon40.dtso | 1 - arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk.dts | 2 ++ 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk-fan-argon40.dtso b/arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk-fan-argon40.dtso index c730ef39c7d7d1..6f10310140b9b2 100644 --- a/arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk-fan-argon40.dtso +++ b/arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk-fan-argon40.dtso @@ -41,7 +41,6 @@ #address-cells = <1>; #size-cells = <0>; clock-frequency = <400000>; - status = "okay"; pwmhat: pwm@1a { compatible = "argon40,fan-hat"; diff --git a/arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk.dts b/arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk.dts index af680290ce8170..a6294fd32daeed 100644 --- a/arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk.dts +++ b/arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk.dts @@ -498,6 +498,7 @@ #size-cells = <0>; pinctrl-0 = <&i2c3_pins>; pinctrl-names = "default"; + status = "okay"; }; /* Page 31 / IO_CN */ @@ -506,6 +507,7 @@ #size-cells = <0>; pinctrl-0 = <&i2c4_pins>; pinctrl-names = "default"; + status = "okay"; }; /* Page 18 / POWER_CORE and Page 19 / POWER_PMIC */ From 3634855070fcb9ba3d8885b6980f0bddb2f1103d Mon Sep 17 00:00:00 2001 From: Claudiu Beznea Date: Fri, 10 Jul 2026 14:36:37 +0300 Subject: [PATCH 423/857] arm64: dts: renesas: rzg3s-smarc-som: Enable I3C The Renesas RZ/G3S SMARC SoM board has a connector for I3C interface. Enable I3C. Reviewed-by: Wolfram Sang Tested-by: Wolfram Sang Signed-off-by: Claudiu Beznea Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260710113637.1328000-6-claudiu.beznea+renesas@tuxon.dev Signed-off-by: Geert Uytterhoeven --- .../boot/dts/renesas/rzg3s-smarc-som.dtsi | 18 ++++++++++++++++++ .../boot/dts/renesas/rzg3s-smarc-switches.h | 4 ++++ 2 files changed, 22 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi b/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi index 9039a927bc46e3..ded6066c91765b 100644 --- a/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi @@ -168,6 +168,14 @@ }; }; +&i3c { + pinctrl-names = "default"; + pinctrl-0 = <&i3c_pins>; + i2c-scl-hz = <400000>; + i3c-scl-hz = <12500000>; + status = "okay"; +}; + &pcie_port0 { clocks = <&versa3 5>; clock-names = "ref"; @@ -299,6 +307,16 @@ }; }; + i3c_pins: i3c { + pins = "I3C_SDA", "I3C_SCL"; +#if SW_CONFIG4 == SW_ON + power-source = <1200>; +#else + power-source = <1800>; +#endif + input-enable; + }; + sdhi0_pins: sd0 { data { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3"; diff --git a/arch/arm64/boot/dts/renesas/rzg3s-smarc-switches.h b/arch/arm64/boot/dts/renesas/rzg3s-smarc-switches.h index bbf908a5322c91..9cccc87da05750 100644 --- a/arch/arm64/boot/dts/renesas/rzg3s-smarc-switches.h +++ b/arch/arm64/boot/dts/renesas/rzg3s-smarc-switches.h @@ -25,9 +25,13 @@ * @SW_CONFIG3: * SW_OFF - SD2 is connected to SoC * SW_ON - SCIF1, SSI0, IRQ0, IRQ1 connected to SoC + * @SW_CONFIG4: + * SW_OFF - I3C voltage is 1.8V + * SW_ON - I3C voltage is 1.2V */ #define SW_CONFIG2 SW_OFF #define SW_CONFIG3 SW_ON +#define SW_CONFIG4 SW_OFF /* * SW_OPT_MUX[x] switches' states: From a169807a7ddbf226d3eda5c6fe64f5fa11457ffa Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:03:01 +0200 Subject: [PATCH 424/857] arm: dts: renesas: gr-peach: Specify ethernet PHY reset timings The LAN8710Ai reference manual [1] DS00002164C page 61 FIGURE 5-3: POWER-ON NRST & CONFIGURATION STRAP TIMING does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for 300 us which should cover every timing option, including the 16 us delay required for MDIO to switch to 25 MHz listed in Note: in Chapter 3.8.5 RESETS. The LAN8710Ai reference manual [1] DS00002164C page 61 TABLE 5-8: POWER-ON NRST & CONFIGURATION STRAP TIMING VALUES row tSR Stable supply voltages to reset high is at minimum 10 ms. Set DT property reset-assert-us to 25ms because the LAN8710Ai RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/LAN8710A-LAN8710Ai-Data-Sheet-DS00002164.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130325.11080-1-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r7s72100-gr-peach.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r7s72100-gr-peach.dts b/arch/arm/boot/dts/renesas/r7s72100-gr-peach.dts index 23ddec21768574..2477db9e5aacac 100644 --- a/arch/arm/boot/dts/renesas/r7s72100-gr-peach.dts +++ b/arch/arm/boot/dts/renesas/r7s72100-gr-peach.dts @@ -131,5 +131,7 @@ reset-gpios = <&port4 2 GPIO_ACTIVE_LOW>; reset-delay-us = <5>; + reset-assert-us = <25000>; + reset-deassert-us = <300>; }; }; From a86329bc44839cb5c7567bb264a7752b3ac1e885 Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:03:02 +0200 Subject: [PATCH 425/857] arm: dts: renesas: armadillo800eva: Specify ethernet PHY reset timings The LAN8710Ai reference manual [1] DS00002164C page 61 FIGURE 5-3: POWER-ON NRST & CONFIGURATION STRAP TIMING does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for 300 us which should cover every timing option, including the 16 us delay required for MDIO to switch to 25 MHz listed in Note: in Chapter 3.8.5 RESETS. The LAN8710Ai reference manual [1] DS00002164C page 61 TABLE 5-8: POWER-ON NRST & CONFIGURATION STRAP TIMING VALUES row tSR Stable supply voltages to reset high is at minimum 10 ms. Set DT property reset-assert-us to 25ms because the LAN8710Ai RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/LAN8710A-LAN8710Ai-Data-Sheet-DS00002164.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130325.11080-2-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7740-armadillo800eva.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7740-armadillo800eva.dts b/arch/arm/boot/dts/renesas/r8a7740-armadillo800eva.dts index 1d56bdef545398..eb65a54d0e5111 100644 --- a/arch/arm/boot/dts/renesas/r8a7740-armadillo800eva.dts +++ b/arch/arm/boot/dts/renesas/r8a7740-armadillo800eva.dts @@ -197,6 +197,8 @@ "ethernet-phy-ieee802.3-c22"; reg = <0>; reset-gpios = <&pfc 18 GPIO_ACTIVE_LOW>; + reset-assert-us = <25000>; + reset-deassert-us = <300>; }; }; From 2eb10f6874c24e599fad5b45e18f9717750d4a0a Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:03:54 +0200 Subject: [PATCH 426/857] arm: dts: renesas: lager: Specify ethernet PHY reset timings The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS row tSR Stable supply voltages to reset high is at minimum 10 ms. Set the DT property reset-assert-us to 10ms because the KSZ8041RNL RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/KSZ8041NL-RNL-Data-Sheet-DS00002245.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130429.11214-1-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7790-lager.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7790-lager.dts b/arch/arm/boot/dts/renesas/r8a7790-lager.dts index 8e766550167557..86548859e803d4 100644 --- a/arch/arm/boot/dts/renesas/r8a7790-lager.dts +++ b/arch/arm/boot/dts/renesas/r8a7790-lager.dts @@ -690,6 +690,8 @@ interrupts-extended = <&irqc0 0 IRQ_TYPE_LEVEL_LOW>; micrel,led-mode = <1>; reset-gpios = <&gpio5 31 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From b1d43f93035213ce22cbc1d14f23e4e8831c0fb0 Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:03:55 +0200 Subject: [PATCH 427/857] arm: dts: renesas: stout: Specify ethernet PHY reset timings The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS row tSR Stable supply voltages to reset high is at minimum 10 ms. Set the DT property reset-assert-us to 10ms because the KSZ8041RNL RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/KSZ8041NL-RNL-Data-Sheet-DS00002245.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130429.11214-2-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7790-stout.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7790-stout.dts b/arch/arm/boot/dts/renesas/r8a7790-stout.dts index 8ba9d85f103896..b062423499e480 100644 --- a/arch/arm/boot/dts/renesas/r8a7790-stout.dts +++ b/arch/arm/boot/dts/renesas/r8a7790-stout.dts @@ -213,6 +213,8 @@ interrupts-extended = <&irqc0 1 IRQ_TYPE_LEVEL_LOW>; micrel,led-mode = <1>; reset-gpios = <&gpio3 31 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From 89bd4b855cd8779b44bdb0a5a3bbe88003a25592 Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:03:56 +0200 Subject: [PATCH 428/857] arm: dts: renesas: koelsch: Specify ethernet PHY reset timings The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS row tSR Stable supply voltages to reset high is at minimum 10 ms. Set the DT property reset-assert-us to 10ms because the KSZ8041RNL RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/KSZ8041NL-RNL-Data-Sheet-DS00002245.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130429.11214-3-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7791-koelsch.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7791-koelsch.dts b/arch/arm/boot/dts/renesas/r8a7791-koelsch.dts index 48db62e0ff8748..54148a5ddff7b9 100644 --- a/arch/arm/boot/dts/renesas/r8a7791-koelsch.dts +++ b/arch/arm/boot/dts/renesas/r8a7791-koelsch.dts @@ -681,6 +681,8 @@ interrupts-extended = <&irqc0 0 IRQ_TYPE_LEVEL_LOW>; micrel,led-mode = <1>; reset-gpios = <&gpio5 22 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From b82db6b9da9cf470ea145abc792853f3040077ac Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:03:57 +0200 Subject: [PATCH 429/857] arm: dts: renesas: porter: Specify ethernet PHY reset timings The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS row tSR Stable supply voltages to reset high is at minimum 10 ms. Set the DT property reset-assert-us to 10ms because the KSZ8041RNL RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/KSZ8041NL-RNL-Data-Sheet-DS00002245.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130429.11214-4-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7791-porter.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7791-porter.dts b/arch/arm/boot/dts/renesas/r8a7791-porter.dts index 811e263452acd9..05989da3be0469 100644 --- a/arch/arm/boot/dts/renesas/r8a7791-porter.dts +++ b/arch/arm/boot/dts/renesas/r8a7791-porter.dts @@ -331,6 +331,8 @@ interrupts-extended = <&irqc0 0 IRQ_TYPE_LEVEL_LOW>; micrel,led-mode = <1>; reset-gpios = <&gpio5 22 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From eb700cb8c67e3b40ab69c77fc07686241fcf705a Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:03:58 +0200 Subject: [PATCH 430/857] arm: dts: renesas: gose: Specify ethernet PHY reset timings The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS row tSR Stable supply voltages to reset high is at minimum 10 ms. Set the DT property reset-assert-us to 10ms because the KSZ8041RNL RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/KSZ8041NL-RNL-Data-Sheet-DS00002245.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130429.11214-5-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7793-gose.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7793-gose.dts b/arch/arm/boot/dts/renesas/r8a7793-gose.dts index 69d9c674bb0329..9687c1eebfd4d4 100644 --- a/arch/arm/boot/dts/renesas/r8a7793-gose.dts +++ b/arch/arm/boot/dts/renesas/r8a7793-gose.dts @@ -621,6 +621,8 @@ interrupts-extended = <&irqc0 0 IRQ_TYPE_LEVEL_LOW>; micrel,led-mode = <1>; reset-gpios = <&gpio5 22 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From f3dd3cda839737c3c1957e3235f7af5b4198ad7f Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:03:59 +0200 Subject: [PATCH 431/857] arm: dts: renesas: alt: Specify ethernet PHY reset timings The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS row tSR Stable supply voltages to reset high is at minimum 10 ms. Set the DT property reset-assert-us to 10ms because the KSZ8041RNL RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/KSZ8041NL-RNL-Data-Sheet-DS00002245.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130429.11214-6-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7794-alt.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7794-alt.dts b/arch/arm/boot/dts/renesas/r8a7794-alt.dts index 5d6d0d8cc4dd8f..92bc4ec4f854fc 100644 --- a/arch/arm/boot/dts/renesas/r8a7794-alt.dts +++ b/arch/arm/boot/dts/renesas/r8a7794-alt.dts @@ -383,6 +383,8 @@ interrupts-extended = <&irqc0 8 IRQ_TYPE_LEVEL_LOW>; micrel,led-mode = <1>; reset-gpios = <&gpio1 24 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From 687711edb90f844088cf08437eaebb46adf772be Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:04:00 +0200 Subject: [PATCH 432/857] arm: dts: renesas: silk: Specify ethernet PHY reset timings The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS row tSR Stable supply voltages to reset high is at minimum 10 ms. Set the DT property reset-assert-us to 10ms because the KSZ8041RNL RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/KSZ8041NL-RNL-Data-Sheet-DS00002245.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130429.11214-7-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7794-silk.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7794-silk.dts b/arch/arm/boot/dts/renesas/r8a7794-silk.dts index af474b1d9676d7..72829ce0524087 100644 --- a/arch/arm/boot/dts/renesas/r8a7794-silk.dts +++ b/arch/arm/boot/dts/renesas/r8a7794-silk.dts @@ -417,6 +417,8 @@ interrupts-extended = <&irqc0 8 IRQ_TYPE_LEVEL_LOW>; micrel,led-mode = <1>; reset-gpios = <&gpio1 24 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From c7406d78ad7c36c5e8327994cbda78e00cd5fedc Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:04:01 +0200 Subject: [PATCH 433/857] arm: dts: renesas: sk-rzg1m: Specify ethernet PHY reset timings The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS row tSR Stable supply voltages to reset high is at minimum 10 ms. Set the DT property reset-assert-us to 10ms because the KSZ8041RNL RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/KSZ8041NL-RNL-Data-Sheet-DS00002245.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130429.11214-8-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7743-sk-rzg1m.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7743-sk-rzg1m.dts b/arch/arm/boot/dts/renesas/r8a7743-sk-rzg1m.dts index 60217797e5345b..dc1b4ca9446225 100644 --- a/arch/arm/boot/dts/renesas/r8a7743-sk-rzg1m.dts +++ b/arch/arm/boot/dts/renesas/r8a7743-sk-rzg1m.dts @@ -75,5 +75,7 @@ interrupts-extended = <&irqc 0 IRQ_TYPE_LEVEL_LOW>; micrel,led-mode = <1>; reset-gpios = <&gpio5 22 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From d7b957fa4b96fd3c25b12bd8f1c23bdf536711e1 Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:04:02 +0200 Subject: [PATCH 434/857] arm: dts: renesas: sk-rzg1e: Specify ethernet PHY reset timings The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS does not indicate how long should the system wait after deassertion of the PHY reset and before start of communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ8041RNL reference manual [1] DS00002245C page 47 TABLE 7-10: POWER-UP/RESET TIMING PARAMETERS row tSR Stable supply voltages to reset high is at minimum 10 ms. Set the DT property reset-assert-us to 10ms because the KSZ8041RNL RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/KSZ8041NL-RNL-Data-Sheet-DS00002245.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130429.11214-9-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7745-sk-rzg1e.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r8a7745-sk-rzg1e.dts b/arch/arm/boot/dts/renesas/r8a7745-sk-rzg1e.dts index 42e82f0697553c..26ee1eec521be5 100644 --- a/arch/arm/boot/dts/renesas/r8a7745-sk-rzg1e.dts +++ b/arch/arm/boot/dts/renesas/r8a7745-sk-rzg1e.dts @@ -70,5 +70,7 @@ interrupts-extended = <&irqc 8 IRQ_TYPE_LEVEL_LOW>; micrel,led-mode = <1>; reset-gpios = <&gpio1 24 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From 2b0a6b2cb28c16e3ad1b90ec02b146838ae11d60 Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:04:57 +0200 Subject: [PATCH 435/857] arm64: dts: renesas: cat875: Specify ethernet PHY reset timings The RTL8211E reference manual [1] page 38 chapter 7.16. PHY Reset (Hardware Reset) states that the PHYRSTB pin must be asserted low for at least 10ms (Tgap in Figure 10) and the system must wait at minimum 30ms (for internal circuits settle time) before accessing the PHY registers. Use 15ms and 35ms respectively to provide some additional margin. [1] https://files.pine64.org/doc/datasheet/pine64/rtl8211e%28g%29-vb%28vl%29-cg_datasheet_1.6.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130515.11262-1-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/cat875.dtsi | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/cat875.dtsi b/arch/arm64/boot/dts/renesas/cat875.dtsi index 5815e9d2d8a933..196cf2e6007ea3 100644 --- a/arch/arm64/boot/dts/renesas/cat875.dtsi +++ b/arch/arm64/boot/dts/renesas/cat875.dtsi @@ -26,6 +26,8 @@ reg = <0>; interrupts-extended = <&gpio2 21 IRQ_TYPE_LEVEL_LOW>; reset-gpios = <&gpio1 20 GPIO_ACTIVE_LOW>; + reset-assert-us = <15000>; + reset-deassert-us = <35000>; }; }; From b716c74ef5d62a664237e4dc0f4e5f2d4d5c6e28 Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:04:58 +0200 Subject: [PATCH 436/857] arm64: dts: renesas: hihope-rzg2-ex: Specify ethernet PHY reset timings The RTL8211E reference manual [1] page 38 chapter 7.16. PHY Reset (Hardware Reset) states that the PHYRSTB pin must be asserted low for at least 10ms (Tgap in Figure 10) and the system must wait at minimum 30ms (for internal circuits settle time) before accessing the PHY registers. Use 15ms and 35ms respectively to provide some additional margin. [1] https://files.pine64.org/doc/datasheet/pine64/rtl8211e%28g%29-vb%28vl%29-cg_datasheet_1.6.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130515.11262-2-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi b/arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi index 83b6c04274ac93..58b787db0f24f4 100644 --- a/arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi +++ b/arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi @@ -28,6 +28,8 @@ reg = <0>; interrupts-extended = <&gpio2 11 IRQ_TYPE_LEVEL_LOW>; reset-gpios = <&gpio2 10 GPIO_ACTIVE_LOW>; + reset-assert-us = <15000>; + reset-deassert-us = <35000>; }; }; From 6e2a4471e679761abf82be3bd3710a2e61a9d95d Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 14 Jul 2026 15:05:28 +0200 Subject: [PATCH 437/857] arm64: dts: renesas: beacon: Specify ethernet PHY reset timings The KSZ9131RNX reference manual [1] DS00002841D page 131 FIGURE 6-2: POWER SEQUENCE TIMING INTERNAL REGULATORS does not indicate how long should the system wait after deassertion of the PHY reset and before starting communication with the PHY via MDIO. Opt for the same value as used for KSZ9031RNX, which is 300 us. The KSZ9131RNX reference manual [1] DS00002841D page 131 FIGURE 6-2: POWER SEQUENCE TIMING INTERNAL REGULATORS row tSR Stable supply voltages to de-assertion of reset is at minimum 10 ms. Set DT property reset-assert-us to 10ms because the KSZ9131RNX RM does not explicitly spell out how long the reset has to be asserted, but this at least covers the worst case scenario. [1] https://ww1.microchip.com/downloads/aemDocuments/documents/UNG/ProductDocuments/DataSheets/00002841D.pdf Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260714130539.11287-1-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/beacon-renesom-som.dtsi | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/beacon-renesom-som.dtsi b/arch/arm64/boot/dts/renesas/beacon-renesom-som.dtsi index f8442b6a85a756..8723e7a76c9ddf 100644 --- a/arch/arm64/boot/dts/renesas/beacon-renesom-som.dtsi +++ b/arch/arm64/boot/dts/renesas/beacon-renesom-som.dtsi @@ -63,6 +63,8 @@ reg = <0>; interrupts-extended = <&gpio2 11 IRQ_TYPE_LEVEL_LOW>; reset-gpios = <&gpio2 10 GPIO_ACTIVE_LOW>; + reset-assert-us = <10000>; + reset-deassert-us = <300>; }; }; From a6cd9d3ee97a77cc4f0ab123a0ad1274aef91695 Mon Sep 17 00:00:00 2001 From: Wolfram Sang Date: Wed, 15 Jul 2026 12:53:07 +0200 Subject: [PATCH 438/857] ARM: dts: renesas: lager: Specify correct connector for i2cexio0 bus It is located on EXIO connector C, not A. Signed-off-by: Wolfram Sang Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260715105306.25147-5-wsa+renesas@sang-engineering.com Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7790-lager.dts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm/boot/dts/renesas/r8a7790-lager.dts b/arch/arm/boot/dts/renesas/r8a7790-lager.dts index 86548859e803d4..5ef001c8f3f3bf 100644 --- a/arch/arm/boot/dts/renesas/r8a7790-lager.dts +++ b/arch/arm/boot/dts/renesas/r8a7790-lager.dts @@ -302,7 +302,7 @@ }; /* - * IIC0/I2C0 is routed to EXIO connector A, pins 114 (SCL) + 116 (SDA) only. + * IIC0/I2C0 is routed to EXIO connector C, pins 114 (SCL) + 116 (SDA) only. * We use the I2C demuxer, so the desired IP core can be selected at runtime * depending on the use case (e.g. DMA with IIC0 or slave support with I2C0). * Note: For testing the I2C slave feature, it is convenient to connect this From 8d1ed99602603a08cc3a8f7d62c1874c035fb332 Mon Sep 17 00:00:00 2001 From: Wolfram Sang Date: Wed, 15 Jul 2026 12:53:08 +0200 Subject: [PATCH 439/857] ARM: dts: renesas: lager: Use inclusive wording The feature is now officially called "target" in I2C documentation. Signed-off-by: Wolfram Sang Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260715105306.25147-6-wsa+renesas@sang-engineering.com Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7790-lager.dts | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/arch/arm/boot/dts/renesas/r8a7790-lager.dts b/arch/arm/boot/dts/renesas/r8a7790-lager.dts index 5ef001c8f3f3bf..7366488f176e26 100644 --- a/arch/arm/boot/dts/renesas/r8a7790-lager.dts +++ b/arch/arm/boot/dts/renesas/r8a7790-lager.dts @@ -304,11 +304,11 @@ /* * IIC0/I2C0 is routed to EXIO connector C, pins 114 (SCL) + 116 (SDA) only. * We use the I2C demuxer, so the desired IP core can be selected at runtime - * depending on the use case (e.g. DMA with IIC0 or slave support with I2C0). - * Note: For testing the I2C slave feature, it is convenient to connect this + * depending on the use case (e.g. DMA with IIC0 or target support with I2C0). + * Note: For testing the I2C target feature, it is convenient to connect this * bus with IIC3 on pins 110 (SCL) + 112 (SDA), select I2C0 at runtime, and - * instantiate the slave device at runtime according to the documentation. - * You can then communicate with the slave via IIC3. + * instantiate the target device at runtime according to the documentation. + * You can then communicate with the target via IIC3. * * IIC0/I2C0 does not appear to support fallback to GPIO. */ From 6f71104288c78c091070c6367b705c68e4faa118 Mon Sep 17 00:00:00 2001 From: Wolfram Sang Date: Thu, 16 Jul 2026 11:59:36 +0200 Subject: [PATCH 440/857] ARM: dts: renesas: r9a06g032-rzn1d400: Don't enable gpiochips which are always on gpiochips are always enabled, no need to enable them explicitly. Signed-off-by: Wolfram Sang Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260716095935.13329-5-wsa+renesas@sang-engineering.com Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-db.dts | 1 - arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts | 4 ---- 2 files changed, 5 deletions(-) diff --git a/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-db.dts b/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-db.dts index 5626d7fd6c3ee8..49f2def1bb128a 100644 --- a/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-db.dts +++ b/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-db.dts @@ -194,7 +194,6 @@ &gpio2 { pinctrl-0 = <&pins_gpio2>; pinctrl-names = "default"; - status = "okay"; }; &i2c2 { diff --git a/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts b/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts index ead379988fb1c6..303b5926d8d173 100644 --- a/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts +++ b/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts @@ -53,10 +53,6 @@ }; }; -&gpio2 { - status = "okay"; -}; - &i2c2 { /* Sensors are different across revisions. All are LM75B compatible */ sensor@49 { From 64ec1a4d2fb5ca997c8dc6cd79779b08a3fffa82 Mon Sep 17 00:00:00 2001 From: Wolfram Sang Date: Thu, 16 Jul 2026 12:10:30 +0200 Subject: [PATCH 441/857] ARM: dts: renesas: r9a06g032: Don't set status to the default value "okay" is the default for 'status', so don't encode it explicitly. Signed-off-by: Wolfram Sang Tested-by: Miquel Raynal Reviewed-by: Miquel Raynal Reviewed-by: Herve Codina Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260716101030.15252-2-wsa+renesas@sang-engineering.com Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r9a06g032.dtsi | 2 -- 1 file changed, 2 deletions(-) diff --git a/arch/arm/boot/dts/renesas/r9a06g032.dtsi b/arch/arm/boot/dts/renesas/r9a06g032.dtsi index 19c9bce0a26d69..8645bab7ee2392 100644 --- a/arch/arm/boot/dts/renesas/r9a06g032.dtsi +++ b/arch/arm/boot/dts/renesas/r9a06g032.dtsi @@ -145,7 +145,6 @@ sysctrl: system-controller@4000c000 { compatible = "renesas,r9a06g032-sysctrl"; reg = <0x4000c000 0x1000>; - status = "okay"; #clock-cells = <1>; #power-domain-cells = <0>; @@ -352,7 +351,6 @@ reg = <0x40067000 0x1000>, <0x51000000 0x480>; clocks = <&sysctrl R9A06G032_HCLK_PINCONFIG>; clock-names = "bus"; - status = "okay"; }; sdio1: mmc@40100000 { From d74cc454a19e55c65eaac83a9bf7f573e4b4335c Mon Sep 17 00:00:00 2001 From: Krzysztof Kozlowski Date: Sat, 1 Aug 2026 23:03:22 +0200 Subject: [PATCH 442/857] ARM: dts: renesas: Correct white-space style Correct a few white-space issues, like using tab after ':' character or spurious space before '{', which will be flagged by dt-check-style ("redundant-whitespace" warning). No functional changes. Signed-off-by: Krzysztof Kozlowski Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260801210321.383773-3-krzysztof.kozlowski@oss.qualcomm.com Signed-off-by: Geert Uytterhoeven --- arch/arm/boot/dts/renesas/r8a7743.dtsi | 6 +++--- arch/arm/boot/dts/renesas/r8a7744.dtsi | 6 +++--- arch/arm/boot/dts/renesas/r8a7745.dtsi | 6 +++--- arch/arm/boot/dts/renesas/r8a77470.dtsi | 6 +++--- arch/arm/boot/dts/renesas/r8a7790-lager.dts | 16 ++++++++-------- arch/arm/boot/dts/renesas/r8a7790-stout.dts | 2 +- arch/arm/boot/dts/renesas/r8a7790.dtsi | 4 ++-- arch/arm/boot/dts/renesas/r8a7791.dtsi | 4 ++-- arch/arm/boot/dts/renesas/r8a7792.dtsi | 4 ++-- arch/arm/boot/dts/renesas/r8a7793.dtsi | 4 ++-- arch/arm/boot/dts/renesas/r8a7794.dtsi | 4 ++-- 11 files changed, 31 insertions(+), 31 deletions(-) diff --git a/arch/arm/boot/dts/renesas/r8a7743.dtsi b/arch/arm/boot/dts/renesas/r8a7743.dtsi index c697942387e1d5..4db849b5bc6b91 100644 --- a/arch/arm/boot/dts/renesas/r8a7743.dtsi +++ b/arch/arm/boot/dts/renesas/r8a7743.dtsi @@ -456,7 +456,7 @@ status = "disabled"; }; - icram0: sram@e63a0000 { + icram0: sram@e63a0000 { compatible = "mmio-sram"; reg = <0 0xe63a0000 0 0x12000>; #address-cells = <1>; @@ -464,7 +464,7 @@ ranges = <0 0 0xe63a0000 0x12000>; }; - icram1: sram@e63c0000 { + icram1: sram@e63c0000 { compatible = "mmio-sram"; reg = <0 0xe63c0000 0 0x1000>; #address-cells = <1>; @@ -477,7 +477,7 @@ }; }; - icram2: sram@e6300000 { + icram2: sram@e6300000 { compatible = "mmio-sram"; reg = <0 0xe6300000 0 0x40000>; #address-cells = <1>; diff --git a/arch/arm/boot/dts/renesas/r8a7744.dtsi b/arch/arm/boot/dts/renesas/r8a7744.dtsi index fed46345807cb0..1a85bc0dc5f399 100644 --- a/arch/arm/boot/dts/renesas/r8a7744.dtsi +++ b/arch/arm/boot/dts/renesas/r8a7744.dtsi @@ -456,7 +456,7 @@ status = "disabled"; }; - icram0: sram@e63a0000 { + icram0: sram@e63a0000 { compatible = "mmio-sram"; reg = <0 0xe63a0000 0 0x12000>; #address-cells = <1>; @@ -464,7 +464,7 @@ ranges = <0 0 0xe63a0000 0x12000>; }; - icram1: sram@e63c0000 { + icram1: sram@e63c0000 { compatible = "mmio-sram"; reg = <0 0xe63c0000 0 0x1000>; #address-cells = <1>; @@ -477,7 +477,7 @@ }; }; - icram2: sram@e6300000 { + icram2: sram@e6300000 { compatible = "mmio-sram"; reg = <0 0xe6300000 0 0x40000>; #address-cells = <1>; diff --git a/arch/arm/boot/dts/renesas/r8a7745.dtsi b/arch/arm/boot/dts/renesas/r8a7745.dtsi index 5424a73562ddba..78ff5a9f8f3860 100644 --- a/arch/arm/boot/dts/renesas/r8a7745.dtsi +++ b/arch/arm/boot/dts/renesas/r8a7745.dtsi @@ -420,7 +420,7 @@ status = "disabled"; }; - icram0: sram@e63a0000 { + icram0: sram@e63a0000 { compatible = "mmio-sram"; reg = <0 0xe63a0000 0 0x12000>; #address-cells = <1>; @@ -428,7 +428,7 @@ ranges = <0 0 0xe63a0000 0x12000>; }; - icram1: sram@e63c0000 { + icram1: sram@e63c0000 { compatible = "mmio-sram"; reg = <0 0xe63c0000 0 0x1000>; #address-cells = <1>; @@ -441,7 +441,7 @@ }; }; - icram2: sram@e6300000 { + icram2: sram@e6300000 { compatible = "mmio-sram"; reg = <0 0xe6300000 0 0x40000>; #address-cells = <1>; diff --git a/arch/arm/boot/dts/renesas/r8a77470.dtsi b/arch/arm/boot/dts/renesas/r8a77470.dtsi index c61790e7667f58..02df768bd2bc3e 100644 --- a/arch/arm/boot/dts/renesas/r8a77470.dtsi +++ b/arch/arm/boot/dts/renesas/r8a77470.dtsi @@ -285,7 +285,7 @@ status = "disabled"; }; - icram0: sram@e63a0000 { + icram0: sram@e63a0000 { compatible = "mmio-sram"; reg = <0 0xe63a0000 0 0x12000>; #address-cells = <1>; @@ -293,7 +293,7 @@ ranges = <0 0 0xe63a0000 0x12000>; }; - icram1: sram@e63c0000 { + icram1: sram@e63c0000 { compatible = "mmio-sram"; reg = <0 0xe63c0000 0 0x1000>; #address-cells = <1>; @@ -306,7 +306,7 @@ }; }; - icram2: sram@e6300000 { + icram2: sram@e6300000 { compatible = "mmio-sram"; reg = <0 0xe6300000 0 0x20000>; #address-cells = <1>; diff --git a/arch/arm/boot/dts/renesas/r8a7790-lager.dts b/arch/arm/boot/dts/renesas/r8a7790-lager.dts index 7366488f176e26..f7345d19337bb5 100644 --- a/arch/arm/boot/dts/renesas/r8a7790-lager.dts +++ b/arch/arm/boot/dts/renesas/r8a7790-lager.dts @@ -815,46 +815,46 @@ cpu0-supply = <&vdd_dvfs>; }; -&i2c0 { +&i2c0 { pinctrl-0 = <&i2c0_pins>; pinctrl-names = "i2c-exio0"; }; -&iic0 { +&iic0 { pinctrl-0 = <&iic0_pins>; pinctrl-names = "i2c-exio0"; }; -&i2c1 { +&i2c1 { pinctrl-0 = <&i2c1_pins>; pinctrl-names = "i2c-exio1"; }; -&iic1 { +&iic1 { pinctrl-0 = <&iic1_pins>; pinctrl-names = "i2c-exio1"; }; -&i2c2 { +&i2c2 { pinctrl-0 = <&i2c2_pins>; pinctrl-names = "i2c-hdmi"; clock-frequency = <100000>; }; -&iic2 { +&iic2 { pinctrl-0 = <&iic2_pins>; pinctrl-names = "i2c-hdmi"; clock-frequency = <100000>; }; -&i2c3 { +&i2c3 { pinctrl-0 = <&i2c3_pins>; pinctrl-names = "i2c-pwr"; }; -&iic3 { +&iic3 { pinctrl-0 = <&iic3_pins>; pinctrl-names = "i2c-pwr"; }; diff --git a/arch/arm/boot/dts/renesas/r8a7790-stout.dts b/arch/arm/boot/dts/renesas/r8a7790-stout.dts index b062423499e480..95adece7681a4b 100644 --- a/arch/arm/boot/dts/renesas/r8a7790-stout.dts +++ b/arch/arm/boot/dts/renesas/r8a7790-stout.dts @@ -291,7 +291,7 @@ cpu0-supply = <&vdd_dvfs>; }; -&iic2 { +&iic2 { status = "okay"; pinctrl-0 = <&iic2_pins>; pinctrl-names = "default"; diff --git a/arch/arm/boot/dts/renesas/r8a7790.dtsi b/arch/arm/boot/dts/renesas/r8a7790.dtsi index 12cce9bdc44991..358c8061290ac7 100644 --- a/arch/arm/boot/dts/renesas/r8a7790.dtsi +++ b/arch/arm/boot/dts/renesas/r8a7790.dtsi @@ -566,7 +566,7 @@ status = "disabled"; }; - icram0: sram@e63a0000 { + icram0: sram@e63a0000 { compatible = "mmio-sram"; reg = <0 0xe63a0000 0 0x12000>; #address-cells = <1>; @@ -574,7 +574,7 @@ ranges = <0 0 0xe63a0000 0x12000>; }; - icram1: sram@e63c0000 { + icram1: sram@e63c0000 { compatible = "mmio-sram"; reg = <0 0xe63c0000 0 0x1000>; #address-cells = <1>; diff --git a/arch/arm/boot/dts/renesas/r8a7791.dtsi b/arch/arm/boot/dts/renesas/r8a7791.dtsi index 35313e8da42650..3a447a18d03882 100644 --- a/arch/arm/boot/dts/renesas/r8a7791.dtsi +++ b/arch/arm/boot/dts/renesas/r8a7791.dtsi @@ -493,7 +493,7 @@ status = "disabled"; }; - icram0: sram@e63a0000 { + icram0: sram@e63a0000 { compatible = "mmio-sram"; reg = <0 0xe63a0000 0 0x12000>; #address-cells = <1>; @@ -501,7 +501,7 @@ ranges = <0 0 0xe63a0000 0x12000>; }; - icram1: sram@e63c0000 { + icram1: sram@e63c0000 { compatible = "mmio-sram"; reg = <0 0xe63c0000 0 0x1000>; #address-cells = <1>; diff --git a/arch/arm/boot/dts/renesas/r8a7792.dtsi b/arch/arm/boot/dts/renesas/r8a7792.dtsi index fbdbcff1cbed41..a63a28dd2bc784 100644 --- a/arch/arm/boot/dts/renesas/r8a7792.dtsi +++ b/arch/arm/boot/dts/renesas/r8a7792.dtsi @@ -415,7 +415,7 @@ status = "disabled"; }; - icram0: sram@e63a0000 { + icram0: sram@e63a0000 { compatible = "mmio-sram"; reg = <0 0xe63a0000 0 0x12000>; #address-cells = <1>; @@ -423,7 +423,7 @@ ranges = <0 0 0xe63a0000 0x12000>; }; - icram1: sram@e63c0000 { + icram1: sram@e63c0000 { compatible = "mmio-sram"; reg = <0 0xe63c0000 0 0x1000>; #address-cells = <1>; diff --git a/arch/arm/boot/dts/renesas/r8a7793.dtsi b/arch/arm/boot/dts/renesas/r8a7793.dtsi index 1ad50070a1a732..4cb3bf305ab556 100644 --- a/arch/arm/boot/dts/renesas/r8a7793.dtsi +++ b/arch/arm/boot/dts/renesas/r8a7793.dtsi @@ -468,7 +468,7 @@ status = "disabled"; }; - icram0: sram@e63a0000 { + icram0: sram@e63a0000 { compatible = "mmio-sram"; reg = <0 0xe63a0000 0 0x12000>; #address-cells = <1>; @@ -476,7 +476,7 @@ ranges = <0 0 0xe63a0000 0x12000>; }; - icram1: sram@e63c0000 { + icram1: sram@e63c0000 { compatible = "mmio-sram"; reg = <0 0xe63c0000 0 0x1000>; #address-cells = <1>; diff --git a/arch/arm/boot/dts/renesas/r8a7794.dtsi b/arch/arm/boot/dts/renesas/r8a7794.dtsi index 7669a67377c989..732e99f830c7ce 100644 --- a/arch/arm/boot/dts/renesas/r8a7794.dtsi +++ b/arch/arm/boot/dts/renesas/r8a7794.dtsi @@ -413,7 +413,7 @@ status = "disabled"; }; - icram0: sram@e63a0000 { + icram0: sram@e63a0000 { compatible = "mmio-sram"; reg = <0 0xe63a0000 0 0x12000>; #address-cells = <1>; @@ -421,7 +421,7 @@ ranges = <0 0 0xe63a0000 0x12000>; }; - icram1: sram@e63c0000 { + icram1: sram@e63c0000 { compatible = "mmio-sram"; reg = <0 0xe63c0000 0 0x1000>; #address-cells = <1>; From b10a17eee62fa7a7d926f3e374bf4eeeeded4c28 Mon Sep 17 00:00:00 2001 From: Krzysztof Kozlowski Date: Sat, 1 Aug 2026 23:03:23 +0200 Subject: [PATCH 443/857] arm64: dts: renesas: Correct white-space style Correct a few white-space issues, like missing space before bracket '{' character or spurious space, which will be flagged by dt-check-style ("redundant-whitespace" warning). No functional changes. Signed-off-by: Krzysztof Kozlowski Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260801210321.383773-4-krzysztof.kozlowski@oss.qualcomm.com Signed-off-by: Geert Uytterhoeven --- .../r8a77970-eagle-function-expansion.dtso | 2 +- arch/arm64/boot/dts/renesas/r8a779md-geist.dts | 2 +- .../boot/dts/renesas/r9a07g044l2-remi-pi.dts | 2 +- arch/arm64/boot/dts/renesas/r9a08g045.dtsi | 2 +- arch/arm64/boot/dts/renesas/r9a09g011.dtsi | 4 ++-- arch/arm64/boot/dts/renesas/r9a09g047.dtsi | 2 +- arch/arm64/boot/dts/renesas/r9a09g056.dtsi | 2 +- arch/arm64/boot/dts/renesas/r9a09g057.dtsi | 2 +- arch/arm64/boot/dts/renesas/r9a09g077.dtsi | 2 +- arch/arm64/boot/dts/renesas/r9a09g087.dtsi | 2 +- .../boot/dts/renesas/salvator-common.dtsi | 2 +- .../ulcb-kf-audio-graph-card2-mix+split.dtsi | 18 +++++++++--------- 12 files changed, 21 insertions(+), 21 deletions(-) diff --git a/arch/arm64/boot/dts/renesas/r8a77970-eagle-function-expansion.dtso b/arch/arm64/boot/dts/renesas/r8a77970-eagle-function-expansion.dtso index ecb35257b9ae18..a9ad009ff7b04b 100644 --- a/arch/arm64/boot/dts/renesas/r8a77970-eagle-function-expansion.dtso +++ b/arch/arm64/boot/dts/renesas/r8a77970-eagle-function-expansion.dtso @@ -112,7 +112,7 @@ reg = <0x70 0x71 0x72 0x73 0x74 0x75 0x60 0x61 0x62 0x63 0x64 0x65>; reg-names = "main", "dpll", "cp", "hdmi", "edid", "repeater", - "infoframe", "cbus", "cec", "sdp", "txa", "txb" ; + "infoframe", "cbus", "cec", "sdp", "txa", "txb"; interrupts-extended = <&gpio3 3 IRQ_TYPE_LEVEL_LOW>, <&gpio3 4 IRQ_TYPE_LEVEL_LOW>; interrupt-names = "intrq1", "intrq2"; diff --git a/arch/arm64/boot/dts/renesas/r8a779md-geist.dts b/arch/arm64/boot/dts/renesas/r8a779md-geist.dts index 0e4724336e7346..b186807926d2b3 100644 --- a/arch/arm64/boot/dts/renesas/r8a779md-geist.dts +++ b/arch/arm64/boot/dts/renesas/r8a779md-geist.dts @@ -367,7 +367,7 @@ reg = <0x70 0x71 0x72 0x73 0x74 0x75 0x60 0x61 0x62 0x63 0x64 0x65>; reg-names = "main", "dpll", "cp", "hdmi", "edid", "repeater", - "infoframe", "cbus", "cec", "sdp", "txa", "txb" ; + "infoframe", "cbus", "cec", "sdp", "txa", "txb"; interrupts-extended = <&gpio6 30 IRQ_TYPE_LEVEL_LOW>, <&gpio6 31 IRQ_TYPE_LEVEL_LOW>; diff --git a/arch/arm64/boot/dts/renesas/r9a07g044l2-remi-pi.dts b/arch/arm64/boot/dts/renesas/r9a07g044l2-remi-pi.dts index 3267e7b75b58ff..dea386f3c836e9 100644 --- a/arch/arm64/boot/dts/renesas/r9a07g044l2-remi-pi.dts +++ b/arch/arm64/boot/dts/renesas/r9a07g044l2-remi-pi.dts @@ -156,7 +156,7 @@ hdmi-bridge@48 { compatible = "lontium,lt8912b"; - reg = <0x48> ; + reg = <0x48>; reset-gpios = <&pinctrl RZG2L_GPIO(42, 2) GPIO_ACTIVE_LOW>; ports { diff --git a/arch/arm64/boot/dts/renesas/r9a08g045.dtsi b/arch/arm64/boot/dts/renesas/r9a08g045.dtsi index ae92d45ede387b..b98ad375a3e9c1 100644 --- a/arch/arm64/boot/dts/renesas/r9a08g045.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a08g045.dtsi @@ -645,7 +645,7 @@ dma-channels = <16>; }; - sdhi0: mmc@11c00000 { + sdhi0: mmc@11c00000 { compatible = "renesas,sdhi-r9a08g045", "renesas,rzg2l-sdhi"; reg = <0x0 0x11c00000 0 0x10000>; interrupts = , diff --git a/arch/arm64/boot/dts/renesas/r9a09g011.dtsi b/arch/arm64/boot/dts/renesas/r9a09g011.dtsi index 42462c138dd236..fbe1e1c24f57a6 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g011.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g011.dtsi @@ -85,7 +85,7 @@ status = "disabled"; }; - sdhi1: mmc@85010000 { + sdhi1: mmc@85010000 { compatible = "renesas,sdhi-r9a09g011", "renesas,rzg2l-sdhi"; reg = <0x0 0x85010000 0 0x2000>; @@ -101,7 +101,7 @@ status = "disabled"; }; - emmc: mmc@85020000 { + emmc: mmc@85020000 { compatible = "renesas,sdhi-r9a09g011", "renesas,rzg2l-sdhi"; reg = <0x0 0x85020000 0 0x2000>; diff --git a/arch/arm64/boot/dts/renesas/r9a09g047.dtsi b/arch/arm64/boot/dts/renesas/r9a09g047.dtsi index 060405d4a2ddea..72120e2abf8b49 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g047.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g047.dtsi @@ -1773,7 +1773,7 @@ status = "disabled"; }; - sdhi0: mmc@15c00000 { + sdhi0: mmc@15c00000 { compatible = "renesas,sdhi-r9a09g047", "renesas,sdhi-r9a09g057"; reg = <0x0 0x15c00000 0 0x10000>; interrupts = , diff --git a/arch/arm64/boot/dts/renesas/r9a09g056.dtsi b/arch/arm64/boot/dts/renesas/r9a09g056.dtsi index 77c2221a9e2a3c..175e24c98e11d2 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g056.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g056.dtsi @@ -1447,7 +1447,7 @@ status = "disabled"; }; - sdhi0: mmc@15c00000 { + sdhi0: mmc@15c00000 { compatible = "renesas,sdhi-r9a09g056", "renesas,sdhi-r9a09g057"; reg = <0x0 0x15c00000 0 0x10000>; interrupts = , diff --git a/arch/arm64/boot/dts/renesas/r9a09g057.dtsi b/arch/arm64/boot/dts/renesas/r9a09g057.dtsi index 188ce9f9c7c21d..bbaab75681c299 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g057.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g057.dtsi @@ -1577,7 +1577,7 @@ status = "disabled"; }; - sdhi0: mmc@15c00000 { + sdhi0: mmc@15c00000 { compatible = "renesas,sdhi-r9a09g057"; reg = <0x0 0x15c00000 0 0x10000>; interrupts = , diff --git a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi index 73cfb62952c4b6..d5eafaf23e75ec 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi @@ -1364,7 +1364,7 @@ status = "disabled"; }; - sdhi0: mmc@92080000 { + sdhi0: mmc@92080000 { compatible = "renesas,sdhi-r9a09g077", "renesas,sdhi-r9a09g057"; reg = <0x0 0x92080000 0 0x10000>; diff --git a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi index 2be6d39a37ad25..fe9003426d70ef 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi @@ -1367,7 +1367,7 @@ status = "disabled"; }; - sdhi0: mmc@92080000 { + sdhi0: mmc@92080000 { compatible = "renesas,sdhi-r9a09g087", "renesas,sdhi-r9a09g057"; reg = <0x0 0x92080000 0 0x10000>; diff --git a/arch/arm64/boot/dts/renesas/salvator-common.dtsi b/arch/arm64/boot/dts/renesas/salvator-common.dtsi index 1317ede2f719be..9f8c545dad34ad 100644 --- a/arch/arm64/boot/dts/renesas/salvator-common.dtsi +++ b/arch/arm64/boot/dts/renesas/salvator-common.dtsi @@ -543,7 +543,7 @@ reg = <0x70 0x71 0x72 0x73 0x74 0x75 0x60 0x61 0x62 0x63 0x64 0x65>; reg-names = "main", "dpll", "cp", "hdmi", "edid", "repeater", - "infoframe", "cbus", "cec", "sdp", "txa", "txb" ; + "infoframe", "cbus", "cec", "sdp", "txa", "txb"; interrupts-extended = <&gpio6 30 IRQ_TYPE_LEVEL_LOW>, <&gpio6 31 IRQ_TYPE_LEVEL_LOW>; diff --git a/arch/arm64/boot/dts/renesas/ulcb-kf-audio-graph-card2-mix+split.dtsi b/arch/arm64/boot/dts/renesas/ulcb-kf-audio-graph-card2-mix+split.dtsi index 67a0057a3383da..2460ad849b9399 100644 --- a/arch/arm64/boot/dts/renesas/ulcb-kf-audio-graph-card2-mix+split.dtsi +++ b/arch/arm64/boot/dts/renesas/ulcb-kf-audio-graph-card2-mix+split.dtsi @@ -76,14 +76,14 @@ * (H) CPU7 * (I) CPU8 */ - fe_c: port@2 { reg = <2>; fe_c_ep: endpoint { remote-endpoint = <&rsnd_c_ep>; }; }; - fe_d: port@3 { reg = <3>; fe_d_ep: endpoint { remote-endpoint = <&rsnd_d_ep>; }; }; - fe_e: port@4 { reg = <4>; fe_e_ep: endpoint { remote-endpoint = <&rsnd_e_ep>; }; }; - fe_f: port@5 { reg = <5>; fe_f_ep: endpoint { remote-endpoint = <&rsnd_f_ep>; }; }; + fe_c: port@2 { reg = <2>; fe_c_ep: endpoint { remote-endpoint = <&rsnd_c_ep>; }; }; + fe_d: port@3 { reg = <3>; fe_d_ep: endpoint { remote-endpoint = <&rsnd_d_ep>; }; }; + fe_e: port@4 { reg = <4>; fe_e_ep: endpoint { remote-endpoint = <&rsnd_e_ep>; }; }; + fe_f: port@5 { reg = <5>; fe_f_ep: endpoint { remote-endpoint = <&rsnd_f_ep>; }; }; - fe_g: port@6 { reg = <6>; fe_g_ep: endpoint { remote-endpoint = <&rsnd_g_ep>; }; }; - fe_h: port@7 { reg = <7>; fe_h_ep: endpoint { remote-endpoint = <&rsnd_h_ep>; }; }; - fe_i: port@8 { reg = <8>; fe_i_ep: endpoint { remote-endpoint = <&rsnd_i_ep>; }; }; + fe_g: port@6 { reg = <6>; fe_g_ep: endpoint { remote-endpoint = <&rsnd_g_ep>; }; }; + fe_h: port@7 { reg = <7>; fe_h_ep: endpoint { remote-endpoint = <&rsnd_h_ep>; }; }; + fe_i: port@8 { reg = <8>; fe_i_ep: endpoint { remote-endpoint = <&rsnd_i_ep>; }; }; }; ports@1 { @@ -96,8 +96,8 @@ * (Y) PCM3168A-p * (Z) PCM3168A-c */ - be_y: port@0 { reg = <0>; be_y_ep: endpoint { remote-endpoint = <&pcm3168a_y_ep>; }; }; - be_z: port@1 { reg = <1>; be_z_ep: endpoint { remote-endpoint = <&pcm3168a_z_ep>; }; }; + be_y: port@0 { reg = <0>; be_y_ep: endpoint { remote-endpoint = <&pcm3168a_y_ep>; }; }; + be_z: port@1 { reg = <1>; be_z_ep: endpoint { remote-endpoint = <&pcm3168a_z_ep>; }; }; }; }; }; From 9ea99025641d06f748ac8de81b3d8d9fd7f2fbe8 Mon Sep 17 00:00:00 2001 From: Tommaso Merciai Date: Mon, 3 Aug 2026 15:28:26 +0200 Subject: [PATCH 444/857] arm64: dts: renesas: r9a09g047e57-smarc: Set I2C1 clock frequency to 1 MHz On the SMARC carrier board I2C1 only carries the DA7212 audio codec. Override the 100 kHz SoC default to 1 MHz. Signed-off-by: Tommaso Merciai Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260803132828.3249424-4-tommaso.merciai.xr@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r9a09g047e57-smarc.dts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/r9a09g047e57-smarc.dts b/arch/arm64/boot/dts/renesas/r9a09g047e57-smarc.dts index 4eed095b683b31..469a6282b2a30c 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g047e57-smarc.dts +++ b/arch/arm64/boot/dts/renesas/r9a09g047e57-smarc.dts @@ -147,6 +147,8 @@ }; &i2c1 { + clock-frequency = <1000000>; + da7212: codec@1a { compatible = "dlg,da7212"; #sound-dai-cells = <0>; From f459d075d126c66c0cdc2bcc840affe05cabc2ac Mon Sep 17 00:00:00 2001 From: Geert Uytterhoeven Date: Wed, 5 Aug 2026 17:20:53 +0200 Subject: [PATCH 445/857] arm64: dts: renesas: r8a78000: Add CPG node Add a device node for the Clock Pulse Generator (CPG) on the R-Car X5H (R8A78000) SoC. Convert all (H)SCIF serial ports from dummy to CPG clocks, removing the need for any dummy clocks. Signed-off-by: Geert Uytterhoeven Reviewed-by: Marek Vasut Link: https://patch.msgid.link/03a40ce94e01a5a390493b8c711c299d328a2db3.1785941595.git.geert+renesas@glider.be --- arch/arm64/boot/dts/renesas/r8a78000.dtsi | 59 +++++++++++++---------- 1 file changed, 34 insertions(+), 25 deletions(-) diff --git a/arch/arm64/boot/dts/renesas/r8a78000.dtsi b/arch/arm64/boot/dts/renesas/r8a78000.dtsi index fb71974ef39050..1fe078c7822c01 100644 --- a/arch/arm64/boot/dts/renesas/r8a78000.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a78000.dtsi @@ -5,6 +5,7 @@ * Copyright (C) 2025 Renesas Electronics Corp. */ +#include #include / { @@ -668,23 +669,6 @@ }; }; - /* - * In the early phase, there is no clock control support, - * so assume that the clocks are enabled by default. - * Therefore, dummy clocks are used. - */ - dummy_clk_sgasyncd16: dummy-clk-sgasyncd16 { - compatible = "fixed-clock"; - #clock-cells = <0>; - clock-frequency = <66666000>; - }; - - dummy_clk_sgasyncd4: dummy-clk-sgasyncd4 { - compatible = "fixed-clock"; - #clock-cells = <0>; - clock-frequency = <266660000>; - }; - extal_clk: extal-clk { compatible = "fixed-clock"; #clock-cells = <0>; @@ -882,7 +866,9 @@ "renesas,rcar-gen5-scif", "renesas,scif"; reg = <0 0xc0700000 0 0x40>; interrupts = ; - clocks = <&dummy_clk_sgasyncd16>, <&dummy_clk_sgasyncd4>, <&scif_clk>; + clocks = <&cpg R8A78000_CPG_SGASYNCD16_PERW_BUS>, + <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; status = "disabled"; }; @@ -892,7 +878,9 @@ "renesas,rcar-gen5-scif", "renesas,scif"; reg = <0 0xc0704000 0 0x40>; interrupts = ; - clocks = <&dummy_clk_sgasyncd16>, <&dummy_clk_sgasyncd4>, <&scif_clk>; + clocks = <&cpg R8A78000_CPG_SGASYNCD16_PERW_BUS>, + <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; status = "disabled"; }; @@ -902,7 +890,9 @@ "renesas,rcar-gen5-scif", "renesas,scif"; reg = <0 0xc0708000 0 0x40>; interrupts = ; - clocks = <&dummy_clk_sgasyncd16>, <&dummy_clk_sgasyncd4>, <&scif_clk>; + clocks = <&cpg R8A78000_CPG_SGASYNCD16_PERW_BUS>, + <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; status = "disabled"; }; @@ -912,7 +902,9 @@ "renesas,rcar-gen5-scif", "renesas,scif"; reg = <0 0xc070c000 0 0x40>; interrupts = ; - clocks = <&dummy_clk_sgasyncd16>, <&dummy_clk_sgasyncd4>, <&scif_clk>; + clocks = <&cpg R8A78000_CPG_SGASYNCD16_PERW_BUS>, + <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; status = "disabled"; }; @@ -922,7 +914,9 @@ "renesas,rcar-gen5-hscif", "renesas,hscif"; reg = <0 0xc0710000 0 0x60>; interrupts = ; - clocks = <&dummy_clk_sgasyncd4>, <&dummy_clk_sgasyncd4>, <&scif_clk>; + clocks = <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; status = "disabled"; }; @@ -932,7 +926,9 @@ "renesas,rcar-gen5-hscif", "renesas,hscif"; reg = <0 0xc0714000 0 0x60>; interrupts = ; - clocks = <&dummy_clk_sgasyncd4>, <&dummy_clk_sgasyncd4>, <&scif_clk>; + clocks = <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; status = "disabled"; }; @@ -942,7 +938,9 @@ "renesas,rcar-gen5-hscif", "renesas,hscif"; reg = <0 0xc0718000 0 0x60>; interrupts = ; - clocks = <&dummy_clk_sgasyncd4>, <&dummy_clk_sgasyncd4>, <&scif_clk>; + clocks = <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; status = "disabled"; }; @@ -952,7 +950,9 @@ "renesas,rcar-gen5-hscif", "renesas,hscif"; reg = <0 0xc071c000 0 0x60>; interrupts = ; - clocks = <&dummy_clk_sgasyncd4>, <&dummy_clk_sgasyncd4>, <&scif_clk>; + clocks = <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, + <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; status = "disabled"; }; @@ -965,6 +965,15 @@ ranges = <0 0x0 0xc1060000 0x1c00>; /* actual transport nodes must be set per board file */ }; + + cpg: clock-controller@c1320000 { + compatible = "renesas,r8a78000-cpg"; + reg = <0 0xc1320000 0 0x10000>; + clocks = <&extal_clk>, <&extalr_clk>; + clock-names = "extal", "extalr"; + #clock-cells = <1>; + bootph-all; + }; }; timer { From 19945371cd6f6660a9596e4b05a9aeca028bf445 Mon Sep 17 00:00:00 2001 From: Geert Uytterhoeven Date: Wed, 5 Aug 2026 17:20:54 +0200 Subject: [PATCH 446/857] arm64: dts: renesas: r8a78000: Add MDLC nodes Add device nodes for the Module Control (MDLC) blocks on the R-Car X5H (R8A78000) SoC. Complete hardware desciption of all (H)SCIF serial ports, by linking them to an MDLC for power domains and resets. Signed-off-by: Geert Uytterhoeven Reviewed-by: Marek Vasut Link: https://patch.msgid.link/9b7f96352ef5fe08e7003169710a9af874fccb68.1785941595.git.geert+renesas@glider.be --- arch/arm64/boot/dts/renesas/r8a78000.dtsi | 241 ++++++++++++++++++++++ 1 file changed, 241 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/r8a78000.dtsi b/arch/arm64/boot/dts/renesas/r8a78000.dtsi index 1fe078c7822c01..9a07753b221541 100644 --- a/arch/arm64/boot/dts/renesas/r8a78000.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a78000.dtsi @@ -6,6 +6,7 @@ */ #include +#include #include / { @@ -870,6 +871,8 @@ <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; + power-domains = <&mdlc_perw R8A78000_MDLC_PD_APL 0x40>; + resets = <&mdlc_perw 0x40>; status = "disabled"; }; @@ -882,6 +885,8 @@ <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; + power-domains = <&mdlc_perw R8A78000_MDLC_PD_APL 0x41>; + resets = <&mdlc_perw 0x41>; status = "disabled"; }; @@ -894,6 +899,8 @@ <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; + power-domains = <&mdlc_perw R8A78000_MDLC_PD_APL 0x42>; + resets = <&mdlc_perw 0x42>; status = "disabled"; }; @@ -906,6 +913,8 @@ <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; + power-domains = <&mdlc_perw R8A78000_MDLC_PD_APL 0x43>; + resets = <&mdlc_perw 0x43>; status = "disabled"; }; @@ -918,6 +927,8 @@ <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; + power-domains = <&mdlc_perw R8A78000_MDLC_PD_APL 0x54>; + resets = <&mdlc_perw 0x54>; status = "disabled"; }; @@ -930,6 +941,8 @@ <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; + power-domains = <&mdlc_perw R8A78000_MDLC_PD_APL 0x55>; + resets = <&mdlc_perw 0x55>; status = "disabled"; }; @@ -942,6 +955,8 @@ <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; + power-domains = <&mdlc_perw R8A78000_MDLC_PD_APL 0x56>; + resets = <&mdlc_perw 0x56>; status = "disabled"; }; @@ -954,6 +969,8 @@ <&cpg R8A78000_CPG_SGASYNCD4_PERW_BUS>, <&scif_clk>; clock-names = "fck", "brg_int", "scif_clk"; + power-domains = <&mdlc_perw R8A78000_MDLC_PD_APL 0x57>; + resets = <&mdlc_perw 0x57>; status = "disabled"; }; @@ -974,6 +991,230 @@ #clock-cells = <1>; bootph-all; }; + + mdlc_aon: system-controller@c1338000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc1338000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_cmnn: system-controller@ca410000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xca410000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_cmns: system-controller@ca510000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xca510000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_ddr0: system-controller@e8000000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xe8000000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_ddr1: system-controller@e8080000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xe8080000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_ddr2: system-controller@e8100000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xe8100000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_ddr3: system-controller@e8180000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xe8180000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_ddr4: system-controller@e8200000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xe8200000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_ddr5: system-controller@e8280000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xe8280000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_ddr6: system-controller@e8300000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xe8300000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_ddr7: system-controller@e8380000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xe8380000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_dsp: system-controller@cbe90000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xcbe90000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_gpc: system-controller@cb510000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xcb510000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_hscn: system-controller@c9c90000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc9c90000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_hscs: system-controller@de200000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xde200000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_imn: system-controller@c1990000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc1990000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_ims: system-controller@c1d90000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc1d90000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_mm: system-controller@e9980000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xe9980000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_npu0: system-controller@d2c30000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xd2c30000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_npu1: system-controller@d6c30000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xd6c30000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_pere: system-controller@c08f0000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc08f0000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_perw: system-controller@c05d0000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc05d0000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_rt: system-controller@19440000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0x19440000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_scp: system-controller@c1330000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc1330000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_top: system-controller@c6480000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc6480000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_vio: system-controller@c5000000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc5000000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_vipn: system-controller@c3060000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc3060000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; + + mdlc_vips: system-controller@c3460000 { + compatible = "renesas,r8a78000-mdlc"; + reg = <0 0xc3460000 0 0x1000>; + #power-domain-cells = <2>; + #reset-cells = <1>; + bootph-all; + }; }; timer { From 7fe73a3a3f2c371803dd7b9aa697a93191854112 Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Sun, 9 Aug 2026 22:03:12 +0200 Subject: [PATCH 447/857] arm64: dts: renesas: r8a779g0: Add GICv3 ITS and update PCIe nodes This SoC implements GIC600 with GICv3 ITS and PCIe host mode on this SoC can use it. Add GIC ITS node into GIC node, update interrupt-map and add msi-map into PCIe controller node. The GIC ITS does have master interface to issue transactions to RAM. The interface does support cacheable transactions, however, it does not support shareable attribute, because the AXI port signals are tied to inactive in this implementation. Therefore, add "dma-noncoherent" DT property into the GIC ITS subnode. The GIC redistributor does not have cacheable/shareable, therefore add "dma-noncoherent" DT property into the GIC node. Co-developed-by: Yoshihiro Shimoda Signed-off-by: Yoshihiro Shimoda Signed-off-by: Marek Vasut Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260809200429.843212-1-marek.vasut+renesas@mailbox.org Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r8a779g0.dtsi | 31 ++++++++++++++++------- 1 file changed, 22 insertions(+), 9 deletions(-) diff --git a/arch/arm64/boot/dts/renesas/r8a779g0.dtsi b/arch/arm64/boot/dts/renesas/r8a779g0.dtsi index 8a291447b90450..42b20ddcbeddba 100644 --- a/arch/arm64/boot/dts/renesas/r8a779g0.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a779g0.dtsi @@ -809,6 +809,7 @@ resets = <&cpg 624>; reset-names = "pwr"; max-link-speed = <4>; + msi-parent = <&its>; num-lanes = <2>; #address-cells = <3>; #size-cells = <2>; @@ -819,10 +820,10 @@ dma-ranges = <0x42000000 0 0x00000000 0 0x00000000 1 0x00000000>; #interrupt-cells = <1>; interrupt-map-mask = <0 0 0 7>; - interrupt-map = <0 0 0 1 &gic GIC_SPI 449 IRQ_TYPE_LEVEL_HIGH>, - <0 0 0 2 &gic GIC_SPI 449 IRQ_TYPE_LEVEL_HIGH>, - <0 0 0 3 &gic GIC_SPI 449 IRQ_TYPE_LEVEL_HIGH>, - <0 0 0 4 &gic GIC_SPI 449 IRQ_TYPE_LEVEL_HIGH>; + interrupt-map = <0 0 0 1 &gic 0 0 GIC_SPI 449 IRQ_TYPE_LEVEL_HIGH>, + <0 0 0 2 &gic 0 0 GIC_SPI 449 IRQ_TYPE_LEVEL_HIGH>, + <0 0 0 3 &gic 0 0 GIC_SPI 449 IRQ_TYPE_LEVEL_HIGH>, + <0 0 0 4 &gic 0 0 GIC_SPI 449 IRQ_TYPE_LEVEL_HIGH>; snps,enable-cdm-check; status = "disabled"; @@ -856,6 +857,7 @@ resets = <&cpg 625>; reset-names = "pwr"; max-link-speed = <4>; + msi-parent = <&its>; num-lanes = <2>; #address-cells = <3>; #size-cells = <2>; @@ -866,10 +868,10 @@ dma-ranges = <0x42000000 0 0x00000000 0 0x00000000 1 0x00000000>; #interrupt-cells = <1>; interrupt-map-mask = <0 0 0 7>; - interrupt-map = <0 0 0 1 &gic GIC_SPI 456 IRQ_TYPE_LEVEL_HIGH>, - <0 0 0 2 &gic GIC_SPI 456 IRQ_TYPE_LEVEL_HIGH>, - <0 0 0 3 &gic GIC_SPI 456 IRQ_TYPE_LEVEL_HIGH>, - <0 0 0 4 &gic GIC_SPI 456 IRQ_TYPE_LEVEL_HIGH>; + interrupt-map = <0 0 0 1 &gic 0 0 GIC_SPI 456 IRQ_TYPE_LEVEL_HIGH>, + <0 0 0 2 &gic 0 0 GIC_SPI 456 IRQ_TYPE_LEVEL_HIGH>, + <0 0 0 3 &gic 0 0 GIC_SPI 456 IRQ_TYPE_LEVEL_HIGH>, + <0 0 0 4 &gic 0 0 GIC_SPI 456 IRQ_TYPE_LEVEL_HIGH>; snps,enable-cdm-check; status = "disabled"; @@ -2148,11 +2150,22 @@ gic: interrupt-controller@f1000000 { compatible = "arm,gic-v3"; #interrupt-cells = <3>; - #address-cells = <0>; + #address-cells = <2>; + #size-cells = <2>; interrupt-controller; reg = <0x0 0xf1000000 0 0x20000>, <0x0 0xf1060000 0 0x110000>; interrupts = ; + dma-noncoherent; + + ranges = <0x0 0x0 0x0 0xf1000000 0x0 0x200000>; + + its: msi-controller@40000 { + compatible = "arm,gic-v3-its"; + reg = <0x0 0x40000 0x0 0x20000>; + dma-noncoherent; + msi-controller; + }; }; csi40: csi2@fe500000 { From 478f85746e89ac1af08bbbc7bde5d84d4116254e Mon Sep 17 00:00:00 2001 From: Wolfram Sang Date: Mon, 17 Aug 2026 12:30:28 +0200 Subject: [PATCH 448/857] ARM: dts: renesas: r9a06g032-rzn1d400-eb: Enable GPIOs on CN12 CN12 offers some GPIOs independently of switch settings. Add the nodes. Signed-off-by: Wolfram Sang Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260817103255.49565-2-wsa+renesas@sang-engineering.com Signed-off-by: Geert Uytterhoeven --- .../arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts b/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts index 303b5926d8d173..eb72955047155c 100644 --- a/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts +++ b/arch/arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts @@ -53,6 +53,11 @@ }; }; +&gpio2b { + pinctrl-0 = <&pins_cn12>; + pinctrl-names = "default"; +}; + &i2c2 { /* Sensors are different across revisions. All are LM75B compatible */ sensor@49 { @@ -81,6 +86,15 @@ }; &pinctrl { + pins_cn12: pins-cn12 { + pinmux = , + , + , + , + , + ; + }; + pins_eth0: pins-eth0 { pinmux = , , From b7c724aaaaf22c7d6048409769968be0596a8363 Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Wed, 19 Aug 2026 19:38:28 +0100 Subject: [PATCH 449/857] arm64: dts: renesas: rzt2h-n2h-evk: Use consistent switch notation The switch comments in the board DT files use inconsistent notation, with some references written as "SWx-y" and others as "SWx[y]" (and "DSWx-y"/ DSWx[y] for the DIP switches on the RZ/N2H EVK). Standardise on the "SWx[y]" / "DSWx[y]" form throughout both files so switch references use a consistent notation. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260819183829.3953529-2-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- .../dts/renesas/r9a09g077m44-rzt2h-evk.dts | 18 ++++++++--------- .../dts/renesas/r9a09g087m44-rzn2h-evk.dts | 20 +++++++++---------- 2 files changed, 19 insertions(+), 19 deletions(-) diff --git a/arch/arm64/boot/dts/renesas/r9a09g077m44-rzt2h-evk.dts b/arch/arm64/boot/dts/renesas/r9a09g077m44-rzt2h-evk.dts index 572be8aa6d7ef5..d21367cef534e2 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g077m44-rzt2h-evk.dts +++ b/arch/arm64/boot/dts/renesas/r9a09g077m44-rzt2h-evk.dts @@ -69,7 +69,7 @@ compatible = "gpio-keys"; #if (!SD1_MICRO_SD) - /* SW2-3: OFF */ + /* SW2[3]: OFF */ key-1 { interrupts-extended = <&pinctrl RZT2H_GPIO(8, 6) IRQ_TYPE_EDGE_FALLING>; linux,code = ; @@ -100,7 +100,7 @@ compatible = "gpio-leds"; led-0 { - /* SW8-9: ON, SW8-10: OFF */ + /* SW8[9]: ON, SW8[10]: OFF */ gpios = <&pinctrl RZT2H_GPIO(23, 1) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -108,7 +108,7 @@ }; led-1 { - /* SW5-1: OFF, SW5-2: ON */ + /* SW5[1]: OFF, SW5[2]: ON */ gpios = <&pinctrl RZT2H_GPIO(32, 2) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -124,7 +124,7 @@ #if (!SD1_MICRO_SD) led-3 { - /* SW2-3: OFF */ + /* SW2[3]: OFF */ gpios = <&pinctrl RZT2H_GPIO(8, 5) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -133,7 +133,7 @@ #endif led-4 { - /* SW8-3: ON, SW8-4: OFF */ + /* SW8[3]: ON, SW8[4]: OFF */ gpios = <&pinctrl RZT2H_GPIO(18, 0) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -141,7 +141,7 @@ }; led-5 { - /* SW8-1: ON, SW8-2: OFF */ + /* SW8[1]: ON, SW8[2]: OFF */ gpios = <&pinctrl RZT2H_GPIO(18, 1) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -149,7 +149,7 @@ }; led-6 { - /* SW5-9: OFF, SW5-10: ON */ + /* SW5[9]: OFF, SW5[10]: ON */ gpios = <&pinctrl RZT2H_GPIO(22, 7) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -157,7 +157,7 @@ }; led-7 { - /* SW5-7: OFF, SW5-8: ON */ + /* SW5[7]: OFF, SW5[8]: ON */ gpios = <&pinctrl RZT2H_GPIO(23, 0) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -165,7 +165,7 @@ }; led-8 { - /* SW7-5: OFF, SW7-6: ON */ + /* SW7[5]: OFF, SW7[6]: ON */ gpios = <&pinctrl RZT2H_GPIO(23, 5) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; diff --git a/arch/arm64/boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts b/arch/arm64/boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts index 4e57d4fe195ca0..76e30a6ecb8bb6 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts +++ b/arch/arm64/boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts @@ -120,7 +120,7 @@ compatible = "gpio-leds"; led-3 { - /* DSW18-7: ON, DSW18-8: OFF */ + /* DSW18[7]: ON, DSW18[8]: OFF */ gpios = <&pinctrl RZT2H_GPIO(31, 6) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -128,7 +128,7 @@ }; led-4 { - /* DSW18-9: ON, DSW18-10: OFF */ + /* DSW18[9]: ON, DSW18[10]: OFF */ gpios = <&pinctrl RZT2H_GPIO(18, 1) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -136,7 +136,7 @@ }; led-5 { - /* DSW18-1: ON, DSW18-2: OFF */ + /* DSW18[1]: ON, DSW18[2]: OFF */ gpios = <&pinctrl RZT2H_GPIO(22, 7) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -144,7 +144,7 @@ }; led-6 { - /* DSW18-3: ON, DSW18-4: OFF */ + /* DSW18[3]: ON, DSW18[4]: OFF */ gpios = <&pinctrl RZT2H_GPIO(23, 0) GPIO_ACTIVE_HIGH>; color = ; function = LED_FUNCTION_DEBUG; @@ -153,8 +153,8 @@ led-7 { /* - * DSW18-5: ON, DSW18-6: OFF - * DSW19-3: OFF, DSW19-4: ON + * DSW18[5]: ON, DSW18[6]: OFF + * DSW19[3]: OFF, DSW19[4]: ON */ gpios = <&pinctrl RZT2H_GPIO(14, 3) GPIO_ACTIVE_HIGH>; color = ; @@ -166,7 +166,7 @@ led-8 { /* * USER_LED0 - * DSW15-8: OFF, DSW15-9: OFF, DSW15-10: ON + * DSW15[8]: OFF, DSW15[9]: OFF, DSW15[10]: ON */ gpios = <&pinctrl RZT2H_GPIO(14, 6) GPIO_ACTIVE_HIGH>; color = ; @@ -179,7 +179,7 @@ led-9 { /* * USER_LED1 - * DSW15-5: OFF, DSW15-6: ON + * DSW15[5]: OFF, DSW15[6]: ON */ gpios = <&pinctrl RZT2H_GPIO(14, 7) GPIO_ACTIVE_HIGH>; color = ; @@ -191,7 +191,7 @@ led-10 { /* * USER_LED2 - * DSW17-3: OFF, DSW17-4: ON + * DSW17[3]: OFF, DSW17[4]: ON */ gpios = <&pinctrl RZT2H_GPIO(2, 7) GPIO_ACTIVE_HIGH>; color = ; @@ -202,7 +202,7 @@ led-11 { /* * USER_LED3 - * DSW17-1: OFF, DSW17-2: ON + * DSW17[1]: OFF, DSW17[2]: ON */ gpios = <&pinctrl RZT2H_GPIO(3, 0) GPIO_ACTIVE_HIGH>; color = ; From 39233ecd4aa8b4c2b8329d26d1b0bd8bf7b7bb8b Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Wed, 19 Aug 2026 19:38:29 +0100 Subject: [PATCH 450/857] arm64: dts: renesas: Add LCDC overlays for RZ/T2H and RZ/N2H EVKs Add DT overlay support for enabling the DU/LCDC pipeline on the RZ/T2H and RZ/N2H evaluation kits when fitted with an ADV7513 HDMI transmitter on CN15/CN20. Share the common ADV7513 and LCDC pin configuration between the two overlays. Disable board functions whose pins are reassigned to LCDC, including the affected LEDs, SDHI0 on the RZ/T2H EVK, and I2C0 on the RZ/N2H EVK. Move the LED8 and LED9 preprocessor conditionals inside the node definitions so that the nodes remain present in the base DTS. This allows the LCDC overlays to reference and disable the LEDs when their pins are reassigned to display functions. Configure the LCDC data and clock pins with slew-rate setting 1 as recommended by the hardware manual. Use slew-rate setting 0 for the HSYNC, VSYNC and DE pins, as testing with the ADV7513 showed unstable display output and visible flicker when these synchronization signals were configured for fast slew rate. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260819183829.3953529-3-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/Makefile | 6 ++ .../renesas/r9a09g077m44-evk-cn15-lcdc.dtso | 53 ++++++++++++++ .../renesas/r9a09g087m44-evk-cn20-lcdc.dtso | 63 ++++++++++++++++ .../dts/renesas/r9a09g087m44-rzn2h-evk.dts | 12 ++-- .../dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi | 72 +++++++++++++++++++ 5 files changed, 202 insertions(+), 4 deletions(-) create mode 100644 arch/arm64/boot/dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso create mode 100644 arch/arm64/boot/dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso create mode 100644 arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi diff --git a/arch/arm64/boot/dts/renesas/Makefile b/arch/arm64/boot/dts/renesas/Makefile index 8bf155badd111d..e4a7d7ab5b0e99 100644 --- a/arch/arm64/boot/dts/renesas/Makefile +++ b/arch/arm64/boot/dts/renesas/Makefile @@ -216,8 +216,14 @@ r9a09g057h48-kakip-pixpaper-dtbs := r9a09g057h48-kakip.dtb r9a09g057h48-kakip-pi dtb-$(CONFIG_ARCH_R9A09G057) += r9a09g057h48-kakip-pixpaper.dtb dtb-$(CONFIG_ARCH_R9A09G077) += r9a09g077m44-rzt2h-evk.dtb +dtb-$(CONFIG_ARCH_R9A09G077) += r9a09g077m44-evk-cn15-lcdc.dtbo +r9a09g077m44-rzt2h-evk-cn15-lcdc-dtbs := r9a09g077m44-rzt2h-evk.dtb r9a09g077m44-evk-cn15-lcdc.dtbo +dtb-$(CONFIG_ARCH_R9A09G077) += r9a09g077m44-rzt2h-evk-cn15-lcdc.dtb dtb-$(CONFIG_ARCH_R9A09G087) += r9a09g087m44-rzn2h-evk.dtb +dtb-$(CONFIG_ARCH_R9A09G087) += r9a09g087m44-evk-cn20-lcdc.dtbo +r9a09g087m44-rzn2h-evk-cn20-lcdc-dtbs := r9a09g087m44-rzn2h-evk.dtb r9a09g087m44-evk-cn20-lcdc.dtbo +dtb-$(CONFIG_ARCH_R9A09G087) += r9a09g087m44-rzn2h-evk-cn20-lcdc.dtb dtb-$(CONFIG_ARCH_RCAR_GEN3) += draak-ebisu-panel-aa104xd12.dtbo dtb-$(CONFIG_ARCH_RCAR_GEN3) += salvator-panel-aa104xd12.dtbo diff --git a/arch/arm64/boot/dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso b/arch/arm64/boot/dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso new file mode 100644 index 00000000000000..aaddfc1154d54e --- /dev/null +++ b/arch/arm64/boot/dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso @@ -0,0 +1,53 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * DT overlay for the RZ/T2H EVK with ADV7513 transmitter + * connected to DU enabled. + * + * Copyright (C) 2026 Renesas Electronics Corp. + */ + +/dts-v1/; +/plugin/; + +#include + +#define RZT2H_IRQ8 24 + +/* + * RZ/T2H LCDC configuration: + * ---------------------------------------------------------- + * Function Pin SW Setting + * ---------------------------------------------------------- + * LCDC_DATG0 P11_0, SW6[3]: OFF, SW6[4]: ON, SW6[5]: OFF + * LCDC_DATB1 P18_0, SW8[3]: OFF, SW8[4]: ON + * LCDC_DATB2 P18_1, SW8[1]: OFF, SW8[2]: ON + * IRQ8 P22_6, SW2[1]: ON, SW2[2]: OFF + */ + +&{/leds/led-4} { + /* P18_0 is used for DU function LCDC_DATB1. */ + status = "disabled"; +}; + +&{/leds/led-5} { + /* P18_1 is used for DU function LCDC_DATB2. */ + status = "disabled"; +}; + +/* + * Disable SDHI0 as SW2 settings for eMMC/SD card conflict with DU pin + * settings. + */ +&sdhi0 { + status = "disabled"; +}; + +#include "rzt2h-n2h-evk-du-adv7513.dtsi" + +&adv7513 { + interrupts-extended = <&icu RZT2H_IRQ8 IRQ_TYPE_LEVEL_LOW>; +}; + +&adv7513_irq_pins { + pinmux = ; /* IRQ8 */ +}; diff --git a/arch/arm64/boot/dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso b/arch/arm64/boot/dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso new file mode 100644 index 00000000000000..9c9e4fc80363e6 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso @@ -0,0 +1,63 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * DT overlay for the RZ/N2H EVK with ADV7513 transmitter + * connected to DU enabled. + * + * Copyright (C) 2026 Renesas Electronics Corp. + */ + +/dts-v1/; +/plugin/; + +#include + +#define RZN2H_IRQ5 21 + +/* + * RZ/N2H LCDC configuration: + * ---------------------------------------------------------- + * Function Pin SW Setting + * ---------------------------------------------------------- + * LCDC_DATG0 P11_0, DSW12[3]: ON, DSW12[4]: OFF + * LCDC_DATG3 P14_3, DSW18[5]: OFF, DSW18[6]: ON, DSW19[3]: OFF, DSW19[4]: ON + * LCDC_DATG6 P14_6, DSW15[8]: ON, DSW15[9]: OFF, DSW15[10]: OFF + * LCDC_DATB2 P18_1, DSW18[9]: OFF, DSW18[10]: ON + * I2C_SCL1 P03_3, DSW7[1]: ON, DSW7[2]: OFF + * I2C_SDA1 P03_4, DSW7[3]: ON, DSW7[4]: OFF + * ------------------------------------------------ + */ + +&{/keys/key-1} { + /* P18_2 is used for DU function LCDC_DATB3. */ + status = "disabled"; +}; + +&{/leds/led-4} { + /* P18_1 is used for DU function LCDC_DATB2. */ + status = "disabled"; +}; + +&{/leds/led-7} { + /* P14_3 is used for DU function LCDC_DATG3. */ + status = "disabled"; +}; + +&{/leds/led-8} { + /* P14_6 is used for DU function LCDC_DATG6. */ + status = "disabled"; +}; + +&i2c0 { + /* P14_6 is used for DU function LCDC_DATG6. */ + status = "disabled"; +}; + +#include "rzt2h-n2h-evk-du-adv7513.dtsi" + +&adv7513 { + interrupts-extended = <&icu RZN2H_IRQ5 IRQ_TYPE_LEVEL_LOW>; +}; + +&adv7513_irq_pins { + pinmux = ; /* IRQ5 */ +}; diff --git a/arch/arm64/boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts b/arch/arm64/boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts index 76e30a6ecb8bb6..9d54e765095771 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts +++ b/arch/arm64/boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts @@ -162,8 +162,8 @@ function-enumerator = <8>; }; -#if LED8 led-8 { +#if LED8 /* * USER_LED0 * DSW15[8]: OFF, DSW15[9]: OFF, DSW15[10]: ON @@ -172,11 +172,13 @@ color = ; function = LED_FUNCTION_DEBUG; function-enumerator = <0>; - }; +#else + status = "disabled"; #endif + }; -#if LED9 led-9 { +#if LED9 /* * USER_LED1 * DSW15[5]: OFF, DSW15[6]: ON @@ -185,8 +187,10 @@ color = ; function = LED_FUNCTION_DEBUG; function-enumerator = <1>; - }; +#else + status = "disabled"; #endif + }; led-10 { /* diff --git a/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi b/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi new file mode 100644 index 00000000000000..d0242dc5516100 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi @@ -0,0 +1,72 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * DT overlay common parts for the RZ/{T2H/N2H} EVKs with ADV7513 + * transmitter connected to DU enabled. + * + * Copyright (C) 2026 Renesas Electronics Corp. + */ + +#include + +#define ADV7513_PARENT_I2C i2c1 +#include "rz-smarc-du-adv7513.dtsi" + +&pinctrl { + du_pins: du-group { + du-data-pins { + pinmux = , /* LCDC_DATR0 */ + , /* LCDC_DATR1 */ + , /* LCDC_DATR2 */ + , /* LCDC_DATR3 */ + , /* LCDC_DATR4 */ + , /* LCDC_DATR5 */ + , /* LCDC_DATR6 */ + , /* LCDC_DATR7 */ + , /* LCDC_DATG0 */ + , /* LCDC_DATG1 */ + , /* LCDC_DATG2 */ + , /* LCDC_DATG3 */ + , /* LCDC_DATG4 */ + , /* LCDC_DATG5 */ + , /* LCDC_DATG6 */ + , /* LCDC_DATG7 */ + , /* LCDC_DATB0 */ + , /* LCDC_DATB1 */ + , /* LCDC_DATB2 */ + , /* LCDC_DATB3 */ + , /* LCDC_DATB4 */ + , /* LCDC_DATB5 */ + , /* LCDC_DATB6 */ + ; /* LCDC_DATB7 */ + drive-strength-microamp = <11800>; + slew-rate = <1>; + }; + + du-clk-pins { + pinmux = ; /* LCDC_CLK */ + drive-strength-microamp = <11800>; + slew-rate = <1>; + }; + + du-sync-pins { + pinmux = , /* LCDC_HSYNC */ + ; /* LCDC_VSYNC */ + drive-strength-microamp = <11800>; + slew-rate = <0>; + }; + + du-de-pins { + pinmux = ; /* LCDC_DE */ + drive-strength-microamp = <11800>; + slew-rate = <0>; + }; + }; + + adv7513_irq_pins: adv7513-irq-pins { + }; +}; + +&adv7513 { + pinctrl-0 = <&adv7513_irq_pins>; + pinctrl-names = "default"; +}; From 3861cd02818af69853d90cbf3d73ca58fd1230e1 Mon Sep 17 00:00:00 2001 From: Krzysztof Kozlowski Date: Sat, 22 Aug 2026 09:58:14 +0200 Subject: [PATCH 451/857] arm64: dts: renesas: Use hyphens in node names DTS coding style encourages to use hyphens instead of underscores in node names. DTC W=2 warns about underscores. Improve existing code because apparently people copy it instead of following DTS coding style for new submissions. Semi-manual validation by checking if no new nodes appeared: $ for i in dts-old/*/*dtb dts-old/*/*/*dtb; do echo $i; fdtdump ${i} > ${i}.fdt ; fdtdump dts-new/${i#dts-old/} > dts-new/${i#dts-old/}.fdt ; diff -ubB ${i}.fdt dts-new/${i#dts-old/}.fdt ; done | grep -v ' {' Signed-off-by: Krzysztof Kozlowski Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260822075813.199336-2-krzysztof.kozlowski@oss.qualcomm.com Signed-off-by: Geert Uytterhoeven --- .../aistarvision-mipi-adapter-2.1.dtsi | 2 +- .../dts/renesas/beacon-renesom-baseboard.dtsi | 6 +++--- .../boot/dts/renesas/beacon-renesom-som.dtsi | 10 +++++----- .../arm64/boot/dts/renesas/condor-common.dtsi | 2 +- arch/arm64/boot/dts/renesas/draak.dtsi | 2 +- arch/arm64/boot/dts/renesas/ebisu.dtsi | 6 +++--- .../boot/dts/renesas/gray-hawk-single.dtsi | 8 ++++---- .../arm64/boot/dts/renesas/hihope-common.dtsi | 4 ++-- arch/arm64/boot/dts/renesas/hihope-rev2.dtsi | 8 ++++---- arch/arm64/boot/dts/renesas/hihope-rev4.dtsi | 4 ++-- .../boot/dts/renesas/hihope-rzg2-ex.dtsi | 4 ++-- arch/arm64/boot/dts/renesas/r8a774a1.dtsi | 14 ++++++------- arch/arm64/boot/dts/renesas/r8a774b1.dtsi | 12 +++++------ .../boot/dts/renesas/r8a774c0-cat874.dts | 4 ++-- arch/arm64/boot/dts/renesas/r8a774c0.dtsi | 12 +++++------ arch/arm64/boot/dts/renesas/r8a774e1.dtsi | 14 ++++++------- arch/arm64/boot/dts/renesas/r8a77951.dtsi | 14 ++++++------- arch/arm64/boot/dts/renesas/r8a77960.dtsi | 14 ++++++------- arch/arm64/boot/dts/renesas/r8a77961.dtsi | 14 ++++++------- arch/arm64/boot/dts/renesas/r8a77965.dtsi | 12 +++++------ .../arm64/boot/dts/renesas/r8a77970-eagle.dts | 2 +- .../arm64/boot/dts/renesas/r8a77970-v3msk.dts | 2 +- arch/arm64/boot/dts/renesas/r8a77970.dtsi | 2 +- .../arm64/boot/dts/renesas/r8a77980-v3hsk.dts | 2 +- arch/arm64/boot/dts/renesas/r8a77980.dtsi | 4 ++-- arch/arm64/boot/dts/renesas/r8a77990.dtsi | 10 +++++----- arch/arm64/boot/dts/renesas/r8a77995.dtsi | 6 +++--- .../boot/dts/renesas/r8a779a0-falcon-cpu.dtsi | 2 +- .../boot/dts/renesas/r8a779a0-falcon.dts | 4 ++-- arch/arm64/boot/dts/renesas/r8a779a0.dtsi | 2 +- .../boot/dts/renesas/r8a779f0-spider-cpu.dtsi | 2 +- arch/arm64/boot/dts/renesas/r8a779f0.dtsi | 2 +- arch/arm64/boot/dts/renesas/r8a779f4-s4sk.dts | 2 +- arch/arm64/boot/dts/renesas/r8a779g0.dtsi | 4 ++-- arch/arm64/boot/dts/renesas/r8a779h0.dtsi | 2 +- .../arm64/boot/dts/renesas/r8a779md-geist.dts | 10 +++++----- .../dts/renesas/r9a09g057h44-rzv2h-evk.dts | 6 +++--- .../dts/renesas/rzg2l-smarc-pinfunction.dtsi | 16 +++++++-------- .../boot/dts/renesas/rzg2l-smarc-som.dtsi | 20 +++++++++---------- arch/arm64/boot/dts/renesas/rzg2l-smarc.dtsi | 2 +- .../dts/renesas/rzg2lc-smarc-pinfunction.dtsi | 16 +++++++-------- .../boot/dts/renesas/rzg2lc-smarc-som.dtsi | 20 +++++++++---------- arch/arm64/boot/dts/renesas/rzg2lc-smarc.dtsi | 2 +- .../dts/renesas/rzg2ul-smarc-pinfunction.dtsi | 16 +++++++-------- .../boot/dts/renesas/rzg2ul-smarc-som.dtsi | 20 +++++++++---------- .../boot/dts/renesas/rzg3s-smarc-som.dtsi | 4 ++-- .../boot/dts/renesas/salvator-common.dtsi | 12 +++++------ arch/arm64/boot/dts/renesas/salvator-xs.dtsi | 2 +- arch/arm64/boot/dts/renesas/ulcb-kf.dtsi | 2 +- arch/arm64/boot/dts/renesas/ulcb.dtsi | 8 ++++---- .../dts/renesas/white-hawk-cpu-common.dtsi | 6 +++--- 51 files changed, 188 insertions(+), 188 deletions(-) diff --git a/arch/arm64/boot/dts/renesas/aistarvision-mipi-adapter-2.1.dtsi b/arch/arm64/boot/dts/renesas/aistarvision-mipi-adapter-2.1.dtsi index 529388f6bf2b70..c5db0516ae83f6 100644 --- a/arch/arm64/boot/dts/renesas/aistarvision-mipi-adapter-2.1.dtsi +++ b/arch/arm64/boot/dts/renesas/aistarvision-mipi-adapter-2.1.dtsi @@ -54,7 +54,7 @@ regulator-always-on; }; - osc25250_clk: osc25250_clk { + osc25250_clk: clock-24000000 { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <24000000>; diff --git a/arch/arm64/boot/dts/renesas/beacon-renesom-baseboard.dtsi b/arch/arm64/boot/dts/renesas/beacon-renesom-baseboard.dtsi index 62ab0a3776e75a..963253866d717d 100644 --- a/arch/arm64/boot/dts/renesas/beacon-renesom-baseboard.dtsi +++ b/arch/arm64/boot/dts/renesas/beacon-renesom-baseboard.dtsi @@ -142,7 +142,7 @@ startup-delay-us = <100000>; }; - sound_card { + sound-card { compatible = "audio-graph-card"; label = "rcar-sound"; dais = <&rsnd_port0>, <&rsnd_port1>; @@ -505,7 +505,7 @@ power-source = <3300>; }; - sdhi0_pins_uhs: sd0_uhs { + sdhi0_pins_uhs: sd0-uhs { groups = "sdhi0_data4", "sdhi0_ctrl"; function = "sdhi0"; power-source = <1800>; @@ -516,7 +516,7 @@ function = "ssi"; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { groups = "audio_clk_a_a", "audio_clk_b_a"; function = "audio_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/beacon-renesom-som.dtsi b/arch/arm64/boot/dts/renesas/beacon-renesom-som.dtsi index 8723e7a76c9ddf..88251314904209 100644 --- a/arch/arm64/boot/dts/renesas/beacon-renesom-som.dtsi +++ b/arch/arm64/boot/dts/renesas/beacon-renesom-som.dtsi @@ -13,7 +13,7 @@ reg = <0x0 0x48000000 0x0 0x78000000>; }; - osc_32k: osc_32k { + osc_32k: clock-32768 { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <32768>; @@ -38,7 +38,7 @@ regulator-always-on; }; - wlan_pwrseq: wlan_pwrseq { + wlan_pwrseq: wlan-pwrseq { compatible = "mmc-pwrseq-simple"; reset-gpios = <&pca9654 1 GPIO_ACTIVE_LOW>; clocks = <&osc_32k>; @@ -211,12 +211,12 @@ function = "avb"; }; - pins_mdio { + pins-mdio { groups = "avb_mdio"; drive-strength = <24>; }; - pins_mii_tx { + pins-mii-tx { pins = "PIN_AVB_TX_CTL", "PIN_AVB_TXC", "PIN_AVB_TD0", "PIN_AVB_TD1", "PIN_AVB_TD2", "PIN_AVB_TD3"; drive-strength = <12>; @@ -253,7 +253,7 @@ function = "scif5"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk_a"; function = "scif_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/condor-common.dtsi b/arch/arm64/boot/dts/renesas/condor-common.dtsi index 9d55509b00b150..953c5bb491ecd1 100644 --- a/arch/arm64/boot/dts/renesas/condor-common.dtsi +++ b/arch/arm64/boot/dts/renesas/condor-common.dtsi @@ -472,7 +472,7 @@ function = "scif0"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk_b"; function = "scif_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/draak.dtsi b/arch/arm64/boot/dts/renesas/draak.dtsi index f2f25fe5d77853..33046295f2d66f 100644 --- a/arch/arm64/boot/dts/renesas/draak.dtsi +++ b/arch/arm64/boot/dts/renesas/draak.dtsi @@ -561,7 +561,7 @@ power-source = <1800>; }; - sdhi2_pins_uhs: sd2_uhs { + sdhi2_pins_uhs: sd2-uhs { groups = "mmc_data8", "mmc_ctrl"; function = "mmc"; power-source = <1800>; diff --git a/arch/arm64/boot/dts/renesas/ebisu.dtsi b/arch/arm64/boot/dts/renesas/ebisu.dtsi index 4b3775afcb0178..842e1a57e22828 100644 --- a/arch/arm64/boot/dts/renesas/ebisu.dtsi +++ b/arch/arm64/boot/dts/renesas/ebisu.dtsi @@ -674,7 +674,7 @@ power-source = <3300>; }; - sdhi0_pins_uhs: sd0_uhs { + sdhi0_pins_uhs: sd0-uhs { groups = "sdhi0_data4", "sdhi0_ctrl"; function = "sdhi0"; power-source = <1800>; @@ -686,7 +686,7 @@ power-source = <3300>; }; - sdhi1_pins_uhs: sd1_uhs { + sdhi1_pins_uhs: sd1-uhs { groups = "sdhi1_data4", "sdhi1_ctrl"; function = "sdhi1"; power-source = <1800>; @@ -698,7 +698,7 @@ power-source = <1800>; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { groups = "audio_clk_a", "audio_clk_b_a", "audio_clk_c_a", "audio_clkout_a", "audio_clkout1_a"; function = "audio_clk"; diff --git a/arch/arm64/boot/dts/renesas/gray-hawk-single.dtsi b/arch/arm64/boot/dts/renesas/gray-hawk-single.dtsi index 274493720b14e6..da6133375727ca 100644 --- a/arch/arm64/boot/dts/renesas/gray-hawk-single.dtsi +++ b/arch/arm64/boot/dts/renesas/gray-hawk-single.dtsi @@ -595,12 +595,12 @@ function = "avb0"; }; - pins_mdio { + pins-mdio { groups = "avb0_mdio"; drive-strength = <21>; }; - pins_mii { + pins-mii { groups = "avb0_rgmii"; drive-strength = <21>; }; @@ -696,7 +696,7 @@ function = "i2c3"; }; - irq0_pins: irq0_pins { + irq0_pins: irq0 { groups = "intc_ex_irq0_a"; function = "intc_ex"; }; @@ -727,7 +727,7 @@ function = "scif_clk2"; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { groups = "audio_clkin", "audio_clkout"; function = "audio_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/hihope-common.dtsi b/arch/arm64/boot/dts/renesas/hihope-common.dtsi index 4e78139d52f6c8..1999f97e0e606f 100644 --- a/arch/arm64/boot/dts/renesas/hihope-common.dtsi +++ b/arch/arm64/boot/dts/renesas/hihope-common.dtsi @@ -229,7 +229,7 @@ function = "scif2"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk_a"; function = "scif_clk"; }; @@ -240,7 +240,7 @@ power-source = <3300>; }; - sdhi0_pins_uhs: sd0_uhs { + sdhi0_pins_uhs: sd0-uhs { groups = "sdhi0_data4", "sdhi0_ctrl"; function = "sdhi0"; power-source = <1800>; diff --git a/arch/arm64/boot/dts/renesas/hihope-rev2.dtsi b/arch/arm64/boot/dts/renesas/hihope-rev2.dtsi index 25c55b32aafe5a..99ade271844a29 100644 --- a/arch/arm64/boot/dts/renesas/hihope-rev2.dtsi +++ b/arch/arm64/boot/dts/renesas/hihope-rev2.dtsi @@ -13,14 +13,14 @@ leds { compatible = "gpio-leds"; - bt_active_led { + bt-active-led { label = "blue:bt"; gpios = <&gpio7 0 GPIO_ACTIVE_HIGH>; linux,default-trigger = "hci0-power"; default-state = "off"; }; - wlan_active_led { + wlan-active-led { label = "yellow:wlan"; gpios = <&gpio7 1 GPIO_ACTIVE_HIGH>; linux,default-trigger = "phy0tx"; @@ -28,7 +28,7 @@ }; }; - wlan_en_reg: regulator-wlan_en { + wlan_en_reg: regulator-wlan-en { compatible = "regulator-fixed"; regulator-name = "wlan-en-regulator"; regulator-min-microvolt = <1800000>; @@ -57,7 +57,7 @@ }; &pfc { - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { groups = "audio_clk_a_a"; function = "audio_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/hihope-rev4.dtsi b/arch/arm64/boot/dts/renesas/hihope-rev4.dtsi index acce3c0452f4a4..a26a985ec48b63 100644 --- a/arch/arm64/boot/dts/renesas/hihope-rev4.dtsi +++ b/arch/arm64/boot/dts/renesas/hihope-rev4.dtsi @@ -20,7 +20,7 @@ clock-frequency = <12288000>; }; - wlan_en_reg: regulator-wlan_en { + wlan_en_reg: regulator-wlan-en { compatible = "regulator-fixed"; regulator-name = "wlan-en-regulator"; regulator-min-microvolt = <1800000>; @@ -68,7 +68,7 @@ function = "i2c2"; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { groups = "audio_clk_a_a", "audio_clk_b_a", "audio_clkout_a"; function = "audio_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi b/arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi index 58b787db0f24f4..0878526506561c 100644 --- a/arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi +++ b/arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi @@ -59,12 +59,12 @@ function = "avb"; }; - pins_mdio { + pins-mdio { groups = "avb_mdio"; drive-strength = <24>; }; - pins_mii_tx { + pins-mii-tx { pins = "PIN_AVB_TX_CTL", "PIN_AVB_TXC", "PIN_AVB_TD0", "PIN_AVB_TD1", "PIN_AVB_TD2", "PIN_AVB_TD3"; drive-strength = <12>; diff --git a/arch/arm64/boot/dts/renesas/r8a774a1.dtsi b/arch/arm64/boot/dts/renesas/r8a774a1.dtsi index e66d86db6e6c44..5d500505b3ff1f 100644 --- a/arch/arm64/boot/dts/renesas/r8a774a1.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a774a1.dtsi @@ -21,19 +21,19 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_c: audio_clk_c { + audio_clk_c: audio-clk-c { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -228,13 +228,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts = , , @@ -243,7 +243,7 @@ interrupt-affinity = <&a53_0>, <&a53_1>, <&a53_2>, <&a53_3>; }; - pmu_a57 { + pmu-a57 { compatible = "arm,cortex-a57-pmu"; interrupts= , ; @@ -2877,7 +2877,7 @@ clock-frequency = <0>; }; - usb_extal_clk: usb_extal { + usb_extal_clk: usb-extal { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; diff --git a/arch/arm64/boot/dts/renesas/r8a774b1.dtsi b/arch/arm64/boot/dts/renesas/r8a774b1.dtsi index 62c6703917db4e..e0d593c49a86a8 100644 --- a/arch/arm64/boot/dts/renesas/r8a774b1.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a774b1.dtsi @@ -21,19 +21,19 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_c: audio_clk_c { + audio_clk_c: audio-clk-c { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -121,13 +121,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a57 { + pmu-a57 { compatible = "arm,cortex-a57-pmu"; interrupts = , ; @@ -2748,7 +2748,7 @@ clock-frequency = <0>; }; - usb_extal_clk: usb_extal { + usb_extal_clk: usb-extal { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; diff --git a/arch/arm64/boot/dts/renesas/r8a774c0-cat874.dts b/arch/arm64/boot/dts/renesas/r8a774c0-cat874.dts index 57a281fc49775d..4be3f82a0460b5 100644 --- a/arch/arm64/boot/dts/renesas/r8a774c0-cat874.dts +++ b/arch/arm64/boot/dts/renesas/r8a774c0-cat874.dts @@ -322,7 +322,7 @@ power-source = <3300>; }; - sdhi0_pins_uhs: sd0_uhs { + sdhi0_pins_uhs: sd0-uhs { groups = "sdhi0_data4", "sdhi0_ctrl"; function = "sdhi0"; power-source = <1800>; @@ -334,7 +334,7 @@ power-source = <1800>; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { groups = "audio_clkout1_a"; function = "audio_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/r8a774c0.dtsi b/arch/arm64/boot/dts/renesas/r8a774c0.dtsi index 3858f4328e9619..37313f525a9323 100644 --- a/arch/arm64/boot/dts/renesas/r8a774c0.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a774c0.dtsi @@ -20,19 +20,19 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_c: audio_clk_c { + audio_clk_c: audio-clk-c { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -112,13 +112,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts= , ; @@ -2014,7 +2014,7 @@ clock-frequency = <0>; }; - usb_extal_clk: usb_extal { + usb_extal_clk: usb-extal { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; diff --git a/arch/arm64/boot/dts/renesas/r8a774e1.dtsi b/arch/arm64/boot/dts/renesas/r8a774e1.dtsi index 0ae9bb72d2dda5..335b94d3fec7db 100644 --- a/arch/arm64/boot/dts/renesas/r8a774e1.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a774e1.dtsi @@ -21,19 +21,19 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_c: audio_clk_c { + audio_clk_c: audio-clk-c { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -290,13 +290,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts = , , @@ -305,7 +305,7 @@ interrupt-affinity = <&a53_0>, <&a53_1>, <&a53_2>, <&a53_3>; }; - pmu_a57 { + pmu-a57 { compatible = "arm,cortex-a57-pmu"; interrupts = , , @@ -3011,7 +3011,7 @@ clock-frequency = <0>; }; - usb_extal_clk: usb_extal { + usb_extal_clk: usb-extal { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; diff --git a/arch/arm64/boot/dts/renesas/r8a77951.dtsi b/arch/arm64/boot/dts/renesas/r8a77951.dtsi index 59a0f2e1479d08..a08b1609bbe685 100644 --- a/arch/arm64/boot/dts/renesas/r8a77951.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a77951.dtsi @@ -25,19 +25,19 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_c: audio_clk_c { + audio_clk_c: audio-clk-c { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -305,13 +305,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts = , , @@ -323,7 +323,7 @@ <&a53_3>; }; - pmu_a57 { + pmu-a57 { compatible = "arm,cortex-a57-pmu"; interrupts = , , @@ -3520,7 +3520,7 @@ clock-frequency = <0>; }; - usb_extal_clk: usb_extal { + usb_extal_clk: usb-extal { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; diff --git a/arch/arm64/boot/dts/renesas/r8a77960.dtsi b/arch/arm64/boot/dts/renesas/r8a77960.dtsi index 4f9989b5e77a89..cdbb7f9556fb69 100644 --- a/arch/arm64/boot/dts/renesas/r8a77960.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a77960.dtsi @@ -20,19 +20,19 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_c: audio_clk_c { + audio_clk_c: audio-clk-c { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -277,13 +277,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts = , , @@ -292,7 +292,7 @@ interrupt-affinity = <&a53_0>, <&a53_1>, <&a53_2>, <&a53_3>; }; - pmu_a57 { + pmu-a57 { compatible = "arm,cortex-a57-pmu"; interrupts = , ; @@ -3135,7 +3135,7 @@ clock-frequency = <0>; }; - usb_extal_clk: usb_extal { + usb_extal_clk: usb-extal { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; diff --git a/arch/arm64/boot/dts/renesas/r8a77961.dtsi b/arch/arm64/boot/dts/renesas/r8a77961.dtsi index ad4491ba948f2f..38f34c5c9f65ea 100644 --- a/arch/arm64/boot/dts/renesas/r8a77961.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a77961.dtsi @@ -20,19 +20,19 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_c: audio_clk_c { + audio_clk_c: audio-clk-c { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -277,13 +277,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts = , , @@ -292,7 +292,7 @@ interrupt-affinity = <&a53_0>, <&a53_1>, <&a53_2>, <&a53_3>; }; - pmu_a57 { + pmu-a57 { compatible = "arm,cortex-a57-pmu"; interrupts = , ; @@ -2956,7 +2956,7 @@ clock-frequency = <0>; }; - usb_extal_clk: usb_extal { + usb_extal_clk: usb-extal { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; diff --git a/arch/arm64/boot/dts/renesas/r8a77965.dtsi b/arch/arm64/boot/dts/renesas/r8a77965.dtsi index 70708f5cf74671..52376c6de208ed 100644 --- a/arch/arm64/boot/dts/renesas/r8a77965.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a77965.dtsi @@ -25,19 +25,19 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_c: audio_clk_c { + audio_clk_c: audio-clk-c { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -156,13 +156,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a57 { + pmu-a57 { compatible = "arm,cortex-a57-pmu"; interrupts = , ; @@ -2964,7 +2964,7 @@ clock-frequency = <0>; }; - usb_extal_clk: usb_extal { + usb_extal_clk: usb-extal { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; diff --git a/arch/arm64/boot/dts/renesas/r8a77970-eagle.dts b/arch/arm64/boot/dts/renesas/r8a77970-eagle.dts index a2ad79ddf73db1..be0ee054fc5f7a 100644 --- a/arch/arm64/boot/dts/renesas/r8a77970-eagle.dts +++ b/arch/arm64/boot/dts/renesas/r8a77970-eagle.dts @@ -335,7 +335,7 @@ function = "scif0"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk_b"; function = "scif_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/r8a77970-v3msk.dts b/arch/arm64/boot/dts/renesas/r8a77970-v3msk.dts index 10c9a2e9ed18d3..f184fa93d2a43b 100644 --- a/arch/arm64/boot/dts/renesas/r8a77970-v3msk.dts +++ b/arch/arm64/boot/dts/renesas/r8a77970-v3msk.dts @@ -215,7 +215,7 @@ function = "i2c0"; }; - mmc_pins: mmc_3_3v { + mmc_pins: mmc-3-3v { groups = "mmc_data8", "mmc_ctrl"; function = "mmc"; power-source = <3300>; diff --git a/arch/arm64/boot/dts/renesas/r8a77970.dtsi b/arch/arm64/boot/dts/renesas/r8a77970.dtsi index f7f1f280fa0b63..3eb519b61fca7b 100644 --- a/arch/arm64/boot/dts/renesas/r8a77970.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a77970.dtsi @@ -72,7 +72,7 @@ bootph-all; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts = , ; diff --git a/arch/arm64/boot/dts/renesas/r8a77980-v3hsk.dts b/arch/arm64/boot/dts/renesas/r8a77980-v3hsk.dts index 52462e61b7194c..32b5ea7d1dc563 100644 --- a/arch/arm64/boot/dts/renesas/r8a77980-v3hsk.dts +++ b/arch/arm64/boot/dts/renesas/r8a77980-v3hsk.dts @@ -207,7 +207,7 @@ function = "scif0"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk_b"; function = "scif_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/r8a77980.dtsi b/arch/arm64/boot/dts/renesas/r8a77980.dtsi index 514dafe3447869..1df6d3db046cc9 100644 --- a/arch/arm64/boot/dts/renesas/r8a77980.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a77980.dtsi @@ -93,13 +93,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts = , , diff --git a/arch/arm64/boot/dts/renesas/r8a77990.dtsi b/arch/arm64/boot/dts/renesas/r8a77990.dtsi index fadb5f4effcf0a..0e9b837a6e5bd2 100644 --- a/arch/arm64/boot/dts/renesas/r8a77990.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a77990.dtsi @@ -20,19 +20,19 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_c: audio_clk_c { + audio_clk_c: audio-clk-c { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -127,13 +127,13 @@ }; /* External PCIe clock - can be overridden by the board */ - pcie_bus_clk: pcie_bus { + pcie_bus_clk: pcie-bus { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts = , ; diff --git a/arch/arm64/boot/dts/renesas/r8a77995.dtsi b/arch/arm64/boot/dts/renesas/r8a77995.dtsi index 522a49db025877..a20fff95407b6a 100644 --- a/arch/arm64/boot/dts/renesas/r8a77995.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a77995.dtsi @@ -21,13 +21,13 @@ * clocks by default. * Boards that provide audio clocks should override them. */ - audio_clk_a: audio_clk_a { + audio_clk_a: audio-clk-a { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; }; - audio_clk_b: audio_clk_b { + audio_clk_b: audio-clk-b { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -69,7 +69,7 @@ bootph-all; }; - pmu_a53 { + pmu-a53 { compatible = "arm,cortex-a53-pmu"; interrupts = ; }; diff --git a/arch/arm64/boot/dts/renesas/r8a779a0-falcon-cpu.dtsi b/arch/arm64/boot/dts/renesas/r8a779a0-falcon-cpu.dtsi index 0916fd57d1f1a0..2675f48d8fd025 100644 --- a/arch/arm64/boot/dts/renesas/r8a779a0-falcon-cpu.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a779a0-falcon-cpu.dtsi @@ -306,7 +306,7 @@ function = "scif0"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk"; function = "scif_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/r8a779a0-falcon.dts b/arch/arm64/boot/dts/renesas/r8a779a0-falcon.dts index ea5dcee73658ad..5708e395a226df 100644 --- a/arch/arm64/boot/dts/renesas/r8a779a0-falcon.dts +++ b/arch/arm64/boot/dts/renesas/r8a779a0-falcon.dts @@ -73,12 +73,12 @@ function = "avb0"; }; - pins_mdio { + pins-mdio { groups = "avb0_mdio"; drive-strength = <21>; }; - pins_mii { + pins-mii { groups = "avb0_rgmii"; drive-strength = <21>; }; diff --git a/arch/arm64/boot/dts/renesas/r8a779a0.dtsi b/arch/arm64/boot/dts/renesas/r8a779a0.dtsi index 0483a5d0714af7..38f35c81981bc4 100644 --- a/arch/arm64/boot/dts/renesas/r8a779a0.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a779a0.dtsi @@ -59,7 +59,7 @@ bootph-all; }; - pmu_a76 { + pmu-a76 { compatible = "arm,cortex-a76-pmu"; interrupts = ; }; diff --git a/arch/arm64/boot/dts/renesas/r8a779f0-spider-cpu.dtsi b/arch/arm64/boot/dts/renesas/r8a779f0-spider-cpu.dtsi index 1781bb79a6196f..0cfbff532f42d7 100644 --- a/arch/arm64/boot/dts/renesas/r8a779f0-spider-cpu.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a779f0-spider-cpu.dtsi @@ -206,7 +206,7 @@ function = "scif0"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk"; function = "scif_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/r8a779f0.dtsi b/arch/arm64/boot/dts/renesas/r8a779f0.dtsi index cbb161c863ac7b..69c247b7f0e28a 100644 --- a/arch/arm64/boot/dts/renesas/r8a779f0.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a779f0.dtsi @@ -279,7 +279,7 @@ clock-frequency = <0>; }; - pmu_a55 { + pmu-a55 { compatible = "arm,cortex-a55-pmu"; interrupts = ; }; diff --git a/arch/arm64/boot/dts/renesas/r8a779f4-s4sk.dts b/arch/arm64/boot/dts/renesas/r8a779f4-s4sk.dts index 67b18f2bffbd06..a0fd7f3e1ba99f 100644 --- a/arch/arm64/boot/dts/renesas/r8a779f4-s4sk.dts +++ b/arch/arm64/boot/dts/renesas/r8a779f4-s4sk.dts @@ -151,7 +151,7 @@ function = "i2c5"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk"; function = "scif_clk"; }; diff --git a/arch/arm64/boot/dts/renesas/r8a779g0.dtsi b/arch/arm64/boot/dts/renesas/r8a779g0.dtsi index 42b20ddcbeddba..ee3fd0df4fca60 100644 --- a/arch/arm64/boot/dts/renesas/r8a779g0.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a779g0.dtsi @@ -16,7 +16,7 @@ interrupt-parent = <&gic>; /* External Audio clock - to be overridden by boards that provide it */ - audio_clkin: audio_clkin { + audio_clkin: audio-clkin { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; @@ -192,7 +192,7 @@ clock-frequency = <0>; }; - pmu_a76 { + pmu-a76 { compatible = "arm,cortex-a76-pmu"; interrupts = ; }; diff --git a/arch/arm64/boot/dts/renesas/r8a779h0.dtsi b/arch/arm64/boot/dts/renesas/r8a779h0.dtsi index 74bc4c4854ecae..aca6ffc68c30a7 100644 --- a/arch/arm64/boot/dts/renesas/r8a779h0.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a779h0.dtsi @@ -16,7 +16,7 @@ interrupt-parent = <&gic>; /* External Audio clock - to be overridden by boards that provide it */ - audio_clkin: audio_clkin { + audio_clkin: audio-clkin { compatible = "fixed-clock"; #clock-cells = <0>; clock-frequency = <0>; diff --git a/arch/arm64/boot/dts/renesas/r8a779md-geist.dts b/arch/arm64/boot/dts/renesas/r8a779md-geist.dts index b186807926d2b3..252d9e9d3940db 100644 --- a/arch/arm64/boot/dts/renesas/r8a779md-geist.dts +++ b/arch/arm64/boot/dts/renesas/r8a779md-geist.dts @@ -460,12 +460,12 @@ function = "avb"; }; - pins_mdio { + pins-mdio { groups = "avb_mdio"; drive-strength = <24>; }; - pins_mii_tx { + pins-mii-tx { pins = "PIN_AVB_TX_CTL", "PIN_AVB_TXC", "PIN_AVB_TD0", "PIN_AVB_TD1", "PIN_AVB_TD2", "PIN_AVB_TD3"; drive-strength = <12>; @@ -507,7 +507,7 @@ function = "scif2"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk_a"; function = "scif_clk"; }; @@ -518,7 +518,7 @@ power-source = <3300>; }; - sdhi0_pins_uhs: sd0_uhs { + sdhi0_pins_uhs: sd0-uhs { groups = "sdhi0_data4", "sdhi0_ctrl"; function = "sdhi0"; power-source = <1800>; @@ -535,7 +535,7 @@ function = "ssi"; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { groups = "audio_clk_a_a", "audio_clk_b_a", "audio_clk_c_a", "audio_clkout_a", "audio_clkout3_a"; function = "audio_clk"; diff --git a/arch/arm64/boot/dts/renesas/r9a09g057h44-rzv2h-evk.dts b/arch/arm64/boot/dts/renesas/r9a09g057h44-rzv2h-evk.dts index 637fc92dcc2632..0926ab891fa5cf 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g057h44-rzv2h-evk.dts +++ b/arch/arm64/boot/dts/renesas/r9a09g057h44-rzv2h-evk.dts @@ -461,20 +461,20 @@ }; sdhi1_pins: sd1 { - sd1_dat_cmd { + sd1-dat-cmd { pins = "SD1DAT0", "SD1DAT1", "SD1DAT2", "SD1DAT3", "SD1CMD"; input-enable; renesas,output-impedance = <3>; slew-rate = <0>; }; - sd1_clk { + sd1-clk { pins = "SD1CLK"; renesas,output-impedance = <3>; slew-rate = <0>; }; - sd1_cd { + sd1-cd { pinmux = ; /* SD1_CD */ }; }; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-smarc-pinfunction.dtsi b/arch/arm64/boot/dts/renesas/rzg2l-smarc-pinfunction.dtsi index 2616dbde4dd597..ee141993d93ada 100644 --- a/arch/arm64/boot/dts/renesas/rzg2l-smarc-pinfunction.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg2l-smarc-pinfunction.dtsi @@ -98,38 +98,38 @@ }; sdhi1_pins: sd1 { - sd1_data { + sd1-data { pins = "SD1_DATA0", "SD1_DATA1", "SD1_DATA2", "SD1_DATA3"; power-source = <3300>; }; - sd1_ctrl { + sd1-ctrl { pins = "SD1_CLK", "SD1_CMD"; power-source = <3300>; }; - sd1_mux { + sd1-mux { pinmux = ; /* SD1_CD */ }; }; - sdhi1_pins_uhs: sd1_uhs { - sd1_data_uhs { + sdhi1_pins_uhs: sd1-uhs { + sd1-data-uhs { pins = "SD1_DATA0", "SD1_DATA1", "SD1_DATA2", "SD1_DATA3"; power-source = <1800>; }; - sd1_ctrl_uhs { + sd1-ctrl-uhs { pins = "SD1_CLK", "SD1_CMD"; power-source = <1800>; }; - sd1_mux_uhs { + sd1-mux-uhs { pinmux = ; /* SD1_CD */ }; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { pins = "AUDIO_CLK1", "AUDIO_CLK2"; input-enable; }; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-smarc-som.dtsi b/arch/arm64/boot/dts/renesas/rzg2l-smarc-som.dtsi index 7eccdaffb221f4..b5f20ff1710968 100644 --- a/arch/arm64/boot/dts/renesas/rzg2l-smarc-som.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg2l-smarc-som.dtsi @@ -269,51 +269,51 @@ }; sdhi0_emmc_pins: sd0emmc { - sd0_emmc_data { + sd0-emmc-data { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; power-source = <1800>; }; - sd0_emmc_ctrl { + sd0-emmc-ctrl { pins = "SD0_CLK", "SD0_CMD"; power-source = <1800>; }; - sd0_emmc_rst { + sd0-emmc-rst { pins = "SD0_RST#"; power-source = <1800>; }; }; sdhi0_pins: sd0 { - sd0_data { + sd0-data { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3"; power-source = <3300>; }; - sd0_ctrl { + sd0-ctrl { pins = "SD0_CLK", "SD0_CMD"; power-source = <3300>; }; - sd0_mux { + sd0-mux { pinmux = ; /* SD0_CD */ }; }; - sdhi0_pins_uhs: sd0_uhs { - sd0_data_uhs { + sdhi0_pins_uhs: sd0-uhs { + sd0-data-uhs { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3"; power-source = <1800>; }; - sd0_ctrl_uhs { + sd0-ctrl-uhs { pins = "SD0_CLK", "SD0_CMD"; power-source = <1800>; }; - sd0_mux_uhs { + sd0-mux-uhs { pinmux = ; /* SD0_CD */ }; }; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-smarc.dtsi b/arch/arm64/boot/dts/renesas/rzg2l-smarc.dtsi index b76b55e7f09dfb..1411e61a53000c 100644 --- a/arch/arm64/boot/dts/renesas/rzg2l-smarc.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg2l-smarc.dtsi @@ -31,7 +31,7 @@ }; }; - sound_card { + sound-card { compatible = "audio-graph-card"; label = "HDMI-Audio"; dais = <&i2s2_port>; diff --git a/arch/arm64/boot/dts/renesas/rzg2lc-smarc-pinfunction.dtsi b/arch/arm64/boot/dts/renesas/rzg2lc-smarc-pinfunction.dtsi index 92c64d58349f36..92ec7935025cb0 100644 --- a/arch/arm64/boot/dts/renesas/rzg2lc-smarc-pinfunction.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg2lc-smarc-pinfunction.dtsi @@ -79,38 +79,38 @@ }; sdhi1_pins: sd1 { - sd1_data { + sd1-data { pins = "SD1_DATA0", "SD1_DATA1", "SD1_DATA2", "SD1_DATA3"; power-source = <3300>; }; - sd1_ctrl { + sd1-ctrl { pins = "SD1_CLK", "SD1_CMD"; power-source = <3300>; }; - sd1_mux { + sd1-mux { pinmux = ; /* SD1_CD */ }; }; - sdhi1_pins_uhs: sd1_uhs { - sd1_data_uhs { + sdhi1_pins_uhs: sd1-uhs { + sd1-data-uhs { pins = "SD1_DATA0", "SD1_DATA1", "SD1_DATA2", "SD1_DATA3"; power-source = <1800>; }; - sd1_ctrl_uhs { + sd1-ctrl-uhs { pins = "SD1_CLK", "SD1_CMD"; power-source = <1800>; }; - sd1_mux_uhs { + sd1-mux-uhs { pinmux = ; /* SD1_CD */ }; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { pins = "AUDIO_CLK1", "AUDIO_CLK2"; input-enable; }; diff --git a/arch/arm64/boot/dts/renesas/rzg2lc-smarc-som.dtsi b/arch/arm64/boot/dts/renesas/rzg2lc-smarc-som.dtsi index 15f2e9eaaf0b62..a043f2f5703632 100644 --- a/arch/arm64/boot/dts/renesas/rzg2lc-smarc-som.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg2lc-smarc-som.dtsi @@ -189,51 +189,51 @@ }; sdhi0_emmc_pins: sd0emmc { - sd0_emmc_data { + sd0-emmc-data { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; power-source = <1800>; }; - sd0_emmc_ctrl { + sd0-emmc-ctrl { pins = "SD0_CLK", "SD0_CMD"; power-source = <1800>; }; - sd0_emmc_rst { + sd0-emmc-rst { pins = "SD0_RST#"; power-source = <1800>; }; }; sdhi0_pins: sd0 { - sd0_data { + sd0-data { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3"; power-source = <3300>; }; - sd0_ctrl { + sd0-ctrl { pins = "SD0_CLK", "SD0_CMD"; power-source = <3300>; }; - sd0_mux { + sd0-mux { pinmux = ; /* SD0_CD */ }; }; - sdhi0_pins_uhs: sd0_uhs { - sd0_data_uhs { + sdhi0_pins_uhs: sd0-uhs { + sd0-data-uhs { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3"; power-source = <1800>; }; - sd0_ctrl_uhs { + sd0-ctrl-uhs { pins = "SD0_CLK", "SD0_CMD"; power-source = <1800>; }; - sd0_mux_uhs { + sd0-mux-uhs { pinmux = ; /* SD0_CD */ }; }; diff --git a/arch/arm64/boot/dts/renesas/rzg2lc-smarc.dtsi b/arch/arm64/boot/dts/renesas/rzg2lc-smarc.dtsi index f3d7eff0d2f2a0..0d4c6b37d7a70f 100644 --- a/arch/arm64/boot/dts/renesas/rzg2lc-smarc.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg2lc-smarc.dtsi @@ -37,7 +37,7 @@ #if (SW_I2S0_I2S1) /delete-node/ sound; - sound_card { + sound-card { compatible = "audio-graph-card"; label = "HDMI-Audio"; dais = <&i2s2_port>; diff --git a/arch/arm64/boot/dts/renesas/rzg2ul-smarc-pinfunction.dtsi b/arch/arm64/boot/dts/renesas/rzg2ul-smarc-pinfunction.dtsi index 355694fe4af686..0be666c9049fb5 100644 --- a/arch/arm64/boot/dts/renesas/rzg2ul-smarc-pinfunction.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg2ul-smarc-pinfunction.dtsi @@ -69,38 +69,38 @@ }; sdhi1_pins: sd1 { - sd1_data { + sd1-data { pins = "SD1_DATA0", "SD1_DATA1", "SD1_DATA2", "SD1_DATA3"; power-source = <3300>; }; - sd1_ctrl { + sd1-ctrl { pins = "SD1_CLK", "SD1_CMD"; power-source = <3300>; }; - sd1_mux { + sd1-mux { pinmux = ; /* SD1_CD */ }; }; - sdhi1_pins_uhs: sd1_uhs { - sd1_data_uhs { + sdhi1_pins_uhs: sd1-uhs { + sd1-data-uhs { pins = "SD1_DATA0", "SD1_DATA1", "SD1_DATA2", "SD1_DATA3"; power-source = <1800>; }; - sd1_ctrl_uhs { + sd1-ctrl-uhs { pins = "SD1_CLK", "SD1_CMD"; power-source = <1800>; }; - sd1_mux_uhs { + sd1-mux-uhs { pinmux = ; /* SD1_CD */ }; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { pins = "AUDIO_CLK1", "AUDIO_CLK2"; input-enable; }; diff --git a/arch/arm64/boot/dts/renesas/rzg2ul-smarc-som.dtsi b/arch/arm64/boot/dts/renesas/rzg2ul-smarc-som.dtsi index 0f917d7c99398f..99d30fa9a1a83d 100644 --- a/arch/arm64/boot/dts/renesas/rzg2ul-smarc-som.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg2ul-smarc-som.dtsi @@ -204,51 +204,51 @@ }; sdhi0_emmc_pins: sd0emmc { - sd0_emmc_data { + sd0-emmc-data { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; power-source = <1800>; }; - sd0_emmc_ctrl { + sd0-emmc-ctrl { pins = "SD0_CLK", "SD0_CMD"; power-source = <1800>; }; - sd0_emmc_rst { + sd0-emmc-rst { pins = "SD0_RST#"; power-source = <1800>; }; }; sdhi0_pins: sd0 { - sd0_data { + sd0-data { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3"; power-source = <3300>; }; - sd0_ctrl { + sd0-ctrl { pins = "SD0_CLK", "SD0_CMD"; power-source = <3300>; }; - sd0_mux { + sd0-mux { pinmux = ; /* SD0_CD */ }; }; - sdhi0_pins_uhs: sd0_uhs { - sd0_data_uhs { + sdhi0_pins_uhs: sd0-uhs { + sd0-data-uhs { pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3"; power-source = <1800>; }; - sd0_ctrl_uhs { + sd0-ctrl-uhs { pins = "SD0_CLK", "SD0_CMD"; power-source = <1800>; }; - sd0_mux_uhs { + sd0-mux-uhs { pinmux = ; /* SD0_CD */ }; }; diff --git a/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi b/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi index ded6066c91765b..462102a620257b 100644 --- a/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi +++ b/arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi @@ -239,7 +239,7 @@ drive-strength-microamp = <5200>; }; - tx_ctl { + tx-ctl { pinmux = ; /* ET0_TX_CTL */ power-source = <1800>; output-enable; @@ -282,7 +282,7 @@ drive-strength-microamp = <5200>; }; - tx_ctl { + tx-ctl { pinmux = ; /* ET1_TX_CTL */ power-source = <1800>; output-enable; diff --git a/arch/arm64/boot/dts/renesas/salvator-common.dtsi b/arch/arm64/boot/dts/renesas/salvator-common.dtsi index 9f8c545dad34ad..b9bdc366088eea 100644 --- a/arch/arm64/boot/dts/renesas/salvator-common.dtsi +++ b/arch/arm64/boot/dts/renesas/salvator-common.dtsi @@ -686,12 +686,12 @@ function = "avb"; }; - pins_mdio { + pins-mdio { groups = "avb_mdio"; drive-strength = <24>; }; - pins_mii_tx { + pins-mii-tx { pins = "PIN_AVB_TX_CTL", "PIN_AVB_TXC", "PIN_AVB_TD0", "PIN_AVB_TD1", "PIN_AVB_TD2", "PIN_AVB_TD3"; drive-strength = <12>; @@ -738,7 +738,7 @@ function = "scif2"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk_a"; function = "scif_clk"; }; @@ -749,7 +749,7 @@ power-source = <3300>; }; - sdhi0_pins_uhs: sd0_uhs { + sdhi0_pins_uhs: sd0-uhs { groups = "sdhi0_data4", "sdhi0_ctrl"; function = "sdhi0"; power-source = <1800>; @@ -767,7 +767,7 @@ power-source = <3300>; }; - sdhi3_pins_uhs: sd3_uhs { + sdhi3_pins_uhs: sd3-uhs { groups = "sdhi3_data4", "sdhi3_ctrl"; function = "sdhi3"; power-source = <1800>; @@ -778,7 +778,7 @@ function = "ssi"; }; - sound_clk_pins: sound_clk { + sound_clk_pins: sound-clk { groups = "audio_clk_a_a", "audio_clk_b_a", "audio_clk_c_a", "audio_clkout_a", "audio_clkout3_a"; function = "audio_clk"; diff --git a/arch/arm64/boot/dts/renesas/salvator-xs.dtsi b/arch/arm64/boot/dts/renesas/salvator-xs.dtsi index 1d18dedb1ff039..f5157940bd60cc 100644 --- a/arch/arm64/boot/dts/renesas/salvator-xs.dtsi +++ b/arch/arm64/boot/dts/renesas/salvator-xs.dtsi @@ -72,7 +72,7 @@ * - Connect GP6_3[01] to BD082065 (USB2.0 ch3's host power). * - Connect GP6_{04,21} to ADV7842. */ - usb2_ch3_pins: usb2_ch3 { + usb2_ch3_pins: usb2-ch3 { groups = "usb2_ch3"; function = "usb2_ch3"; }; diff --git a/arch/arm64/boot/dts/renesas/ulcb-kf.dtsi b/arch/arm64/boot/dts/renesas/ulcb-kf.dtsi index 97014bcfbb1d22..603c791e438db6 100644 --- a/arch/arm64/boot/dts/renesas/ulcb-kf.dtsi +++ b/arch/arm64/boot/dts/renesas/ulcb-kf.dtsi @@ -84,7 +84,7 @@ regulator-always-on; }; - wlan_en: regulator-wlan_en { + wlan_en: regulator-wlan-en { compatible = "regulator-fixed"; regulator-name = "wlan-en-regulator"; diff --git a/arch/arm64/boot/dts/renesas/ulcb.dtsi b/arch/arm64/boot/dts/renesas/ulcb.dtsi index 119f2b5024b3d9..cedff8e0157edf 100644 --- a/arch/arm64/boot/dts/renesas/ulcb.dtsi +++ b/arch/arm64/boot/dts/renesas/ulcb.dtsi @@ -311,12 +311,12 @@ function = "avb"; }; - pins_mdio { + pins-mdio { groups = "avb_mdio"; drive-strength = <24>; }; - pins_mii_tx { + pins-mii-tx { pins = "PIN_AVB_TX_CTL", "PIN_AVB_TXC", "PIN_AVB_TD0", "PIN_AVB_TD1", "PIN_AVB_TD2", "PIN_AVB_TD3"; drive-strength = <12>; @@ -338,7 +338,7 @@ function = "scif2"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk_a"; function = "scif_clk"; }; @@ -349,7 +349,7 @@ power-source = <3300>; }; - sdhi0_pins_uhs: sd0_uhs { + sdhi0_pins_uhs: sd0-uhs { groups = "sdhi0_data4", "sdhi0_ctrl"; function = "sdhi0"; power-source = <1800>; diff --git a/arch/arm64/boot/dts/renesas/white-hawk-cpu-common.dtsi b/arch/arm64/boot/dts/renesas/white-hawk-cpu-common.dtsi index c5045bda45c338..2bcd486b03e93b 100644 --- a/arch/arm64/boot/dts/renesas/white-hawk-cpu-common.dtsi +++ b/arch/arm64/boot/dts/renesas/white-hawk-cpu-common.dtsi @@ -320,12 +320,12 @@ function = "avb0"; }; - pins_mdio { + pins-mdio { groups = "avb0_mdio"; drive-strength = <21>; }; - pins_mii { + pins-mii { groups = "avb0_rgmii"; drive-strength = <21>; }; @@ -368,7 +368,7 @@ function = "qspi0"; }; - scif_clk_pins: scif_clk { + scif_clk_pins: scif-clk { groups = "scif_clk"; function = "scif_clk"; }; From 84c3808811a9a99d5bed4abe4a87ac3a1b75052d Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Mon, 24 Aug 2026 08:59:34 +0100 Subject: [PATCH 452/857] arm64: dts: renesas: r9a09g077: Add RTC node Add Real Time Clock (RTC) node to SoC DTSI. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260824075936.2904336-2-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r9a09g077.dtsi | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi index d5eafaf23e75ec..33f953c110bb39 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi @@ -1173,6 +1173,19 @@ power-domains = <&cpg>; }; + rtc0: rtc@81009000 { + compatible = "renesas,r9a09g077-rtc"; + reg = <0 0x81009000 0 0x1000>; + interrupts = , + , + ; + interrupt-names = "alarm", "timer", "pps"; + clocks = <&cpg CPG_MOD 605>, <&cpg CPG_CORE R9A09G077_PCLKRTC>; + clock-names = "hclk", "xtal"; + power-domains = <&cpg>; + status = "disabled"; + }; + gic: interrupt-controller@83000000 { compatible = "arm,gic-v3"; reg = <0x0 0x83000000 0 0x40000>, From 61cb9aee632fbed9941c1712a3ff48c021cca3dd Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Mon, 24 Aug 2026 08:59:35 +0100 Subject: [PATCH 453/857] arm64: dts: renesas: r9a09g087: Add RTC node Add Real Time Clock (RTC) node to SoC DTSI. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260824075936.2904336-3-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/r9a09g087.dtsi | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi index fe9003426d70ef..d5b3eb6c173689 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi @@ -1176,6 +1176,19 @@ power-domains = <&cpg>; }; + rtc0: rtc@81009000 { + compatible = "renesas,r9a09g087-rtc", "renesas,r9a09g077-rtc"; + reg = <0 0x81009000 0 0x1000>; + interrupts = , + , + ; + interrupt-names = "alarm", "timer", "pps"; + clocks = <&cpg CPG_MOD 605>, <&cpg CPG_CORE R9A09G087_PCLKRTC>; + clock-names = "hclk", "xtal"; + power-domains = <&cpg>; + status = "disabled"; + }; + gic: interrupt-controller@83000000 { compatible = "arm,gic-v3"; reg = <0x0 0x83000000 0 0x40000>, From a7893cc3194f0adfa032624421530ba1f0324976 Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Mon, 24 Aug 2026 08:59:36 +0100 Subject: [PATCH 454/857] arm64: dts: renesas: rzt2h-n2h-evk: Enable RTC support Enable RTC support on RZ/T2H and RZ/N2H EVKs. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260824075936.2904336-4-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi b/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi index ffcc0960903bf4..1994ce33942e4b 100644 --- a/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi +++ b/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi @@ -20,6 +20,7 @@ i2c1 = &i2c1; mmc0 = &sdhi0; mmc1 = &sdhi1; + rtc0 = &rtc0; serial0 = &sci0; spi0 = &xspi0; spi1 = &xspi1; @@ -514,6 +515,10 @@ }; }; +&rtc0 { + status = "okay"; +}; + &sci0 { pinctrl-0 = <&sci0_pins>; pinctrl-names = "default"; From 5c9b0eda22ec63c98286c52dc05ae348869b04eb Mon Sep 17 00:00:00 2001 From: Lad Prabhakar Date: Wed, 26 Aug 2026 12:28:21 +0100 Subject: [PATCH 455/857] arm64: dts: renesas: rzt2h-n2h-evk: Mark XSPI1 boot partitions read-only Mark the BL2 and FIP partitions on XSPI1 as read-only to prevent them from being modified through the MTD partition interface. The corresponding boot partitions on XSPI0 are already marked read-only. Signed-off-by: Lad Prabhakar Reviewed-by: Geert Uytterhoeven Link: https://patch.msgid.link/20260826112821.194877-1-prabhakar.mahadev-lad.rj@bp.renesas.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi b/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi index 1994ce33942e4b..224d133e3a0f7b 100644 --- a/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi +++ b/arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-common.dtsi @@ -652,11 +652,13 @@ partition@0 { label = "bl2-1"; reg = <0x00000000 0x00060000>; + read-only; }; partition@60000 { label = "fip-1"; reg = <0x00060000 0x007a0000>; + read-only; }; partition@800000 { From a15f90964998e1d7b5f3aa34d29f3acac5971038 Mon Sep 17 00:00:00 2001 From: Linmao Li Date: Fri, 28 Aug 2026 14:19:49 +0800 Subject: [PATCH 456/857] hwmon: (corsair-cpro) Remove debugfs entries when probe fails ccp_debugfs_init() registers debugfs files whose private data is the devm allocated ccp. If hwmon_device_register_with_info() fails right after it, ccp_probe() returns without removing them: the HID core then frees ccp, and ccp_remove() is not called for a failed probe, so the files stay behind. Reading one of them dereferences the freed pointer. Remove the debugfs entries on that error path. debugfs_remove_recursive() waits for readers already inside the show callbacks, so ccp is no longer reachable through debugfs by the time probe returns. Reported-by: Sashiko Closes: https://lore.kernel.org/linux-hwmon/20260708031612.BD7E61F000E9@smtp.kernel.org/ Fixes: 5997eb60f896 ("hwmon: (corsair-cpro) Add firmware and bootloader information") Signed-off-by: Linmao Li Link: https://patch.msgid.link/20260828061949.3151191-1-lilinmao@kylinos.cn Signed-off-by: Guenter Roeck --- drivers/hwmon/corsair-cpro.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/drivers/hwmon/corsair-cpro.c b/drivers/hwmon/corsair-cpro.c index 56de0fe0f544c5..c0964515261329 100644 --- a/drivers/hwmon/corsair-cpro.c +++ b/drivers/hwmon/corsair-cpro.c @@ -642,13 +642,15 @@ static int ccp_probe(struct hid_device *hdev, const struct hid_device_id *id) ccp, &ccp_chip_info, NULL); if (IS_ERR(ccp->hwmon_dev)) { ret = PTR_ERR(ccp->hwmon_dev); - goto out_hw_close; + goto out_debugfs_remove; } ccp_debugfs_init(ccp, fw_valid, bl_valid); return 0; +out_debugfs_remove: + debugfs_remove_recursive(ccp->debugfs); out_hw_close: hid_hw_close(hdev); hid_device_io_stop(hdev); From a20c7ae6c80fddbd8f14a8a624d91f3acbd20ec1 Mon Sep 17 00:00:00 2001 From: Diego Oliva Date: Wed, 2 Sep 2026 11:42:07 +0100 Subject: [PATCH 457/857] smb: client: reject out-of-bounds DataOffset in CIFSSMBRead() The SMB1 synchronous read helper CIFSSMBRead() validates the server's DataLength against CIFSMaxBufSize and the caller's count, but never validates DataOffset. The copy source is formed as &pSMBr->hdr.Protocol + le16_to_cpu(pSMBr->DataOffset) and memcpy()'d for DataLength bytes with no check that the [DataOffset, DataOffset + DataLength) range lies within the response actually received from the server. A malicious or compromised SMB1 server can return a response carrying an in-range DataLength and a large DataOffset, driving the source pointer past the end of the response buffer. The memcpy() then copies adjacent kernel heap into the caller's read buffer (information disclosure), or reads unmapped memory and oopses (denial of service). SMB1 is not negotiated by default; reaching this code requires an explicit vers=1.0 mount. Both DataOffset and the received response length recorded in rsp_iov.iov_len are relative to the start of the SMB header, so reject the response unless DataOffset + DataLength fits within that length, using overflow-safe arithmetic, before forming the source pointer. The response length has been validated by the previous patch, so the DataOffset and DataLength fields can be read safely here. While here, make data_length unsigned. It holds a length derived from unsigned on-the-wire fields and is only ever compared against unsigned quantities; print it with %u accordingly, and add __func__ to the cifs_dbg() calls in this function. Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Cc: stable@vger.kernel.org # 6.19.x Assisted-by: Bynario AI Signed-off-by: Diego Oliva Reviewed-by: David Howells Signed-off-by: Paulo Alcantara --- fs/smb/client/cifssmb.c | 17 ++++++++++++----- fs/smb/client/trace.h | 1 + 2 files changed, 13 insertions(+), 5 deletions(-) diff --git a/fs/smb/client/cifssmb.c b/fs/smb/client/cifssmb.c index be13ab37039d97..f3cba16f6e1794 100644 --- a/fs/smb/client/cifssmb.c +++ b/fs/smb/client/cifssmb.c @@ -1728,7 +1728,8 @@ CIFSSMBRead(const unsigned int xid, struct cifs_io_parms *io_parms, rsp_iov.iov_len, tcon->ses->server->vals->read_rsp_size); *nbytes = 0; } else { - int data_length = le16_to_cpu(pSMBr->DataLengthHigh); + unsigned int data_length = le16_to_cpu(pSMBr->DataLengthHigh); + __u16 data_offset = le16_to_cpu(pSMBr->DataOffset); data_length = data_length << 16; data_length += le16_to_cpu(pSMBr->DataLength); *nbytes = data_length; @@ -1736,14 +1737,20 @@ CIFSSMBRead(const unsigned int xid, struct cifs_io_parms *io_parms, /*check that DataLength would not go beyond end of SMB */ if ((data_length > CIFSMaxBufSize) || (data_length > count)) { - cifs_dbg(FYI, "bad length %d for count %d\n", - data_length, count); + cifs_dbg(FYI, "%s: bad length %u for count %u\n", + __func__, data_length, count); rc = smb_EIO2(smb_eio_trace_read_overlarge, data_length, count); *nbytes = 0; + } else if ((size_t)data_offset + data_length > rsp_iov.iov_len) { + /* check that the data lies within the received response */ + cifs_dbg(FYI, "%s: bad data offset %u length %u for response of %zu\n", + __func__, data_offset, data_length, rsp_iov.iov_len); + rc = smb_EIO2(smb_eio_trace_read_bad_offset, + data_offset, data_length); + *nbytes = 0; } else { - pReadData = (char *) (&pSMBr->hdr.Protocol) + - le16_to_cpu(pSMBr->DataOffset); + pReadData = (char *) (&pSMBr->hdr.Protocol) + data_offset; /* if (rc = copy_to_user(buf, pReadData, data_length)) { cifs_dbg(VFS, "Faulting on read rc = %d\n",rc); rc = -EFAULT; diff --git a/fs/smb/client/trace.h b/fs/smb/client/trace.h index 12241abb8e2ebc..b442cccd153085 100644 --- a/fs/smb/client/trace.h +++ b/fs/smb/client/trace.h @@ -79,6 +79,7 @@ EM(smb_eio_trace_qreparse_setup_count, "qreparse_setup_count") \ EM(smb_eio_trace_qreparse_sizes_wrong, "qreparse_sizes_wrong") \ EM(smb_eio_trace_qsym_bcc_too_small, "qsym_bcc_too_small") \ + EM(smb_eio_trace_read_bad_offset, "read_bad_offset") \ EM(smb_eio_trace_read_mid_state_unknown, "read_mid_state_unknown") \ EM(smb_eio_trace_read_overlarge, "read_overlarge") \ EM(smb_eio_trace_read_rsp_malformed, "read_rsp_malformed") \ From 4ee5025d18677575c7b303fdd1bc5f3b8e7ab29c Mon Sep 17 00:00:00 2001 From: Aohan Mei Date: Wed, 2 Sep 2026 20:52:13 +0800 Subject: [PATCH 458/857] smb: client: reject userspace cifs.idmap descriptions cifs.idmap key descriptions carry authority-bearing fields (owner and group SIDs and uid/gid values in "os:"/"gs:"/"oi:"/"gi:" form) that the cifs.idmap upcall helper treats as kernel-originating inputs. Unlike its sibling cifs.spnego, the cifs.idmap key type has no vet_description hook, so userspace can create keys of this type through request_key(2)/add_key(2) and supply those fields without CIFS origin. A request_key(2) call with a non-NULL callout then drives a root usermodehelper upcall (/sbin/request-key -> cifs.idmap) that consumes the unvetted description in root context. Only accept cifs.idmap descriptions while CIFS is using its private root_cred to request the key. id_to_sid()/sid_to_id() already run under override_creds(root_cred), so the kernel-originated path is unaffected. This mirrors commit 3da1fdf4efbc ("smb: client: reject userspace cifs.spnego descriptions"), which applied the same restriction to cifs.spnego. Fixes: 4d79dba0e007 ("cifs: Add idmap key and related data structures and functions (try #17 repost)") Reported-by: TencentOS Corvus AI Cc: stable@vger.kernel.org Assisted-by: CodeBuddy:Kimi-K3 Signed-off-by: Aohan Mei Acked-by: David Howells Signed-off-by: Paulo Alcantara --- fs/smb/client/cifsacl.c | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/fs/smb/client/cifsacl.c b/fs/smb/client/cifsacl.c index 12005f46307dea..213a421bf8e93b 100644 --- a/fs/smb/client/cifsacl.c +++ b/fs/smb/client/cifsacl.c @@ -100,8 +100,23 @@ cifs_idmap_key_destroy(struct key *key) kfree(key->payload.data[0]); } +static int +cifs_idmap_key_vet_description(const char *description) +{ + /* + * cifs.idmap descriptions are authority-bearing inputs to the + * cifs.idmap upcall helper. Only allow the kernel to create this + * type of key using the private root_cred installed in + * init_cifs_idmap; reject userspace request_key(2)/add_key(2). + */ + if (current_cred() != root_cred) + return -EPERM; + return 0; +} + static struct key_type cifs_idmap_key_type = { .name = "cifs.idmap", + .vet_description = cifs_idmap_key_vet_description, .instantiate = cifs_idmap_key_instantiate, .destroy = cifs_idmap_key_destroy, .describe = user_describe, From c6d60c24cd7b3d74d1f7ad5db5651cb581801cc4 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Wed, 2 Sep 2026 00:04:33 +0300 Subject: [PATCH 459/857] Bluetooth: L2CAP: take lock for l2cap_chan_del in l2cap_ecred_rsp_defer l2cap_ecred_rsp_defer() calls l2cap_chan_del without holding chan->lock, which ends up calling ops->teardown() with wrong lock context. Fix by taking chan->lock in l2cap_ecred_rsp_defer(). AB-BA deadlocks between sibling l2cap_chan are avoided here via requiring l2cap_conn::lock to serialize all nested l2cap_chan locking on same nesting level. In current code, there is no nested l2cap_chan locking on same nesting level, so we can add this new requirement. Also return early from __l2cap_ecred_conn_rsp_defer() if chan did not have FLAG_DEFER_SETUP, as then no RSP shall be sent for it, to make sure SMP channels are excluded. Also hold chan reference over l2cap_chan_del(), in case chan_l reference was the last. Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 4 +++ net/bluetooth/l2cap_core.c | 60 +++++++++++++++++++++++++++++++++++ 2 files changed, 64 insertions(+) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index 2315a3993c7bdb..c7e642abe40404 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -758,6 +758,10 @@ enum { * otherwise considers all channels equal and will e.g. complain about a * connection oriented channel triggering SMP procedures or a listening * channel creating and locking a child channel. + * + * Lock nesting of channels at the same nesting level is allowed if the channels + * have the same l2cap_chan::conn and l2cap_chan::conn.lock is taken before the + * nested locks. l2cap_chan_try_sibling_lock() must be used. */ enum { L2CAP_NESTING_SMP, diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 9da689f0a50a9f..805623a48baef0 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -848,6 +848,8 @@ static void __l2cap_chan_close(struct l2cap_chan *chan, int reason) BT_DBG("chan %p state %s", chan, state_to_string(chan->state)); + lockdep_assert_held(&chan->lock); + switch (chan->state) { case BT_LISTEN: chan->ops->teardown(chan, 0); @@ -3950,6 +3952,7 @@ static void l2cap_ecred_list_defer(struct l2cap_chan *chan, void *data) } struct l2cap_ecred_rsp_data { + struct l2cap_chan *locked_chan; struct { struct l2cap_ecred_conn_rsp_hdr rsp; __le16 scid[L2CAP_ECRED_MAX_CID]; @@ -3957,11 +3960,42 @@ struct l2cap_ecred_rsp_data { int count; }; +/* Lock @chan if it is not @locked_chan, and has same or lower nesting level. + * + * They must have the same chan->conn, and conn->lock must be held. + * + * Caller must ensure @chan has lock nesting level <= that of @locked_chan, as + * nested locking of l2cap_chan of different levels is allowed also without + * holding conn->lock. + * + * See l2cap.h for the global l2cap_chan locking rules. + */ +static bool l2cap_chan_try_sibling_lock(struct l2cap_chan *chan, + struct l2cap_chan *locked_chan) + __must_hold(&locked_chan->lock) + __must_hold(&locked_chan->conn->lock) + __cond_acquires(true, &chan->lock) +{ + if (chan == locked_chan) + return false; + + if (WARN_ON_ONCE(locked_chan->conn != chan->conn)) + return false; + + if (WARN_ON_ONCE(atomic_read(&locked_chan->nesting) + < atomic_read(&chan->nesting))) + return false; + + mutex_lock_nest_lock(&chan->lock, &locked_chan->conn->lock); + return true; +} + static void l2cap_ecred_rsp_defer(struct l2cap_chan *chan, void *data) { struct l2cap_ecred_rsp_data *rsp = data; struct l2cap_ecred_conn_rsp *rsp_flex = container_of(&rsp->pdu.rsp, struct l2cap_ecred_conn_rsp, hdr); + bool locked; if (chan->mode != L2CAP_MODE_EXT_FLOWCTL) return; @@ -3973,6 +4007,22 @@ static void l2cap_ecred_rsp_defer(struct l2cap_chan *chan, void *data) !test_and_clear_bit(FLAG_DEFER_SETUP, &chan->flags)) return; + lockdep_assert_held(&rsp->locked_chan->lock); + lockdep_assert_held(&rsp->locked_chan->conn->lock); + + l2cap_chan_hold(chan); + + locked = l2cap_chan_try_sibling_lock(chan, rsp->locked_chan); + + /* Cannot occur: PARENT channels do not appear in chan_l, and SMP + * channels never have FLAG_DEFER_SETUP. + */ + if (context_unsafe(!locked && chan != rsp->locked_chan)) + goto done; + + lockdep_assert_held(&chan->lock); + lockdep_assert_held(&chan->conn->lock); + /* Reset ident so only one response is sent */ chan->ident = 0; @@ -3985,6 +4035,12 @@ static void l2cap_ecred_rsp_defer(struct l2cap_chan *chan, void *data) rsp_flex->dcid[rsp->count++] = cpu_to_le16(chan->scid); else l2cap_chan_del(chan, ECONNRESET); + +done: + if (locked) + l2cap_chan_unlock(chan); + + l2cap_chan_put(chan); } void __l2cap_ecred_conn_rsp_defer(struct l2cap_chan *chan) @@ -3996,11 +4052,15 @@ void __l2cap_ecred_conn_rsp_defer(struct l2cap_chan *chan) if (!id) return; + if (!test_bit(FLAG_DEFER_SETUP, &chan->flags)) + return; BT_DBG("chan %p id %d", chan, id); memset(&data, 0, sizeof(data)); + data.locked_chan = chan; + data.pdu.rsp.mtu = cpu_to_le16(chan->imtu); data.pdu.rsp.mps = cpu_to_le16(chan->mps); data.pdu.rsp.credits = cpu_to_le16(chan->rx_credits); From 6873eb51dcdd9ae01f8c682e482c8915dbbb138f Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Wed, 2 Sep 2026 00:04:34 +0300 Subject: [PATCH 460/857] Bluetooth: L2CAP: annotate locking for l2cap_chan_del() Add context analysis annotations for chan->lock and chan->conn->lock involving l2cap_chan_del() usage. Add necessary annotations and related lockdep_assert_held to callers. Move struct l2cap_ops definition after struct l2cap_conn, so that the callbacks can be annotated. In l2cap_chan_close_unlocked() we consider chan->conn->lock as locked even if chan->conn == NULL, to avoid needing to define separate __l2cap_chan_close/del for this NULL case. Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 56 +++++++++++++++++++---------------- net/bluetooth/6lowpan.c | 2 ++ net/bluetooth/l2cap_core.c | 49 +++++++++++++++++++++++++----- 3 files changed, 74 insertions(+), 33 deletions(-) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index c7e642abe40404..3e1e2b36d7b6db 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -614,31 +614,6 @@ struct l2cap_chan { struct mutex lock; }; -struct l2cap_ops { - char *name; - - int (*new_connection)(struct l2cap_chan *chan, - struct l2cap_chan *new_chan); - int (*recv) (struct l2cap_chan * chan, - struct sk_buff *skb); - void (*teardown) (struct l2cap_chan *chan, int err); - void (*close) (struct l2cap_chan *chan); - void (*state_change) (struct l2cap_chan *chan, - int state, int err); - void (*ready) (struct l2cap_chan *chan); - void (*defer) (struct l2cap_chan *chan); - void (*resume) (struct l2cap_chan *chan); - void (*suspend) (struct l2cap_chan *chan); - void (*set_shutdown) (struct l2cap_chan *chan); - long (*get_sndtimeo) (struct l2cap_chan *chan); - struct pid *(*get_peer_pid) (struct l2cap_chan *chan); - struct sk_buff *(*alloc_skb) (struct l2cap_chan *chan, - unsigned long hdr_len, - unsigned long len, int nb); - int (*filter) (struct l2cap_chan * chan, - struct sk_buff *skb); -}; - struct l2cap_conn { struct hci_conn *hcon; struct hci_chan *hchan; @@ -674,6 +649,34 @@ struct l2cap_conn { struct list_head users; }; +struct l2cap_ops { + char *name; + + int (*new_connection)(struct l2cap_chan *chan, + struct l2cap_chan *new_chan); + int (*recv) (struct l2cap_chan * chan, + struct sk_buff *skb); + void (*teardown) (struct l2cap_chan *chan, int err) + __must_hold(&chan->lock); + void (*close) (struct l2cap_chan *chan); + void (*state_change) (struct l2cap_chan *chan, + int state, int err); + void (*ready) (struct l2cap_chan *chan) + __must_hold(&chan->lock) + __must_hold(&chan->conn->lock); + void (*defer) (struct l2cap_chan *chan); + void (*resume) (struct l2cap_chan *chan); + void (*suspend) (struct l2cap_chan *chan); + void (*set_shutdown) (struct l2cap_chan *chan); + long (*get_sndtimeo) (struct l2cap_chan *chan); + struct pid *(*get_peer_pid) (struct l2cap_chan *chan); + struct sk_buff *(*alloc_skb) (struct l2cap_chan *chan, + unsigned long hdr_len, + unsigned long len, int nb); + int (*filter) (struct l2cap_chan * chan, + struct sk_buff *skb); +}; + struct l2cap_user { struct list_head list; int (*probe) (struct l2cap_conn *conn, struct l2cap_user *user); @@ -983,7 +986,8 @@ void __l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan) typedef void (*l2cap_chan_func_t)(struct l2cap_chan *chan, void *data); void l2cap_chan_list(struct l2cap_conn *conn, l2cap_chan_func_t func, void *data); -void l2cap_chan_del(struct l2cap_chan *chan, int err); +void l2cap_chan_del(struct l2cap_chan *chan, int err) + __must_hold(&chan->lock) __must_hold(&chan->conn->lock); void l2cap_send_conn_req(struct l2cap_chan *chan); struct l2cap_conn *l2cap_conn_get(struct l2cap_conn *conn); diff --git a/net/bluetooth/6lowpan.c b/net/bluetooth/6lowpan.c index ddcdd2aff91fe2..836add41f5d16f 100644 --- a/net/bluetooth/6lowpan.c +++ b/net/bluetooth/6lowpan.c @@ -722,6 +722,8 @@ static int setup_netdev(struct l2cap_chan *chan, struct lowpan_btle_dev **dev) } static inline void chan_ready_cb(struct l2cap_chan *chan) + __must_hold(&chan->lock) + __must_hold(&chan->conn->lock) { struct lowpan_btle_dev *dev; bool new_netdev = false; diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 805623a48baef0..219d92431be060 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -59,7 +59,8 @@ static void l2cap_tx(struct l2cap_chan *chan, struct l2cap_ctrl *control, static void l2cap_retrans_timeout(struct work_struct *work); static void l2cap_monitor_timeout(struct work_struct *work); static void l2cap_ack_timeout(struct work_struct *work); -static void __l2cap_chan_close(struct l2cap_chan *chan, int reason); +static void __l2cap_chan_close(struct l2cap_chan *chan, int reason) + __must_hold(&chan->lock) __must_hold(&chan->conn->lock); static inline u8 bdaddr_type(u8 link_type, u8 bdaddr_type) { @@ -681,6 +682,8 @@ void l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan) void l2cap_chan_del(struct l2cap_chan *chan, int err) { + lockdep_assert(!chan->conn || lockdep_is_held(&chan->conn->lock)); + __clear_chan_timer(chan); BT_DBG("chan %p, err %d, state %s", chan, err, @@ -812,12 +815,11 @@ static void l2cap_chan_le_connect_reject(struct l2cap_chan *chan) } static void l2cap_chan_ecred_connect_reject(struct l2cap_chan *chan) + __must_hold(&chan->lock) + __must_hold(&chan->conn->lock) { l2cap_state_change(chan, BT_DISCONN); - lockdep_assert_held(&chan->lock); - lockdep_assert_held(&chan->conn->lock); - __l2cap_ecred_conn_rsp_defer(chan); } @@ -848,8 +850,6 @@ static void __l2cap_chan_close(struct l2cap_chan *chan, int reason) BT_DBG("chan %p state %s", chan, state_to_string(chan->state)); - lockdep_assert_held(&chan->lock); - switch (chan->state) { case BT_LISTEN: chan->ops->teardown(chan, 0); @@ -934,7 +934,10 @@ void l2cap_chan_close_unlocked(struct l2cap_chan *chan, int reason) bool have_conn; have_conn = l2cap_chan_lock_conn(chan); - __l2cap_chan_close(chan, reason); + + /* Context analysis: consider chan->conn->lock held also if conn NULL */ + context_unsafe(__l2cap_chan_close(chan, reason)); + l2cap_chan_unlock_conn(chan, have_conn); } EXPORT_SYMBOL(l2cap_chan_close_unlocked); @@ -1336,6 +1339,8 @@ void l2cap_send_conn_req(struct l2cap_chan *chan) } static void l2cap_chan_ready(struct l2cap_chan *chan) + __must_hold(&chan->lock) + __must_hold(&chan->conn->lock) { /* The channel may have already been flagged as connected in * case of receiving data before the L2CAP info req/rsp @@ -1471,6 +1476,7 @@ static void l2cap_ecred_connect(struct l2cap_chan *chan) } static void l2cap_le_start(struct l2cap_chan *chan) + __must_hold(&chan->lock) __must_hold(&chan->conn->lock) { struct l2cap_conn *conn = chan->conn; @@ -1492,6 +1498,7 @@ static void l2cap_le_start(struct l2cap_chan *chan) } static void l2cap_start_connection(struct l2cap_chan *chan) + __must_hold(&chan->lock) __must_hold(&chan->conn->lock) { if (chan->conn->hcon->type == LE_LINK) { @@ -1542,6 +1549,7 @@ static bool l2cap_check_enc_key_size(struct hci_conn *hcon, } static void l2cap_do_start(struct l2cap_chan *chan) + __must_hold(&chan->lock) __must_hold(&chan->conn->lock) { struct l2cap_conn *conn = chan->conn; @@ -1893,6 +1901,8 @@ static void l2cap_conn_del(struct hci_conn *hcon, int err) l2cap_chan_hold(chan); l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + l2cap_chan_del(chan, err); chan->ops->close(chan); @@ -4218,6 +4228,8 @@ static struct l2cap_chan *l2cap_new_connection(struct l2cap_conn *conn, __l2cap_chan_add(conn, chan); + lockdep_assert_held(&chan->conn->lock); + if (pchan->ops->new_connection && pchan->ops->new_connection(pchan, chan) < 0) { l2cap_chan_del(chan, 0); @@ -4419,6 +4431,8 @@ static int l2cap_connect_create_rsp(struct l2cap_conn *conn, l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + switch (result) { case L2CAP_CR_SUCCESS: if (__l2cap_get_chan_by_dcid(conn, dcid)) { @@ -4520,6 +4534,8 @@ static inline int l2cap_config_req(struct l2cap_conn *conn, l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + if (chan->state != BT_CONFIG && chan->state != BT_CONNECT2 && chan->state != BT_CONNECTED) { cmd_reject_invalid_cid(conn, cmd->ident, chan->scid, @@ -4634,6 +4650,8 @@ static inline int l2cap_config_rsp(struct l2cap_conn *conn, l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + switch (result) { case L2CAP_CONF_SUCCESS: l2cap_conf_rfc_get(chan, rsp->data, len); @@ -4743,6 +4761,8 @@ static inline int l2cap_disconnect_req(struct l2cap_conn *conn, l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + rsp.dcid = cpu_to_le16(chan->scid); rsp.scid = cpu_to_le16(chan->dcid); l2cap_send_cmd(conn, cmd->ident, L2CAP_DISCONN_RSP, sizeof(rsp), &rsp); @@ -4783,6 +4803,8 @@ static inline int l2cap_disconnect_rsp(struct l2cap_conn *conn, l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + if (chan->state != BT_DISCONN) { l2cap_chan_unlock(chan); l2cap_chan_put(chan); @@ -4995,6 +5017,8 @@ static int l2cap_le_connect_rsp(struct l2cap_conn *conn, l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + switch (result) { case L2CAP_CR_LE_SUCCESS: if (__l2cap_get_chan_by_dcid(conn, dcid)) { @@ -5213,6 +5237,8 @@ static int l2cap_le_connect_req(struct l2cap_conn *conn, l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + bacpy(&chan->src, &conn->hcon->src); bacpy(&chan->dst, &conn->hcon->dst); chan->src_type = bdaddr_src_type(conn->hcon); @@ -5444,6 +5470,8 @@ static inline int l2cap_ecred_conn_req(struct l2cap_conn *conn, l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + bacpy(&chan->src, &conn->hcon->src); bacpy(&chan->dst, &conn->hcon->dst); chan->src_type = bdaddr_src_type(conn->hcon); @@ -5533,6 +5561,8 @@ static inline int l2cap_ecred_conn_rsp(struct l2cap_conn *conn, l2cap_chan_hold(chan); l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + /* Check that there is a dcid for each pending channel */ if (cmd_len < sizeof(dcid)) { l2cap_chan_del(chan, ECONNREFUSED); @@ -5760,6 +5790,8 @@ static inline int l2cap_ecred_reconf_rsp(struct l2cap_conn *conn, continue; l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + l2cap_chan_del(chan, ECONNRESET); l2cap_chan_unlock(chan); @@ -5789,6 +5821,7 @@ static inline int l2cap_le_command_rej(struct l2cap_conn *conn, goto done; l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); l2cap_chan_del(chan, ECONNREFUSED); l2cap_chan_unlock(chan); l2cap_chan_put(chan); @@ -7176,6 +7209,8 @@ static void l2cap_data_channel(struct l2cap_conn *conn, u16 cid, l2cap_chan_lock(chan); + lockdep_assert_held(&chan->conn->lock); + BT_DBG("chan %p, len %d", chan, skb->len); /* If we receive data on a fixed channel before the info req/rsp From 4149ba2a806c46853122d1a7063748429d8fc415 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Wed, 2 Sep 2026 00:04:35 +0300 Subject: [PATCH 461/857] Bluetooth: L2CAP: annotate locking for l2cap_ops callbacks Annotate current locking context for l2cap_ops callbacks. Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- include/net/bluetooth/l2cap.h | 13 +++++++++---- net/bluetooth/l2cap_core.c | 1 + 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/include/net/bluetooth/l2cap.h b/include/net/bluetooth/l2cap.h index 3e1e2b36d7b6db..efb9b7f422d1d8 100644 --- a/include/net/bluetooth/l2cap.h +++ b/include/net/bluetooth/l2cap.h @@ -653,21 +653,26 @@ struct l2cap_ops { char *name; int (*new_connection)(struct l2cap_chan *chan, - struct l2cap_chan *new_chan); + struct l2cap_chan *new_chan) + __must_hold(&chan->lock) + __must_hold(&new_chan->lock); int (*recv) (struct l2cap_chan * chan, struct sk_buff *skb); void (*teardown) (struct l2cap_chan *chan, int err) __must_hold(&chan->lock); - void (*close) (struct l2cap_chan *chan); + void (*close) (struct l2cap_chan *chan) + __must_hold(&chan->lock); void (*state_change) (struct l2cap_chan *chan, int state, int err); void (*ready) (struct l2cap_chan *chan) __must_hold(&chan->lock) __must_hold(&chan->conn->lock); void (*defer) (struct l2cap_chan *chan); - void (*resume) (struct l2cap_chan *chan); + void (*resume) (struct l2cap_chan *chan) + __must_hold(&chan->lock); void (*suspend) (struct l2cap_chan *chan); - void (*set_shutdown) (struct l2cap_chan *chan); + void (*set_shutdown) (struct l2cap_chan *chan) + __must_hold(&chan->lock); long (*get_sndtimeo) (struct l2cap_chan *chan); struct pid *(*get_peer_pid) (struct l2cap_chan *chan); struct sk_buff *(*alloc_skb) (struct l2cap_chan *chan, diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index 219d92431be060..b7d5fa6f6a8338 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -4214,6 +4214,7 @@ static inline int l2cap_command_rej(struct l2cap_conn *conn, static struct l2cap_chan *l2cap_new_connection(struct l2cap_conn *conn, struct l2cap_chan *pchan) __must_hold(&conn->lock) + __must_hold(&pchan->lock) { struct l2cap_chan *chan; From 6696072ffe07205255cf83621a95a1aa2f9f6e62 Mon Sep 17 00:00:00 2001 From: Pauli Virtanen Date: Wed, 2 Sep 2026 00:04:36 +0300 Subject: [PATCH 462/857] Bluetooth: L2CAP: refuse __l2cap_chan_add if chan already has conn l2cap_chan may be linked to l2cap_conn at most once. This is assumed in several places, eg l2cap_chan_del cleanup. There is a TOCTOU race where the invariant is violated: [Task 1] [Task 2] l2cap_chan_connect l2cap_sock_bind l2cap_chan_lock lock_sock l2cap_state_change if (sk->sk_state != BT_OPEN) chan->state = BT_CONNECT l2cap_sock_state_change_cb chan->state = BT_BOUND sk->sk_state = BT_BOUND lock_sock <------------------ release_sock sk->sk_state = BT_CONNECT l2cap_sock_connect() does not check sk->sk_state, so since chan->state is now BT_BOUND, subsequent connect() ends up with second __l2cap_chan_add. Explicitly document and check the invariant in __l2cap_chan_add with WARN_ON_ONCE. The only callsite where it could be hit is l2cap_chan_connect, so add pre-check there to avoid relying on chan->state. chan->state read/write is not properly guarded currently so there can be other TOCTOUC problems. Add l2cap_lock_chan in l2cap_sock_bind() to guard chan->state write. Fixes: b66774b48dd9 ("Bluetooth: L2CAP: Fix UAF in channel timeout by holding conn ref") Assisted-by: deepseek-4-flash # finding the race condition Signed-off-by: Pauli Virtanen Signed-off-by: Luiz Augusto von Dentz --- net/bluetooth/l2cap_core.c | 7 ++++++- net/bluetooth/l2cap_sock.c | 2 ++ 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/net/bluetooth/l2cap_core.c b/net/bluetooth/l2cap_core.c index b7d5fa6f6a8338..a22edd2baf5e5d 100644 --- a/net/bluetooth/l2cap_core.c +++ b/net/bluetooth/l2cap_core.c @@ -623,6 +623,10 @@ void __l2cap_chan_add(struct l2cap_conn *conn, struct l2cap_chan *chan) BT_DBG("conn %p, psm 0x%2.2x, dcid 0x%4.4x", conn, __le16_to_cpu(chan->psm), chan->dcid); + /* Caller must ensure l2cap_chan is linked to l2cap_conn only once */ + if (WARN_ON_ONCE(chan->conn || test_bit(FLAG_DEL, &chan->flags))) + return; + conn->disc_reason = HCI_ERROR_REMOTE_USER_TERM; chan->conn = l2cap_conn_get(conn); @@ -7623,7 +7627,8 @@ int l2cap_chan_connect(struct l2cap_chan *chan, __le16 psm, u16 cid, } } - if (cid && __l2cap_get_chan_by_dcid(conn, cid)) { + if ((cid && __l2cap_get_chan_by_dcid(conn, cid)) || chan->conn || + test_bit(FLAG_DEL, &chan->flags)) { hci_conn_drop(hcon); err = -EBUSY; goto chan_unlock; diff --git a/net/bluetooth/l2cap_sock.c b/net/bluetooth/l2cap_sock.c index dee3025f0ec24f..278adb05c4c909 100644 --- a/net/bluetooth/l2cap_sock.c +++ b/net/bluetooth/l2cap_sock.c @@ -109,6 +109,7 @@ static int l2cap_sock_bind(struct socket *sock, struct sockaddr_unsized *addr, i return -EINVAL; } + l2cap_chan_lock(chan); lock_sock(sk); if (sk->sk_state != BT_OPEN) { @@ -174,6 +175,7 @@ static int l2cap_sock_bind(struct socket *sock, struct sockaddr_unsized *addr, i done: release_sock(sk); + l2cap_chan_unlock(chan); return err; } From 54c5771588b113938508a0b421340d38e07b37e1 Mon Sep 17 00:00:00 2001 From: "James C. Owens" Date: Fri, 14 Aug 2026 12:24:05 -0400 Subject: [PATCH 463/857] btrfs: scrub: report the failing sector's address, not the stripe base scrub_stripe_report_errors() iterates over the sectors of a stripe, but every message it emits passes stripe->logical, the address of the first sector of the 64KiB stripe, rather than the address of the sector being reported. The physical address is likewise computed once, before the loop, from stripe->logical. This matters because scrub_print_common_warning() uses that logical address for the backref walk which produces the "root %llu inode %llu offset %llu ... (path: ...)" part of the message. As the address is always the stripe base, the reported root/inode/offset/path can identify a different file from the one whose sector actually failed. A 64KiB stripe routinely spans several extents belonging to unrelated files. On the machine where this was found, the stripe at logical 0x17D9380000 holds four sectors of /usr/share/plasma/emoji/bg.dict, then a file inside a docker volume, then sectors referenced only by snapshots. Every error anywhere in that stripe is attributed to bg.dict. The effect is visible statistically: across ten months and four kernel series that machine logged 81 distinct flagged logical addresses, and every one of them is exactly 64KiB aligned. Since BTRFS_STRIPE_LEN is 64KiB and stripe->logical is stripe aligned by construction, real failures distributed across sectors could not produce that. Report the address of the sector actually being examined. Adding the sector offset to the physical address is valid because BTRFS_STRIPE_LEN is the unit contiguous on a single device for every profile, so a stripe never crosses a device boundary. Fixes: 0096580713ff ("btrfs: scrub: introduce error reporting functionality for scrub_stripe") Reviewed-by: Qu Wenruo Signed-off-by: James C. Owens Signed-off-by: David Sterba --- fs/btrfs/scrub.c | 24 ++++++++++++++---------- 1 file changed, 14 insertions(+), 10 deletions(-) diff --git a/fs/btrfs/scrub.c b/fs/btrfs/scrub.c index f209e75f0ff54d..c09d4213ad8915 100644 --- a/fs/btrfs/scrub.c +++ b/fs/btrfs/scrub.c @@ -1023,6 +1023,10 @@ static void scrub_stripe_report_errors(struct scrub_ctx *sctx, skip: for_each_set_bit(sector_nr, &extent_bitmap, stripe->nr_sectors) { + const u64 sector_logical = stripe->logical + + ((u64)sector_nr << fs_info->sectorsize_bits); + const u64 sector_physical = physical + + ((u64)sector_nr << fs_info->sectorsize_bits); bool repaired = false; if (scrub_bitmap_test_bit_is_metadata(stripe, sector_nr)) { @@ -1051,12 +1055,12 @@ static void scrub_stripe_report_errors(struct scrub_ctx *sctx, if (dev) { btrfs_err_rl(fs_info, "scrub: fixed up error at logical %llu on dev %s physical %llu", - stripe->logical, btrfs_dev_name(dev), - physical); + sector_logical, btrfs_dev_name(dev), + sector_physical); } else { btrfs_err_rl(fs_info, "scrub: fixed up error at logical %llu on mirror %u", - stripe->logical, stripe->mirror_num); + sector_logical, stripe->mirror_num); } continue; } @@ -1065,30 +1069,30 @@ static void scrub_stripe_report_errors(struct scrub_ctx *sctx, if (dev) { btrfs_err_rl(fs_info, "scrub: unable to fixup (regular) error at logical %llu on dev %s physical %llu", - stripe->logical, btrfs_dev_name(dev), - physical); + sector_logical, btrfs_dev_name(dev), + sector_physical); } else { btrfs_err_rl(fs_info, "scrub: unable to fixup (regular) error at logical %llu on mirror %u", - stripe->logical, stripe->mirror_num); + sector_logical, stripe->mirror_num); } if (scrub_bitmap_test_bit_io_error(stripe, sector_nr)) if (__ratelimit(&rs) && dev) scrub_print_common_warning("i/o error", dev, false, - stripe->logical, physical); + sector_logical, sector_physical); if (scrub_bitmap_test_bit_csum_error(stripe, sector_nr)) if (__ratelimit(&rs) && dev) scrub_print_common_warning("checksum error", dev, false, - stripe->logical, physical); + sector_logical, sector_physical); if (scrub_bitmap_test_bit_meta_error(stripe, sector_nr)) if (__ratelimit(&rs) && dev) scrub_print_common_warning("header error", dev, false, - stripe->logical, physical); + sector_logical, sector_physical); if (scrub_bitmap_test_bit_meta_gen_error(stripe, sector_nr)) if (__ratelimit(&rs) && dev) scrub_print_common_warning("generation error", dev, false, - stripe->logical, physical); + sector_logical, sector_physical); } /* Update the device stats. */ From 0f89a2afecff6506db7a566c84db137aa024076e Mon Sep 17 00:00:00 2001 From: Shuangpeng Bai Date: Sun, 16 Aug 2026 22:15:12 -0400 Subject: [PATCH 464/857] btrfs: fix transaction use-after-free in raid stripe insertion If allocation of a RAID stripe extent fails, btrfs_insert_one_raid_extent() aborts and ends the transaction before returning -ENOMEM. btrfs_finish_one_ordered(), the production caller through btrfs_insert_raid_extent(), still owns the transaction handle. It handles the error by aborting the transaction and then reaches the common exit path, which ends the transaction again. The premature end can free the handle and drop its transaction reference. Transaction cleanup can then free the transaction before the caller's second abort accesses the handle and transaction, resulting in use-after-free. Keep the abort at the failure site, but let the caller's common exit path end the transaction once, after it has finished using both objects. Fixes: 02c372e1f016 ("btrfs: add support for inserting raid stripe extents") Assisted-by: Codex:GPT-5 Reviewed-by: Qu Wenruo Signed-off-by: Shuangpeng Bai Signed-off-by: David Sterba --- fs/btrfs/raid-stripe-tree.c | 1 - 1 file changed, 1 deletion(-) diff --git a/fs/btrfs/raid-stripe-tree.c b/fs/btrfs/raid-stripe-tree.c index b210371ce91e38..89e259a47d8de1 100644 --- a/fs/btrfs/raid-stripe-tree.c +++ b/fs/btrfs/raid-stripe-tree.c @@ -337,7 +337,6 @@ int btrfs_insert_one_raid_extent(struct btrfs_trans_handle *trans, stripe_extent = kzalloc(item_size, GFP_NOFS); if (unlikely(!stripe_extent)) { btrfs_abort_transaction(trans, -ENOMEM); - btrfs_end_transaction(trans); return -ENOMEM; } From 3730a69e4eeeb07740cb0679b0229874e3ab9b62 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Mon, 17 Aug 2026 14:43:53 +0930 Subject: [PATCH 465/857] btrfs: fix the possible bioc_list memory leak during error There are two possible ways to leak bioc memory on btrfs_ordered_extent::bioc_list: - An error occurred for btrfs_insert_one_raid_extent() Then the function btrfs_insert_raid_extent() immediately return without freeing any bioc in the bioc_list. - An ordered extent hit an IO error In that case the ordered extent will have BTRFS_ORDERED_IOERR set, and skip the call on btrfs_insert_raid_extent() completely. Fix the problem by: - Introduce a new helper, btrfs_cleanup_ordered_bioc_list() Which will remove all bioc from the bioc_list, and release the bioc. - Call the above helper for btrfs_insert_raid_extent() So that the cleanup helper is always called no matter what. - Call the above helper for btrfs_finish_one_ordered() This is called just before the final release on the ordered extent. This was reported by Sashiko when reviewing another patch. Link: https://sashiko.dev/#/patchset/20260817021512.3010812-1-shuangpeng.kernel%40gmail.com Fixes: 02c372e1f016 ("btrfs: add support for inserting raid stripe extents") Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/inode.c | 3 +++ fs/btrfs/raid-stripe-tree.c | 18 ++++++++++++------ fs/btrfs/raid-stripe-tree.h | 1 + 3 files changed, 16 insertions(+), 6 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 3c10a0ef000231..93ef3cec191e36 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3436,6 +3436,9 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) */ btrfs_remove_ordered_extent(ordered_extent); + /* Cleanup any remaining biocs attached to the OE. */ + btrfs_cleanup_ordered_bioc_list(ordered_extent); + /* once for us */ btrfs_put_ordered_extent(ordered_extent); /* once for the tree */ diff --git a/fs/btrfs/raid-stripe-tree.c b/fs/btrfs/raid-stripe-tree.c index 89e259a47d8de1..6291775dbe0e72 100644 --- a/fs/btrfs/raid-stripe-tree.c +++ b/fs/btrfs/raid-stripe-tree.c @@ -373,7 +373,7 @@ int btrfs_insert_raid_extent(struct btrfs_trans_handle *trans, struct btrfs_ordered_extent *ordered_extent) { struct btrfs_io_context *bioc; - int ret; + int ret = 0; if (!btrfs_fs_incompat(trans->fs_info, RAID_STRIPE_TREE)) return 0; @@ -381,17 +381,23 @@ int btrfs_insert_raid_extent(struct btrfs_trans_handle *trans, list_for_each_entry(bioc, &ordered_extent->bioc_list, rst_ordered_entry) { ret = btrfs_insert_one_raid_extent(trans, bioc); if (ret) - return ret; + break; } - while (!list_empty(&ordered_extent->bioc_list)) { - bioc = list_first_entry(&ordered_extent->bioc_list, + btrfs_cleanup_ordered_bioc_list(ordered_extent); + return ret; +} + +void btrfs_cleanup_ordered_bioc_list(struct btrfs_ordered_extent *ordered) +{ + while (!list_empty(&ordered->bioc_list)) { + struct btrfs_io_context *bioc; + + bioc = list_first_entry(&ordered->bioc_list, typeof(*bioc), rst_ordered_entry); list_del(&bioc->rst_ordered_entry); btrfs_put_bioc(bioc); } - - return 0; } int btrfs_get_raid_extent_offset(struct btrfs_fs_info *fs_info, diff --git a/fs/btrfs/raid-stripe-tree.h b/fs/btrfs/raid-stripe-tree.h index 69942ad431408c..eb02cf48511bc9 100644 --- a/fs/btrfs/raid-stripe-tree.h +++ b/fs/btrfs/raid-stripe-tree.h @@ -28,6 +28,7 @@ int btrfs_get_raid_extent_offset(struct btrfs_fs_info *fs_info, u32 stripe_index, struct btrfs_io_stripe *stripe); int btrfs_insert_raid_extent(struct btrfs_trans_handle *trans, struct btrfs_ordered_extent *ordered_extent); +void btrfs_cleanup_ordered_bioc_list(struct btrfs_ordered_extent *ordered); #ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS int btrfs_insert_one_raid_extent(struct btrfs_trans_handle *trans, From 00eb519501276e5b85e1c92c4b3ec6d12b7b3c5a Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Mon, 17 Aug 2026 14:43:54 +0930 Subject: [PATCH 466/857] btrfs: return proper negative error code for update_raid_extent_item() The function btrfs_abort_transaction() only accepts negative error code, and have the macro VERIFY_NEGATIVE_ERROR() to verify that error code. But inside update_raid_extent_item(), if there is such key found, we return 1, breaking the negative error code scheme. Furthermore if we hit some real error during the tree search, e.g. -EIO, then the error code is always over-written to -EINVAL. Fix both problems by following other call sites by overwriting @ret to -ENOENT if the btrfs_search_slot() failed to locate the key. This is very unlikely to hit, as we only enter update_raid_extent_item() if there is a conflicting key already in the raid stripe tree. This was reported by Sashiko when reviewing another patch. Link: https://sashiko.dev/#/patchset/20260817021512.3010812-1-shuangpeng.kernel%40gmail.com Fixes: 8c4cba2adbb0 ("btrfs: update stripe extents for existing logical addresses") Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/raid-stripe-tree.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/raid-stripe-tree.c b/fs/btrfs/raid-stripe-tree.c index 6291775dbe0e72..d9e660447205f2 100644 --- a/fs/btrfs/raid-stripe-tree.c +++ b/fs/btrfs/raid-stripe-tree.c @@ -310,8 +310,10 @@ static int update_raid_extent_item(struct btrfs_trans_handle *trans, ret = btrfs_search_slot(trans, trans->fs_info->stripe_root, key, path, 0, 1); - if (ret) - return (ret == 1 ? ret : -EINVAL); + if (ret > 0) + ret = -ENOENT; + if (ret < 0) + return ret; leaf = path->nodes[0]; slot = path->slots[0]; From f136b4e7ecbd8a18ac84d866ec3fdce214223353 Mon Sep 17 00:00:00 2001 From: ZhengYuan Huang Date: Mon, 17 Aug 2026 21:20:51 +0800 Subject: [PATCH 467/857] btrfs: send: reject extents for non-regular inodes [BUG] A corrupted subvolume tree can leave an EXTENT_DATA item attached to an inode whose mode is not S_IFREG or S_IFLNK. During send, such an item can be treated as file data and crash through a NULL address_space operation: BUG: kernel NULL pointer dereference, address: 0000000000000000 #PF: supervisor instruction fetch in kernel mode #PF: error_code(0x0010) - not-present page Call Trace: read_pages+0x80b/0xb30 mm/readahead.c:173 page_cache_ra_unbounded+0x40d/0x890 mm/readahead.c:302 do_page_cache_ra mm/readahead.c:332 [inline] page_cache_ra_order+0xa16/0xcd0 mm/readahead.c:535 page_cache_sync_ra+0x5ce/0x9d0 mm/readahead.c:626 page_cache_sync_readahead include/linux/pagemap.h:1379 [inline] put_file_data fs/btrfs/send.c:5224 [inline] send_write fs/btrfs/send.c:5291 [inline] send_extent_data+0x16b2/0x29b0 fs/btrfs/send.c:5715 send_write_or_clone fs/btrfs/send.c:6135 [inline] process_extent+0x5d4/0x17b0 fs/btrfs/send.c:6504 changed_extent fs/btrfs/send.c:7079 [inline] changed_cb+0x22f9/0x3cd0 fs/btrfs/send.c:7245 full_send_tree fs/btrfs/send.c:7318 [inline] send_subvol fs/btrfs/send.c:7910 [inline] btrfs_ioctl_send+0x46a9/0x57f0 fs/btrfs/send.c:8248 ... [CAUSE] process_extent() skips extent items for symlinks but assumes every other inode with an extent item is a regular file. For a corrupted non-regular inode, btrfs_iget() does not install the regular file address_space operations. The readahead fallback can then call a NULL read_folio callback before the existing validation in btrfs_get_extent() can run. [FIX] Reject extent items for inode types other than regular files and symlinks at the common send extent-processing boundary. Symlink handling is left unchanged because send emits symlink data from read_symlink(). This covers full, incremental and new-generation sends without adding a check to the regular I/O path. Reviewed-by: Qu Wenruo Signed-off-by: ZhengYuan Huang Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/send.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c index dca3570168c76b..f88623bbc491d6 100644 --- a/fs/btrfs/send.c +++ b/fs/btrfs/send.c @@ -6417,6 +6417,13 @@ static int process_extent(struct send_ctx *sctx, if (S_ISLNK(sctx->cur_inode_mode)) return 0; + if (unlikely(!S_ISREG(sctx->cur_inode_mode))) { + btrfs_crit(sctx->send_root->fs_info, + "send: extent for non-regular inode %llu root %llu mode 0%llo", + key->objectid, btrfs_root_id(sctx->send_root), + sctx->cur_inode_mode & S_IFMT); + return -EUCLEAN; + } if (sctx->parent_root && !sctx->cur_inode_new) { ret = is_extent_unchanged(sctx, path, key); From daee9e62cfb1ab18866a9ea8c17dc028d695b44b Mon Sep 17 00:00:00 2001 From: Johannes Thumshirn Date: Wed, 19 Aug 2026 12:26:36 +0200 Subject: [PATCH 468/857] btrfs: zoned: finish active block group cleanup if call_zone_finish() fails do_zone_finish() clears BLOCK_GROUP_FLAG_ZONE_IS_ACTIVE before finishing the zones. If call_zone_finish() then fails it returned early, leaving the now inactive block group on fs_info->zone_active_bgs, leaking its reference, the BTRFS_FS_NEED_ZONE_FINISH waiters are never woken, and as its alloc_offset equals the zone capacity btrfs_zone_finish_one_bg() keeps selecting it, spinning btrfs_zoned_activate_one_bg(). Fall through to the cleanup on failure too and return the error, but keep the block group read-only as its zones are left inconsistent. Fixes: d70cbdda75da ("btrfs: zoned: consolidate zone finish functions") Link: https://sashiko.dev/#/patchset/20260818100037.1366563-1-johannes.thumshirn%40wdc.com Reviewed-by: Qu Wenruo Signed-off-by: Johannes Thumshirn Signed-off-by: David Sterba --- fs/btrfs/zoned.c | 11 ++++------- 1 file changed, 4 insertions(+), 7 deletions(-) diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index a016cb471beb47..7f0dde6398d4d3 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -2626,16 +2626,13 @@ static int do_zone_finish(struct btrfs_block_group *block_group, bool fully_writ down_read(&dev_replace->rwsem); map = block_group->physical_map; for (i = 0; i < map->num_stripes; i++) { - ret = call_zone_finish(block_group, &map->stripes[i]); - if (ret) { - up_read(&dev_replace->rwsem); - return ret; - } + if (ret) + break; } up_read(&dev_replace->rwsem); - if (!fully_written) + if (!ret && !fully_written) btrfs_dec_block_group_ro(block_group); spin_lock(&fs_info->zone_active_bgs_lock); @@ -2648,7 +2645,7 @@ static int do_zone_finish(struct btrfs_block_group *block_group, bool fully_writ clear_and_wake_up_bit(BTRFS_FS_NEED_ZONE_FINISH, &fs_info->flags); - return 0; + return ret; } int btrfs_zone_finish(struct btrfs_block_group *block_group) From 27c3d95091ce28a8c616bd7a338e03d952bc9a00 Mon Sep 17 00:00:00 2001 From: Johannes Thumshirn Date: Tue, 18 Aug 2026 12:00:37 +0200 Subject: [PATCH 469/857] btrfs: zoned: propagate do_zone_finish() error in btrfs_zone_finish_endio() btrfs_zone_finish_endio() ignored the return value of do_zone_finish() and always returned 0, silently dropping a failed zone finish. Instead propagate any error from do_zone_finish() as the caller btrfs_finish_ordered_io() already handles it. Reviewed-by: Qu Wenruo Signed-off-by: Johannes Thumshirn Signed-off-by: David Sterba --- fs/btrfs/zoned.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index 7f0dde6398d4d3..9cc2c9c1a606b6 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -2710,6 +2710,7 @@ int btrfs_zone_finish_endio(struct btrfs_fs_info *fs_info, u64 logical, u64 leng { struct btrfs_block_group *block_group; u64 min_alloc_bytes; + int ret = 0; if (!btrfs_is_zoned(fs_info)) return 0; @@ -2729,11 +2730,11 @@ int btrfs_zone_finish_endio(struct btrfs_fs_info *fs_info, u64 logical, u64 leng block_group->start + block_group->zone_capacity) goto out; - do_zone_finish(block_group, true); + ret = do_zone_finish(block_group, true); out: btrfs_put_block_group(block_group); - return 0; + return ret; } static void btrfs_zone_finish_endio_workfn(struct work_struct *work) From 86ce1e8bd5bd1b2f61aa82fc0104e51168a1a529 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Mon, 17 Aug 2026 17:00:43 +0930 Subject: [PATCH 470/857] btrfs: refactor read_key_bytes() to remove the dest_folio parameter The function read_key_bytes() have 3 call sites: - For BTRFS_VERITY_DESC_ITEM_KEY offset 0 inside btrfs_get_verity_descriptor() - For BTRFS_VERITY_DESC_ITEM_KEY offset 1 inside btrfs_get_verity_descriptor() Those are to read the description items, which are pretty small with fixed item size. Those call sites do not utilize the @dest_folio parameter. - For btrfs_read_merkle_tree_page() This is to read the BTRFS_VERITY_MERKLE_ITEM_KEY, which can be pretty large and split into multiple items. This is the only call site utilizing the @dest_folio parameter. Just for the only btrfs_read_merkle_tree_page() call site, we have a complex scheme for @dest and @dest_folio parameters. Since @dest can be NULL, it means if we pass @dest as NULL, then no matter if @dest_folio is provided, the merkle data will not be loaded into that @dest_folio. This can lead to a bug where a highmem folio is not mapped, then we pass folio_address(folio), which is NULL, into read_key_bytes(), causing no data to be written into @dest_folio. To address the complex scheme between @dest and @dest_folio, remove the @dest_folio parameter completely, and let the only caller to map the folio and pass the mapped kernel address into read_key_bytes() instead. This not only reduces the parameter list, but also make it much clear on the @dest parameter handling. The only downside is a longer duration of locally mapped page, but this should still be fine, as kmap_local_folio() can survive context switch. Reported-by: Hongling Zeng Link: https://lore.kernel.org/linux-btrfs/20260817022012.19658-1-zenghongling@kylinos.cn/ Fixes: 146054090b08 ("btrfs: initial fsverity support") Reviewed-by: David Sterba Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/verity.c | 31 ++++++++++++++----------------- 1 file changed, 14 insertions(+), 17 deletions(-) diff --git a/fs/btrfs/verity.c b/fs/btrfs/verity.c index 4e0ab584227419..d432fec21c15bb 100644 --- a/fs/btrfs/verity.c +++ b/fs/btrfs/verity.c @@ -272,21 +272,17 @@ static int write_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, * @dest: Buffer to read into. This parameter has slightly tricky * semantics. If it is NULL, the function will not do any copying * and will just return the size of all the items up to len bytes. - * If dest_page is passed, then the function will kmap_local the - * page and ignore dest, but it must still be non-NULL to avoid the - * counting-only behavior. * @len: length in bytes to read - * @dest_folio: copy into this folio instead of the dest buffer * * Helper function to read items from the btree. This returns the number of * bytes read or < 0 for errors. We can return short reads if the items don't * exist on disk or aren't big enough to fill the desired length. Supports - * reading into a provided buffer (dest) or into the page cache + * reading into a provided buffer (dest). * * Returns number of bytes read or a negative error code on failure. */ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, - char *dest, u64 len, struct folio *dest_folio) + char *dest, u64 len) { BTRFS_PATH_AUTO_FREE(path); struct btrfs_root *root = inode->root; @@ -306,7 +302,11 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, if (!path) return -ENOMEM; - if (dest_folio) + /* + * Merkle items can be large and split across multiple items, so enable + * readahead for such cases. + */ + if (key_type == BTRFS_VERITY_MERKLE_ITEM_KEY) path->reada = READA_FORWARD; key.objectid = btrfs_ino(inode); @@ -350,7 +350,7 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, break; } - /* desc = NULL to just sum all the item lengths */ + /* dest == NULL to just sum all the item lengths */ if (!dest) copy_end = item_end; else @@ -363,16 +363,10 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, copy_offset = offset - key.offset; if (dest) { - if (dest_folio) - kaddr = kmap_local_folio(dest_folio, 0); - data = btrfs_item_ptr(leaf, path->slots[0], void); read_extent_buffer(leaf, kaddr + dest_offset, (unsigned long)data + copy_offset, copy_bytes); - - if (dest_folio) - kunmap_local(kaddr); } offset += copy_bytes; @@ -663,7 +657,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size) memset(&item, 0, sizeof(item)); ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 0, - (char *)&item, sizeof(item), NULL); + (char *)&item, sizeof(item)); if (ret < 0) return ret; @@ -680,7 +674,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size) return -ERANGE; ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 1, - buf, buf_size, NULL); + buf, buf_size); if (ret < 0) return ret; if (ret != true_size) @@ -706,6 +700,7 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, struct folio *folio; u64 off = (u64)index << PAGE_SHIFT; loff_t merkle_pos = merkle_file_pos(inode); + void *kaddr; int ret; if (merkle_pos < 0) @@ -749,6 +744,7 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, } read_folio: + kaddr = kmap_local_folio(folio, 0); /* * Merkle item keys are indexed from byte 0 in the merkle tree. * They have the form: @@ -756,7 +752,8 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, * [ inode objectid, BTRFS_MERKLE_ITEM_KEY, offset in bytes ] */ ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_MERKLE_ITEM_KEY, off, - folio_address(folio), PAGE_SIZE, folio); + kaddr, PAGE_SIZE); + kunmap_local(kaddr); if (ret < 0) { folio_unlock(folio); folio_put(folio); From ece9e2904072c603ec30638949477474d4ca533d Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:16 +0930 Subject: [PATCH 471/857] btrfs: replace btrfs_repair_io_failure() to use bio for page iteration Currently btrfs_repair_io_failure() uses a @paddrs[] array to iterate pages. Such a parameter is required for bs > ps cases, as one fs block crosses several pages. However there is a much simpler and existing way to iterate pages: bio and bvec_iter. This changes btrfs_repair_io_failure() by: - Use a const @bvec_iter pointer to locate where the pages are - Extract file offset/logical from the @bbio - Require no @step parameter Above features allow us to shorten the parameter list. - Rename the function to btrfs_repair_bbio_failure() - Change the caller in btrfs_repair_eb_io_failure() to allocate a bbio Unlike the data read path, we do not have a handy bbio in that case. So we need to allocate one just for btrfs_repair_bbio_failure(). - Change the error reporting in btrfs_repair_bbio_failure() to include root id and use inode number directly Now for btree inode we will report a proper inode number (1). Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/bio.c | 61 ++++++++++++++++++++++++++-------------------- fs/btrfs/bio.h | 5 ++-- fs/btrfs/disk-io.c | 25 +++++++++++++------ 3 files changed, 55 insertions(+), 36 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index cc0bd03048bae6..f8d4c2d550073a 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -186,7 +186,6 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, */ struct bvec_iter saved_iter = repair_bbio->saved_iter; const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u64 logical = repair_bbio->saved_iter.bi_sector << SECTOR_SHIFT; const u32 nr_steps = repair_bbio->saved_iter.bi_size / step; int mirror = repair_bbio->mirror_num; phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; @@ -220,9 +219,8 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, do { mirror = prev_repair_mirror(fbio, mirror); - btrfs_repair_io_failure(fs_info, btrfs_ino(inode), - repair_bbio->file_offset, fs_info->sectorsize, - logical, paddrs, step, mirror); + btrfs_repair_bbio_failure(repair_bbio, &repair_bbio->saved_iter, + fs_info->sectorsize, mirror); } while (mirror != fbio->bbio->mirror_num); done: @@ -925,21 +923,23 @@ void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num) * The I/O is issued synchronously to block the repair read completion from * freeing the bio. * - * @ino: Offending inode number - * @fileoff: File offset inside the inode + * @bbio: Original bbio where the repair is needed + * @orig_iter: Points to where the repair start is * @length: Length of the repair write - * @logical: Logical address of the range - * @paddrs: Physical address array of the content - * @step: Length of for each paddrs * @mirror_num: Mirror number to write to. Must not be zero */ -int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, - u32 length, u64 logical, const phys_addr_t paddrs[], - unsigned int step, int mirror_num) +int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + u32 length, int mirror_num) { - const u32 nr_steps = DIV_ROUND_UP_POW2(length, step); + struct btrfs_inode *inode = bbio->inode; + struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_io_stripe smap = { 0 }; - struct bio *bio = NULL; + struct bvec_iter iter = *orig_iter; + struct bio *repair_bio = NULL; + const u64 logical = iter.bi_sector << SECTOR_SHIFT; + const u64 fileoff = bbio->file_offset + + ((iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT); + u32 cur = 0; int ret = 0; BUG_ON(!mirror_num); @@ -950,8 +950,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, ASSERT(IS_ALIGNED(fileoff, fs_info->sectorsize)); /* Either it's a single data or metadata block. */ ASSERT(length <= BTRFS_MAX_BLOCKSIZE); - ASSERT(step <= length); - ASSERT(is_power_of_2(step)); + + /* Our current iter should not be before the original bbio saved_iter. */ + ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector); /* * The fs either mounted RO or hit critical errors, no need @@ -979,15 +980,22 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, goto out_counter_dec; } - bio = bio_alloc(smap.dev->bdev, nr_steps, REQ_OP_WRITE | REQ_SYNC, GFP_NOFS); - bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT; - for (int i = 0; i < nr_steps; i++) { - ret = bio_add_page(bio, phys_to_page(paddrs[i]), step, offset_in_page(paddrs[i])); - /* We should have allocated enough slots to contain all the different pages. */ - ASSERT(ret == step); + repair_bio = bio_alloc(smap.dev->bdev, max(1, length >> PAGE_SHIFT), + REQ_OP_WRITE | REQ_SYNC, GFP_NOFS); + repair_bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT; + while (cur < length) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), length - cur); + + ret = bio_add_page(repair_bio, page, cur_len, pg_off); + ASSERT(ret == cur_len); + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; } - ret = submit_bio_wait(bio); - bio_put(bio); + + ret = submit_bio_wait(repair_bio); + bio_put(repair_bio); if (ret) { /* try to remap that extent elsewhere? */ btrfs_dev_stat_inc_and_print(smap.dev, BTRFS_DEV_STAT_WRITE_ERRS); @@ -995,8 +1003,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, } btrfs_info_rl(fs_info, - "read error corrected: ino %llu off %llu (dev %s sector %llu)", - ino, fileoff, btrfs_dev_name(smap.dev), + "read error corrected: root %llu ino %llu off %llu (dev %s sector %llu)", + btrfs_root_id(inode->root), btrfs_ino(inode), fileoff, + btrfs_dev_name(smap.dev), smap.physical >> SECTOR_SHIFT); ret = 0; diff --git a/fs/btrfs/bio.h b/fs/btrfs/bio.h index 303ed6c7103d92..b7bd377a016249 100644 --- a/fs/btrfs/bio.h +++ b/fs/btrfs/bio.h @@ -126,8 +126,7 @@ void btrfs_bio_end_io(struct btrfs_bio *bbio, blk_status_t status); void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num); void btrfs_submit_repair_write(struct btrfs_bio *bbio, int mirror_num, bool dev_replace); -int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, - u32 length, u64 logical, const phys_addr_t paddrs[], - unsigned int step, int mirror_num); +int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + u32 length, int mirror_num); #endif diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 819727460bcf4d..466fadb1815a81 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -176,19 +176,24 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb, int mirror_num) { struct btrfs_fs_info *fs_info = eb->fs_info; - const u32 step = min(fs_info->nodesize, PAGE_SIZE); - const u32 nr_steps = eb->len / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; + struct btrfs_bio *bbio; + int ret; if (sb_rdonly(fs_info->sb)) return -EROFS; + /* + * This bbio is only to queue all pages for btrfs_repair_bbio_failure(). + * Thus it will never get its endio called. + */ + bbio = btrfs_bio_alloc(max(1, fs_info->nodesize >> PAGE_SHIFT), REQ_OP_READ, + BTRFS_I(fs_info->btree_inode), eb->start, NULL, NULL); + bbio->bio.bi_iter.bi_sector = eb->start >> SECTOR_SHIFT; for (int i = 0; i < num_extent_pages(eb); i++) { struct folio *folio = eb->folios[i]; /* No large folio support yet. */ ASSERT(folio_order(folio) == 0); - ASSERT(i < nr_steps); /* * For nodesize < page size, there is just one paddr, with some @@ -197,11 +202,17 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb, * For nodesize >= page size, it's one or more paddrs, and eb->start * must be aligned to page boundary. */ - paddrs[i] = page_to_phys(&folio->page) + offset_in_page(eb->start); + ret = bio_add_page(&bbio->bio, &folio->page, min(PAGE_SIZE, fs_info->nodesize), + offset_in_page(eb->start)); + ASSERT(ret == min(PAGE_SIZE, fs_info->nodesize)); } + /* Since the bbio is never submitted, we have to save the iter manually. */ + bbio->saved_iter = bbio->bio.bi_iter; - return btrfs_repair_io_failure(fs_info, 0, eb->start, eb->len, - eb->start, paddrs, step, mirror_num); + ret = btrfs_repair_bbio_failure(bbio, &bbio->saved_iter, fs_info->nodesize, + mirror_num); + bio_put(&bbio->bio); + return ret; } /* From 8a5d77f319793d25d5859736c858973a6a85dbd4 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:17 +0930 Subject: [PATCH 472/857] btrfs: enhance btrfs_data_csum_ok() to use bio for page iteration Currently btrfs_data_csum_ok() requires a @paddr[] array to iterate all possible pages for bs > ps cases. However for all btrfs_data_csum_ok() call sites, we already have a btrfs_bio, and the bio infrastructure has many flexible ways to iterate multiple pages already. Change btrfs_data_csum_ok() to make full use of btrfs_bio by: - Change the parameter list to require a @bvec_iter pointer And remove @bio_offset, which can be calculated through @bvec_iter and bbio->saved_iter. Also remove paddrs[], we will iterate all the pages using bio interfaces. - Make the same parameter changes to repair_one_sector() - Use bio interfaces to iterate pages from a bio - Rename the function to btrfs_bio_data_csum_ok() - Remove on-stack paddrs[] array usage Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/bio.c | 82 +++++++++++++++--------------------------- fs/btrfs/btrfs_inode.h | 4 +-- fs/btrfs/inode.c | 49 +++++++++++++++++++------ 3 files changed, 70 insertions(+), 65 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index f8d4c2d550073a..19b4855969f536 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -180,29 +180,13 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, struct btrfs_failed_bio *fbio = repair_bbio->private; struct btrfs_inode *inode = repair_bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - /* - * We can not move forward the saved_iter, as it will be later - * utilized by repair_bbio again. - */ - struct bvec_iter saved_iter = repair_bbio->saved_iter; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = repair_bbio->saved_iter.bi_size / step; int mirror = repair_bbio->mirror_num; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; - unsigned int slot = 0; - /* Repair bbio should be eaxctly one block sized. */ + /* Repair bbio should be exactly one block sized. */ ASSERT(repair_bbio->saved_iter.bi_size == fs_info->sectorsize); - btrfs_bio_for_each_block(paddr, &repair_bbio->bio, &saved_iter, step) { - ASSERT(slot < nr_steps); - paddrs[slot] = paddr; - slot++; - } - if (repair_bbio->bio.bi_status || - !btrfs_data_csum_ok(repair_bbio, dev, 0, paddrs)) { + !btrfs_bio_data_csum_ok(repair_bbio, &repair_bbio->saved_iter, dev)) { bio_reset(&repair_bbio->bio, NULL, REQ_OP_READ); repair_bbio->bio.bi_iter = repair_bbio->saved_iter; @@ -236,25 +220,21 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, * read succeeded to restore the redundancy. */ static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio, - u32 bio_offset, - phys_addr_t paddrs[], + const struct bvec_iter *orig_iter, struct btrfs_failed_bio *fbio) { struct btrfs_inode *inode = failed_bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - const u32 sectorsize = fs_info->sectorsize; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = sectorsize / step; - /* - * For bs > ps cases, the saved_iter can be partially moved forward. - * In that case we should round it down to the block boundary. - */ - const u64 logical = round_down(failed_bbio->saved_iter.bi_sector << SECTOR_SHIFT, - sectorsize); struct btrfs_bio *repair_bbio; struct bio *repair_bio; + struct bvec_iter iter = *orig_iter; + const u32 sectorsize = fs_info->sectorsize; + const u32 bio_offset = ((iter.bi_sector - failed_bbio->saved_iter.bi_sector) << + SECTOR_SHIFT); + const u64 logical = (iter.bi_sector << SECTOR_SHIFT); int num_copies; int mirror; + u32 cur = 0; btrfs_debug(fs_info, "repair read error: read error at %llu", failed_bbio->file_offset + bio_offset); @@ -275,17 +255,21 @@ static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio, atomic_inc(&fbio->repair_count); - repair_bio = bio_alloc_bioset(NULL, nr_steps, REQ_OP_READ, GFP_NOFS, - &btrfs_repair_bioset); + repair_bio = bio_alloc_bioset(NULL, max(1, sectorsize >> PAGE_SHIFT), + REQ_OP_READ, GFP_NOFS, &btrfs_repair_bioset); repair_bio->bi_iter.bi_sector = logical >> SECTOR_SHIFT; - for (int i = 0; i < nr_steps; i++) { + while (cur < sectorsize) { + struct page *page = bio_iter_page(&failed_bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&failed_bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&failed_bbio->bio, iter), + sectorsize - cur); int ret; - ASSERT(offset_in_page(paddrs[i]) + step <= PAGE_SIZE); + ret = bio_add_page(repair_bio, page, cur_len, pg_off); + ASSERT(ret == cur_len); - ret = bio_add_page(repair_bio, phys_to_page(paddrs[i]), step, - offset_in_page(paddrs[i])); - ASSERT(ret == step); + bio_advance_iter_single(&failed_bbio->bio, &iter, cur_len); + cur += cur_len; } repair_bbio = btrfs_bio(repair_bio); @@ -303,18 +287,16 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; const u32 sectorsize = fs_info->sectorsize; - const u32 step = min(sectorsize, PAGE_SIZE); - const u32 nr_steps = sectorsize / step; - struct bvec_iter *iter = &bbio->saved_iter; + struct bvec_iter iter; blk_status_t status = bbio->bio.bi_status; struct btrfs_failed_bio *fbio = NULL; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; - u32 offset = 0; /* Read-repair requires the inode field to be set by the submitter. */ ASSERT(inode); + /* The original bbio should be sectorsize aligned. */ + ASSERT(IS_ALIGNED(bbio->saved_iter.bi_size, sectorsize)); + /* * Hand off repair bios to the repair code as there is no upper level * submitter for them. @@ -327,16 +309,10 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de /* Clear the I/O error. A failed repair will reset it. */ bbio->bio.bi_status = BLK_STS_OK; - btrfs_bio_for_each_block(paddr, &bbio->bio, iter, step) { - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; - - if (IS_ALIGNED(offset, sectorsize)) { - if (status || - !btrfs_data_csum_ok(bbio, dev, offset - sectorsize, paddrs)) - fbio = repair_one_sector(bbio, offset - sectorsize, - paddrs, fbio); - } + for (iter = bbio->saved_iter; iter.bi_size; + bio_advance_iter(&bbio->bio, &iter, sectorsize)) { + if (status || !btrfs_bio_data_csum_ok(bbio, &iter, dev)) + fbio = repair_one_sector(bbio, &iter, fbio); } if (bbio->csum != bbio->csum_inline) kvfree(bbio->csum); @@ -924,7 +900,7 @@ void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num) * freeing the bio. * * @bbio: Original bbio where the repair is needed - * @orig_iter: Points to where the repair start is + * @orig_iter: Points to where the repair starts * @length: Length of the repair write * @mirror_num: Mirror number to write to. Must not be zero */ diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 1082fa92c1457a..171f96bdb8aa76 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -513,8 +513,8 @@ void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, const phys_addr_t paddrs[], u8 *dest); int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, const u8 * const csum_expected); -bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, - u32 bio_offset, const phys_addr_t paddrs[]); +bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + struct btrfs_device *dev); noinline int can_nocow_extent(struct btrfs_inode *inode, u64 offset, u64 *len, struct btrfs_file_extent *file_extent, bool nowait); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 93ef3cec191e36..005f8f9da8b139 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3536,27 +3536,31 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 * different noncontiguous pages. * * @bbio: btrfs_io_bio which contains the csum - * @dev: device the sector is on - * @bio_offset: offset to the beginning of the bio (in bytes) - * @paddrs: physical addresses which back the fs block + * @orig_iter: bvec iter pointing to the start of the block + * @dev: device the sector is on (optional) * * Check if the checksum on a data block is valid. When a checksum mismatch is * detected, report the error and fill the corrupted range with zero. * * Return %true if the sector is ok or had no checksum to start with, else %false. */ -bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, - u32 bio_offset, const phys_addr_t paddrs[]) +bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, + const struct bvec_iter *orig_iter, + struct btrfs_device *dev) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; + struct bvec_iter iter = *orig_iter; + struct btrfs_csum_ctx cctx; const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; + const u32 bio_offset = (iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT; u64 file_offset = bbio->file_offset + bio_offset; u64 end = file_offset + blocksize - 1; u8 *csum_expected; u8 csum[BTRFS_CSUM_SIZE]; + u32 cur = 0; + + ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector); if (!bbio->csum) return true; @@ -3572,7 +3576,22 @@ bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, csum_expected = bbio->csum + (bio_offset >> fs_info->sectorsize_bits) * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum); + btrfs_csum_init(&cctx, fs_info->csum_type); + while (cur < blocksize) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); + void *kaddr; + + kaddr = kmap_local_page(page) + pg_off; + btrfs_csum_update(&cctx, kaddr, cur_len); + kunmap_local(kaddr); + + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; + } + btrfs_csum_final(&cctx, csum); + if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) goto zeroit; return true; @@ -3582,8 +3601,18 @@ bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, bbio->mirror_num); if (dev) btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_CORRUPTION_ERRS); - for (int i = 0; i < nr_steps; i++) - memzero_page(phys_to_page(paddrs[i]), offset_in_page(paddrs[i]), step); + cur = 0; + iter = *orig_iter; + while (cur < blocksize) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); + + memzero_page(page, pg_off, cur_len); + + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; + } return false; } From fa00ac8a9873e6cbf86a7108897637c4cc464a2b Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:18 +0930 Subject: [PATCH 473/857] btrfs: use a shared helper to calculate data checksum for a bio Since we are already calculating data checksum using bio interface, extract the generation part into btrfs_csum_one_bio_block(), and use that to replace the paddrs[] array based solution in csum_one_bio(). This will reduce 128 bytes on-stack memory usage for csum_one_bio(). Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 2 ++ fs/btrfs/file-item.c | 19 +++++------------ fs/btrfs/inode.c | 46 +++++++++++++++++++++++++----------------- 3 files changed, 34 insertions(+), 33 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 171f96bdb8aa76..e137a99151bd07 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -515,6 +515,8 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 const u8 * const csum_expected); bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, struct btrfs_device *dev); +void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, + const struct bvec_iter *orig_iter, u8 *csum); noinline int can_nocow_extent(struct btrfs_inode *inode, u64 offset, u64 *len, struct btrfs_file_extent *file_extent, bool nowait); diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index cf50fd623f41a8..581ca5653be93a 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -801,25 +801,16 @@ static void csum_one_bio(struct btrfs_bio *bbio, struct bvec_iter *src) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct bio *bio = &bbio->bio; struct btrfs_ordered_sum *sums = bbio->sums; - struct bvec_iter iter = *src; - phys_addr_t paddr; + struct bvec_iter iter; const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - u32 offset = 0; int index = 0; - btrfs_bio_for_each_block(paddr, bio, &iter, step) { - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; + for (iter = *src; iter.bi_size; bio_advance_iter(&bbio->bio, &iter, blocksize)) { + btrfs_csum_one_bio_block(fs_info, &bbio->bio, &iter, + sums->sums + index); - if (IS_ALIGNED(offset, blocksize)) { - btrfs_calculate_block_csum_pages(fs_info, paddrs, sums->sums + index); - index += fs_info->csum_size; - } + index += fs_info->csum_size; } } diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 005f8f9da8b139..11f8aad8601f50 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3531,6 +3531,32 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 return 0; } +/* Generate data checksum for a single fs block, pointed to by @orig_iter. */ +void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, + const struct bvec_iter *orig_iter, u8 *csum) +{ + struct btrfs_csum_ctx cctx; + struct bvec_iter iter = *orig_iter; + const u32 blocksize = fs_info->sectorsize; + u32 cur = 0; + + btrfs_csum_init(&cctx, fs_info->csum_type); + while (cur < blocksize) { + struct page *page = bio_iter_page(bio, iter); + const u32 pg_off = bio_iter_offset(bio, iter); + const u32 cur_len = min(bio_iter_len(bio, iter), blocksize - cur); + void *kaddr; + + kaddr = kmap_local_page(page) + pg_off; + btrfs_csum_update(&cctx, kaddr, cur_len); + kunmap_local(kaddr); + + bio_advance_iter_single(bio, &iter, cur_len); + cur += cur_len; + } + btrfs_csum_final(&cctx, csum); +} + /* * Verify the checksum of a single data sector, which can be scattered at * different noncontiguous pages. @@ -3551,7 +3577,6 @@ bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; struct bvec_iter iter = *orig_iter; - struct btrfs_csum_ctx cctx; const u32 blocksize = fs_info->sectorsize; const u32 bio_offset = (iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT; u64 file_offset = bbio->file_offset + bio_offset; @@ -3576,22 +3601,7 @@ bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, csum_expected = bbio->csum + (bio_offset >> fs_info->sectorsize_bits) * fs_info->csum_size; - btrfs_csum_init(&cctx, fs_info->csum_type); - while (cur < blocksize) { - struct page *page = bio_iter_page(&bbio->bio, iter); - const u32 pg_off = bio_iter_offset(&bbio->bio, iter); - const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); - void *kaddr; - - kaddr = kmap_local_page(page) + pg_off; - btrfs_csum_update(&cctx, kaddr, cur_len); - kunmap_local(kaddr); - - bio_advance_iter_single(&bbio->bio, &iter, cur_len); - cur += cur_len; - } - btrfs_csum_final(&cctx, csum); - + btrfs_csum_one_bio_block(fs_info, &bbio->bio, orig_iter, csum); if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) goto zeroit; return true; @@ -3601,8 +3611,6 @@ bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, bbio->mirror_num); if (dev) btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_CORRUPTION_ERRS); - cur = 0; - iter = *orig_iter; while (cur < blocksize) { struct page *page = bio_iter_page(&bbio->bio, iter); const u32 pg_off = bio_iter_offset(&bbio->bio, iter); From 13522d6632e4b4fab4e8b0153997f0554508a4e1 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:19 +0930 Subject: [PATCH 474/857] btrfs: remove on-stack paddrs[] array usage Since the bs > ps support, we have to handle cases where a data block is inside several discontiguous pages. Thus we need a local paddrs[] array to assemble a data block for bs > ps cases. However to handle all possible bs/ps combinations, we have to declare such array using the max block size vs page size, no matter the current block size and page size. This adds 128 bytes on-stack memory usage for several call sites, and also introduced several duplicated helpers to calculate checksum for a data block: - btrfs_calculate_block_csum_folio() - btrfs_calculate_block_csum_pages() - btrfs_check_block_csum() The differences are mostly in how the data is passed. The first one accepts a contiguous paddr range. The second one accepts an array of paddrs[]. The last one is just a simple wrapper of the first one. However the most common interface to iterate a data block is through bio, and we have already converted most callers to use the bio based interface, e.g. btrfs_bio_data_csum_ok() and btrfs_csum_one_bio_block(). Convert the remaining two call sites to address the remaining paddrs[] usage: - btrfs_calculate_block_csum_pages() inside verify_bio_data_sectors() This can be switched to btrfs_csum_one_bio_block(). This removes the 128 bytes on-stack memory usage. - btrfs_calculate_block_csum_pages() inside verify_one_sector() This call site doesn't use on-stack memory for paddrs[], but reuses the existing btrfs_raid_bio::bio_paddrs[] or btrfs_raid_bio::stripe_paddrs[]. So implement a local version called calculate_block_csum_paddrs(). Now there is no fixed on-stack paddrs[] usage anymore. Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 6 ---- fs/btrfs/inode.c | 75 ------------------------------------------ fs/btrfs/raid56.c | 46 +++++++++++++++----------- 3 files changed, 27 insertions(+), 100 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index e137a99151bd07..89e5e9c0c904f6 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -507,12 +507,6 @@ static inline void btrfs_set_inode_mapping_order(struct btrfs_inode *inode) inode->root->fs_info->block_max_order); } -void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info, - const phys_addr_t paddr, u8 *dest); -void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, - const phys_addr_t paddrs[], u8 *dest); -int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, - const u8 * const csum_expected); bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, struct btrfs_device *dev); void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 11f8aad8601f50..2967a307d8f066 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3456,81 +3456,6 @@ int btrfs_finish_ordered_io(struct btrfs_ordered_extent *ordered) return btrfs_finish_one_ordered(ordered); } -/* - * Calculate the checksum of an fs block at physical memory address @paddr, - * and save the result to @dest. - * - * The folio containing @paddr must be large enough to contain a full fs block. - */ -void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info, - const phys_addr_t paddr, u8 *dest) -{ - struct folio *folio = page_folio(phys_to_page(paddr)); - const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - - /* The full block must be inside the folio. */ - ASSERT(offset_in_folio(folio, paddr) + blocksize <= folio_size(folio)); - - for (int i = 0; i < nr_steps; i++) { - u32 pindex = offset_in_folio(folio, paddr + i * step) >> PAGE_SHIFT; - - /* - * For bs <= ps cases, we will only run the loop once, so the offset - * inside the page will only added to paddrs[0]. - * - * For bs > ps cases, the block must be page aligned, thus offset - * inside the page will always be 0. - */ - paddrs[i] = page_to_phys(folio_page(folio, pindex)) + offset_in_page(paddr); - } - return btrfs_calculate_block_csum_pages(fs_info, paddrs, dest); -} - -/* - * Calculate the checksum of a fs block backed by multiple noncontiguous pages - * at @paddrs[] and save the result to @dest. - * - * The folio containing @paddr must be large enough to contain a full fs block. - */ -void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, - const phys_addr_t paddrs[], u8 *dest) -{ - const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - struct btrfs_csum_ctx csum; - - btrfs_csum_init(&csum, fs_info->csum_type); - for (int i = 0; i < nr_steps; i++) { - const phys_addr_t paddr = paddrs[i]; - void *kaddr; - - ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE); - kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr); - btrfs_csum_update(&csum, kaddr, step); - kunmap_local(kaddr); - } - btrfs_csum_final(&csum, dest); -} - -/* - * Verify the checksum for a single sector without any extra action that depend - * on the type of I/O. - * - * @kaddr must be a properly kmapped address. - */ -int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, - const u8 * const csum_expected) -{ - btrfs_calculate_block_csum_folio(fs_info, paddr, csum); - if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) - return -EIO; - return 0; -} - /* Generate data checksum for a single fs block, pointed to by @orig_iter. */ void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, const struct bvec_iter *orig_iter, u8 *csum) diff --git a/fs/btrfs/raid56.c b/fs/btrfs/raid56.c index 1ee52a9dcee36f..a5d0ef09d92abb 100644 --- a/fs/btrfs/raid56.c +++ b/fs/btrfs/raid56.c @@ -1652,12 +1652,7 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio, struct bio *bio) { struct btrfs_fs_info *fs_info = rbio->bioc->fs_info; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = rbio->sector_nsteps; int total_sector_nr = get_bio_sector_nr(rbio, bio); - u32 offset = 0; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; /* No data csum for the whole stripe, no need to verify. */ if (!rbio->csum_bitmap || !rbio->csum_buf) @@ -1667,28 +1662,20 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio, if (total_sector_nr >= rbio->nr_data * rbio->stripe_nsectors) return; - btrfs_bio_for_each_block_all(paddr, bio, step) { + for (struct bvec_iter iter = init_bvec_iter_for_bio(bio); + iter.bi_size; + bio_advance_iter(bio, &iter, fs_info->sectorsize), total_sector_nr++) { u8 csum_buf[BTRFS_CSUM_SIZE]; u8 *expected_csum; - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; - - /* Not yet covering the full fs block, continue to the next step. */ - if (!IS_ALIGNED(offset, fs_info->sectorsize)) - continue; - /* No csum for this sector, skip to the next sector. */ - if (!test_bit(total_sector_nr, rbio->csum_bitmap)) { - total_sector_nr++; + if (!test_bit(total_sector_nr, rbio->csum_bitmap)) continue; - } expected_csum = rbio->csum_buf + total_sector_nr * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf); + btrfs_csum_one_bio_block(fs_info, bio, &iter, csum_buf); if (unlikely(memcmp(csum_buf, expected_csum, fs_info->csum_size) != 0)) set_bit(total_sector_nr, rbio->error_bitmap); - total_sector_nr++; } } @@ -1879,6 +1866,27 @@ void raid56_parity_write(struct bio *bio, struct btrfs_io_context *bioc) start_async_work(rbio, rmw_rbio_work); } +static void calculate_block_csum_paddrs(struct btrfs_fs_info *fs_info, + const phys_addr_t paddrs[], u8 *dest) +{ + const u32 blocksize = fs_info->sectorsize; + const u32 step = min(blocksize, PAGE_SIZE); + const u32 nr_steps = blocksize / step; + struct btrfs_csum_ctx csum; + + btrfs_csum_init(&csum, fs_info->csum_type); + for (int i = 0; i < nr_steps; i++) { + const phys_addr_t paddr = paddrs[i]; + void *kaddr; + + ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE); + kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr); + btrfs_csum_update(&csum, kaddr, step); + kunmap_local(kaddr); + } + btrfs_csum_final(&csum, dest); +} + static int verify_one_sector(struct btrfs_raid_bio *rbio, int stripe_nr, int sector_nr) { @@ -1906,7 +1914,7 @@ static int verify_one_sector(struct btrfs_raid_bio *rbio, csum_expected = rbio->csum_buf + (stripe_nr * rbio->stripe_nsectors + sector_nr) * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf); + calculate_block_csum_paddrs(fs_info, paddrs, csum_buf); if (unlikely(memcmp(csum_buf, csum_expected, fs_info->csum_size) != 0)) return -EIO; return 0; From 528df404c09db62b1cc68a45110e7097b9a49dca Mon Sep 17 00:00:00 2001 From: Leo Martins Date: Tue, 18 Aug 2026 17:40:10 -0700 Subject: [PATCH 475/857] btrfs: abort transaction before releasing tree_log_mutex on commit failure When transaction metadata writeout fails in btrfs_commit_transaction(), the current code only logs the error, drops tree_log_mutex and then goes through cleanup_transaction(), which aborts the transaction and records the fs error. That is too late for the tree log side. A log sync can already be waiting on tree_log_mutex, because the committing transaction is moved to TRANS_STATE_UNBLOCKED while that mutex is held, which lets fsyncs join the next transaction and queue up in btrfs_sync_log(). Once the failed commit drops tree_log_mutex, such a log sync acquires it, sees BTRFS_FS_ERROR() still clear, and writes super_for_commit. That superblock holds the roots prepared for the transaction that has just failed to write out its metadata, so it can point at tree blocks that never reached the disk, and the next mount fails with a parent transid mismatch. Commit 165ea85f1483 ("btrfs: do not write supers if we have an fs error") fixed this class of problem by making btrfs_sync_log() check for an fs error right after taking tree_log_mutex. That check only works if the commit path publishes the fs error before it releases the same mutex, and commit 68d4ece9c30e ("btrfs: don't call btrfs_handle_fs_error() in btrfs_commit_transaction()") removed the only thing that did so. Restore the ordering by aborting the transaction while tree_log_mutex is still held. We have a transaction handle here, so this does not need to bring back the btrfs_handle_fs_error() call: __btrfs_abort_transaction() records the fs error itself, which is all btrfs_sync_log() looks at, and the error message put in its place is kept. This is what commit 3810ab40afa5 ("btrfs: abort transaction on error in write_all_supers()") already does for the next call in this function. This is reproducible on an unmodified kernel by failing the first couple of bios of a transaction commit with fail_make_request while a concurrent fsync workload keeps log syncs queued on tree_log_mutex. Fixes: 68d4ece9c30e ("btrfs: don't call btrfs_handle_fs_error() in btrfs_commit_transaction()") CC: stable@vger.kernel.org # 7.0+ Reviewed-by: Boris Burkov Reviewed-by: jlayton@meta.com Reviewed-by: Filipe Manana Signed-off-by: Leo Martins Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/transaction.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index bafc62cf5ebcce..39909d363591a6 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -2583,6 +2583,12 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) ret = btrfs_write_and_wait_transaction(trans); if (unlikely(ret)) { btrfs_err(fs_info, "error while writing out transaction: %pe", ERR_PTR(ret)); + /* + * Abort before releasing tree_log_mutex, so a log sync waiting + * on it sees the fs error and skips writing super_for_commit + * for this failed transaction. See btrfs_sync_log(). + */ + btrfs_abort_transaction(trans, ret); mutex_unlock(&fs_info->tree_log_mutex); goto scrub_continue; } From 5463e0a9016e7612e7bd88d3ea7c5316a5fd7de2 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 24 Aug 2026 04:54:45 -0700 Subject: [PATCH 476/857] btrfs: skip extent tree lock in the shrinker for inodes without extent maps find_first_inode_to_shrink() takes inode->extent_tree.lock in write mode on every inode it walks, only to find out whether that inode has any extent maps. Most inodes have none, so the lock is taken and dropped again without any work being done. Check whether the tree is empty before taking the lock. tree->root is only modified with the tree lock held for write, so the unlocked read is a harmless race: a false empty just defers the inode to a later scan, which already happens whenever the write_trylock() below fails, and a false non-empty falls through to the existing check under the lock. Across the Meta production fleet the extent map shrinker is ~0.35% of non-idle kernel CPU. Attributing callees to their caller, find_first_inode_to_shrink() is ~65% of that, and the write_trylock() it does is ~30% of the whole shrinker. Micro benchmark: a 6 GiB btrfs on a loop device, 100000 empty files kept open, plus 200 1 MiB files created last so they get the highest inode numbers and every scan has to walk all the empty ones first. Each round drops the page cache, re-reads the data files to recreate the extent maps, then triggers the shrinker with "echo 2 > /proc/sys/vm/drop_caches". 15 rounds per run on ARM64 (Neoverse V2), 8 CPUs, no lock debugging. Cost of find_first_inode_to_shrink() from the ftrace function profiler, in ns per inode walked, median of runs: base patched delta idle 46.4 40.1 -13.6% 4 concurrent readers 47.8 38.4 -19.7% A separate build with CONFIG_LOCK_STAT, same test, for the extent map tree rwlock. The shrinker is not the only user of that lock, every extent map insert and lookup takes it too, which is why the acquisition count drops by two thirds rather than to nothing: base patched delta write acquisitions 628016 228000 -63.7% hold time total (us) 47512 22717 -52.2% acq cacheline bounces 1574 1288 -18.2% Reviewed-by: Filipe Manana Signed-off-by: Breno Leitao Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/extent_map.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/fs/btrfs/extent_map.c b/fs/btrfs/extent_map.c index 6ad7b39ae358b2..86d9c6f5ff4bd2 100644 --- a/fs/btrfs/extent_map.c +++ b/fs/btrfs/extent_map.c @@ -1219,6 +1219,14 @@ static struct btrfs_inode *find_first_inode_to_shrink(struct btrfs_root *root, tree = &inode->extent_tree; + /* + * Most inodes have no extent maps, so check without the lock. + * The race is harmless: a false empty just defers the inode to + * a later scan, and a false non-empty is caught under the lock. + */ + if (data_race(RB_EMPTY_ROOT(&tree->root))) + goto next; + /* * We want to be fast so if the lock is busy we don't want to * spend time waiting for it (some task is about to do IO for From dc9314442eab6191169f1ede90faa3ec26a272c4 Mon Sep 17 00:00:00 2001 From: Avi Weiss Date: Mon, 10 Aug 2026 12:47:01 +0300 Subject: [PATCH 477/857] btrfs: send: fix lost error return value in will_overwrite_ref() The direct-return refactoring in commit b3047a42f55d ("btrfs: send: directly return from will_overwrite_ref() and simplify it") changed will_overwrite_ref() to return directly instead of going through the common out label. That resulted in a negative return value from is_inode_existent() to start being converted to 0, making lookup errors unable to be distinguished from the inode not existing. process_recorded_refs() expects negative errors from will_overwrite_ref() and aborts processing when it receives one. Return the value from is_inode_existent() to restore the previous error propagation behavior as it was before the refactor. Fixes: b3047a42f55d ("btrfs: send: directly return from will_overwrite_ref() and simplify it") Reviewed-by: Filipe Manana Signed-off-by: Avi Weiss Signed-off-by: Filipe Manana Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/send.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c index f88623bbc491d6..5c59b9abedcd71 100644 --- a/fs/btrfs/send.c +++ b/fs/btrfs/send.c @@ -2065,7 +2065,7 @@ static int will_overwrite_ref(struct send_ctx *sctx, u64 dir, u64 dir_gen, ret = is_inode_existent(sctx, dir, dir_gen, NULL, &parent_root_dir_gen); if (ret <= 0) - return 0; + return ret; /* * If we have a parent root we need to verify that the parent dir was From 6f47ab17f783558575adc8b1e72b24067e7a4949 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 20 Aug 2026 18:28:48 +0930 Subject: [PATCH 478/857] btrfs: do not force reloc root creation during qgroup_account_snapshot() [BUG] When running btrfs/252 with quota enabled through MKFS_OPTIONS="-O quota", it has a high chance to trigger the following kernel warning and flips the fs RO: BTRFS info (device dm-2): relocating block group 30408704 flags metadata|dup ------------[ cut here ]------------ WARNING: fs/btrfs/extent-tree.c:879 at lookup_inline_extent_backref+0x74b/0x960 [btrfs], CPU#4: btrfs/2173 CPU: 4 UID: 0 PID: 2173 Comm: btrfs Not tainted 7.2.0-rc6-custom+ #457 PREEMPT(full) 3adc6528fb66f7a55fe1095385818e742f200aab Hardware name: QEMU Standard PC (Q35 + ICH9, 2009), BIOS unknown 02/02/2022 RIP: 0010:lookup_inline_extent_backref+0x74b/0x960 [btrfs] Call Trace: insert_inline_extent_backref+0x7c/0x160 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] __btrfs_inc_extent_ref+0xa9/0x270 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] __btrfs_run_delayed_refs+0x4af/0x11c0 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] btrfs_run_delayed_refs+0x9d/0xf0 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] create_pending_snapshot+0x39d/0xf00 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] create_pending_snapshots+0x9b/0xc0 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] btrfs_commit_transaction+0x280/0xeb0 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] prepare_to_relocate+0x147/0x200 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] relocate_block_group+0x6b/0x5e0 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] btrfs_relocate_block_group+0x92c/0x2380 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] btrfs_relocate_chunk+0x3f/0x1a0 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] btrfs_balance+0xa2c/0x19c0 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] btrfs_ioctl+0x2839/0x2d30 [btrfs 32f09462c54d9c922fca74a3e4866f4aa7737b72] __x64_sys_ioctl+0x416/0x9a0 do_syscall_64+0xe1/0x790 entry_SYSCALL_64_after_hwframe+0x4b/0x53 ---[ end trace 0000000000000000 ]--- BTRFS info (device dm-2): leaf 4593991680 gen 233 total ptrs 175 free space 5953 owner 2 BTRFS info (device dm-2): refs 3 lock_owner 2173 current 2173 item 0 key (166772736 METADATA_ITEM 1) itemoff 16250 itemsize 33 extent refs 1 gen 222 flags 2 ref#0: tree block backref root 266 [ Skip the tree dump ] item 174 key (263225344 METADATA_ITEM 0) itemoff 10328 itemsize 33 extent refs 1 gen 162 flags 258 ref#0: tree block backref root 267 BTRFS error (device dm-2): extent item not found for insert, bytenr 179847168 num_bytes 16384 parent 4594335744 root_objectid 273 owner 0 offset 0 BTRFS error (device dm-2): failed to run delayed ref for logical 179847168 num_bytes 16384 type 182 action 1 ref_mod 1: -117 [CAUSE] The above error is showing that there is a tree reference to a metadata extent that is no longer there. With "ref_verify" mount option (requires CONFIG_BTRFS_DEBUG), there is some extra debug output: BTRFS error (device dm-2): dumping block entry [180961280 16384], num_refs 0, metadata 1, from disk 0 BTRFS error (device dm-2): root entry 256, num_refs 18446744073709551615 BTRFS error (device dm-2): root entry 273, num_refs 18446744073709551615 BTRFS error (device dm-2): Ref action 3, root 273, ref_root 273, parent 0, owner 0, offset 0, num_refs 1 btrfs_force_cow_block+0x129/0x7d0 [btrfs] btrfs_cow_block+0x10a/0x250 [btrfs] btrfs_search_slot+0x5eb/0xf40 [btrfs] btrfs_insert_empty_items+0x3a/0x70 [btrfs] insert_with_overflow+0x53/0x130 [btrfs] btrfs_insert_dir_item+0x125/0x290 [btrfs] btrfs_add_link+0xaa/0x410 [btrfs] btrfs_rename+0x5ea/0xcd0 [btrfs] btrfs_rename2+0x28/0x60 [btrfs] vfs_rename+0x5b2/0xe10 filename_renameat2+0x244/0x430 __x64_sys_rename+0x48/0x70 do_syscall_64+0xe1/0x790 entry_SYSCALL_64_after_hwframe+0x4b/0x53 BTRFS error (device dm-2): Ref action 2, root 273, ref_root 273, parent 0, owner 0, offset 0, num_refs 18446744073709551615 btrfs_force_cow_block+0x327/0x7d0 [btrfs] btrfs_cow_block+0x10a/0x250 [btrfs] btrfs_search_slot+0x5eb/0xf40 [btrfs] btrfs_lookup_file_extent+0x4d/0x70 [btrfs] btrfs_drop_extents+0x151/0xf00 [btrfs] insert_reserved_file_extent+0xfe/0x3e0 [btrfs] btrfs_finish_one_ordered+0x549/0xc40 [btrfs] btrfs_work_helper+0xde/0x350 [btrfs] process_one_work+0x198/0x380 worker_thread+0x1c8/0x330 kthread+0xee/0x120 ret_from_fork+0x28f/0x310 ret_from_fork_asm+0x11/0x20 BTRFS error (device dm-2): Ref action 1, root 273, ref_root 0, parent 4594335744, owner 0, offset 0, num_refs 1 __btrfs_mod_ref+0x1c5/0x2d0 [btrfs] btrfs_copy_root+0x262/0x390 [btrfs] create_reloc_root+0xb9/0x370 [btrfs] btrfs_init_reloc_root+0xb0/0x1b0 [btrfs] record_root_in_trans+0xa6/0xd0 [btrfs] create_pending_snapshot+0x383/0xf00 [btrfs] create_pending_snapshots+0x9b/0xc0 [btrfs] btrfs_commit_transaction+0x280/0xeb0 [btrfs] prepare_to_relocate+0x147/0x200 [btrfs] relocate_block_group+0x6b/0x5e0 [btrfs] btrfs_relocate_block_group+0x92c/0x2380 [btrfs] btrfs_relocate_chunk+0x3f/0x1a0 [btrfs] btrfs_balance+0xa2c/0x19c0 [btrfs] btrfs_ioctl+0x2839/0x2d30 [btrfs] __x64_sys_ioctl+0x416/0x9a0 do_syscall_64+0xe1/0x790 The above shows the direct cause, Ref action 3 is the oldest operation, which shows the tree block is created by COW. Then ref action 2 shows it's COWed away, by a metadata update, meaning the tree block is already released, should not be referred any more. Then the final one, is trying to create a reloc tree for subvolume 273, and that reloc root creation is referring to the already dropped tree block. The root cause is that, during qgroup_account_snapshot(), we are calling record_root_in_trans() with "force = true". So if the root has no reloc root, we will create one, but at that timing it's already too late. Normally reloc root should be created before the commit and current roots diverge, to avoid the same problem we are hitting. But during relocation initialization, we are committing the current running transaction, with a new reloc_control attached halfway. And if qgroup is enabled, the record_root_in_trans() with "force = true" calls will force reloc root creation even if we do not and should not create reloc root at that timing. [FIX] Do not force reloc root creation during record_root_in_trans() with "force = true" cases, which is only called by qgroup_account_snapshot(). If we're really under relocation, the reloc root should be created way early, before the commit and current root diverge. If the root has no reloc tree yet, it means we're still initializing the reloc, and do not need a reloc root. So skipping the reloc tree creation in qgroup_account_snapshot() should be safe. Link: https://bugzilla.suse.com/show_bug.cgi?id=1275740 Fixes: 4d31778aa2fa ("btrfs: qgroup: Fix root item corruption when multiple same source snapshots are created with quota enabled") Assisted-by: LLM (initial analysis, but incorrect conclusion with too many burnt tokens) Tested-by: Disha Goel Reviewed-by: Filipe Manana Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/transaction.c | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 39909d363591a6..6802b94ed76ff4 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -458,8 +458,19 @@ static int record_root_in_trans(struct btrfs_trans_handle *trans, * through btrfs_record_root_in_trans without having to take the * lock. smp_wmb() makes sure that all the writes above are * done before we pop in the zero below + * + * If @force is true, it means the call is from + * qgroup_account_snapshot(), which only requires radix tree + * tracking. + * We should not force reloc root creation here, as the root + * may have already been modified, and in that case + * root->commit_root has already been dropped. + * + * Using that commit root will cause the reloc root to refer + * to a deleted extent, causing extent tree corruption. */ - ret = btrfs_init_reloc_root(trans, root); + if (!force) + ret = btrfs_init_reloc_root(trans, root); smp_mb__before_atomic(); clear_bit(BTRFS_ROOT_IN_TRANS_SETUP, &root->state); } From bd8fcab8e60d3ddaebfe74c758e22eb6a86f2623 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 25 Aug 2026 13:42:30 +0930 Subject: [PATCH 479/857] btrfs: remove unused variable flags from btrfs_read_qgroup_config() Since commit e562a8bdf652 ("btrfs: introduce BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN"), that @flags variable is no longer utilized. Just remove it. Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/qgroup.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index f68b696b4bf720..f3685cbb8f2e39 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -426,7 +426,6 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) struct extent_buffer *l; int slot; int ret = 0; - u64 flags = 0; u64 rescan_progress = 0; if (!fs_info->quota_root) @@ -609,7 +608,6 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) } out: btrfs_free_path(path); - fs_info->qgroup_flags |= flags; if (ret >= 0) { if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON) set_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags); From 0f9a2f5588f7274a8216f532273e56bb128877b5 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 25 Aug 2026 13:42:31 +0930 Subject: [PATCH 480/857] btrfs: qgroup: use atomic operations for btrfs_fs_info::qgroup_flags Currently we define btrfs_fs_info::qgroup_flags as u64, to match the on-disk qgroup status item's flag. But for now we have only 4 bits utilized for that flag, and since it's u64 we have no way to properly use the existing atomic bit operations (requires an unsigned long pointer). This results in a lot of non-atomic operations inside qgroup code. Some maybe fine as other locks are involved, but still it's not a good practice. Remove those non-atomic operations by: - Re-define btrfs_fs_info::qgroup_flags as unsigned long - Define BTRFS_QGROUP_STATUS_BIT_* and BTRFS_QGROUP_RUNTIME_BIT_* Instead of the old value define the bit number. - Use set_bit()/clear_bit()/test_bit() to replace open-coded bit operations - Add one extra check at qgroup status item read time To make sure the on-disk flag is still inside ULONG_MAX. Otherwise reject the status item and disable qgroup. - Get rid of unnecessary spinlock when checking a single bit Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/fs.h | 2 +- fs/btrfs/ioctl.c | 2 +- fs/btrfs/qgroup.c | 97 ++++++++++++++++----------------- fs/btrfs/qgroup.h | 4 +- fs/btrfs/sysfs.c | 8 +-- include/uapi/linux/btrfs_tree.h | 21 ++++--- 6 files changed, 67 insertions(+), 67 deletions(-) diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h index 10e15a319b93de..3eba8438593cde 100644 --- a/fs/btrfs/fs.h +++ b/fs/btrfs/fs.h @@ -811,7 +811,7 @@ struct btrfs_fs_info { struct btrfs_discard_ctl discard_ctl; /* Is qgroup tracking in a consistent state? */ - u64 qgroup_flags; + unsigned long qgroup_flags; /* Holds configuration and tracking. Protected by qgroup_lock. */ struct rb_root qgroup_tree; diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index e4b2da31a0d5de..54960351fbd171 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -3881,7 +3881,7 @@ static long btrfs_ioctl_quota_rescan_status(struct btrfs_fs_info *fs_info, if (!capable(CAP_SYS_ADMIN)) return -EPERM; - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { qsa.flags = 1; qsa.progress = fs_info->qgroup_rescan_progress.objectid; } diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index f3685cbb8f2e39..2c2ac0f16b1e5b 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -34,7 +34,7 @@ enum btrfs_qgroup_mode btrfs_qgroup_mode(const struct btrfs_fs_info *fs_info) { if (!test_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags)) return BTRFS_QGROUP_MODE_DISABLED; - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags)) return BTRFS_QGROUP_MODE_SIMPLE; return BTRFS_QGROUP_MODE_FULL; } @@ -384,14 +384,14 @@ static bool squota_check_parent_usage(struct btrfs_fs_info *fs_info, struct btrf __printf(2, 3) static void qgroup_mark_inconsistent(struct btrfs_fs_info *fs_info, const char *fmt, ...) { - const u64 old_flags = fs_info->qgroup_flags; + const unsigned long old_flags = fs_info->qgroup_flags; if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_SIMPLE) return; - fs_info->qgroup_flags |= (BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT | - BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN | - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING); - if (!(old_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) { + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); + if (!test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &old_flags)) { struct va_format vaf; va_list args; @@ -472,8 +472,12 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) "old qgroup version, quota disabled"); goto out; } - fs_info->qgroup_flags = btrfs_qgroup_status_flags(l, ptr); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE) + if (btrfs_qgroup_status_flags(l, ptr) > ULONG_MAX) { + btrfs_err(fs_info, "invalid qgroup status flags, quota disabled"); + goto out; + } + fs_info->qgroup_flags = (unsigned long)btrfs_qgroup_status_flags(l, ptr); + if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags)) qgroup_read_enable_gen(fs_info, l, slot, ptr); else if (btrfs_qgroup_status_generation(l, ptr) != fs_info->generation) qgroup_mark_inconsistent(fs_info, "qgroup generation mismatch"); @@ -609,12 +613,12 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) out: btrfs_free_path(path); if (ret >= 0) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) set_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) ret = qgroup_rescan_init(fs_info, rescan_progress, 0); } else { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); btrfs_sysfs_del_qgroups(fs_info); } @@ -1099,9 +1103,9 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, struct btrfs_qgroup_status_item); btrfs_set_qgroup_status_generation(leaf, ptr, trans->transid); btrfs_set_qgroup_status_version(leaf, ptr, BTRFS_QGROUP_STATUS_VERSION); - fs_info->qgroup_flags = BTRFS_QGROUP_STATUS_FLAG_ON; + fs_info->qgroup_flags = (1UL << BTRFS_QGROUP_STATUS_BIT_ON); if (simple) { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE; + set_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); btrfs_set_fs_incompat(fs_info, SIMPLE_QUOTA); /* * Set the enable generation to the next transaction, as we cannot @@ -1111,7 +1115,7 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, */ btrfs_set_qgroup_status_enable_gen(leaf, ptr, trans->transid + 1); } else { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); } btrfs_set_qgroup_status_flags(leaf, ptr, fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAGS_MASK); @@ -1401,8 +1405,8 @@ int btrfs_quota_disable(struct btrfs_fs_info *fs_info) spin_lock(&fs_info->qgroup_lock); quota_root = fs_info->quota_root; fs_info->quota_root = NULL; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE; + clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); fs_info->qgroup_drop_subtree_thres = BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT; spin_unlock(&fs_info->qgroup_lock); @@ -1552,7 +1556,7 @@ static int quick_update_accounting(struct btrfs_fs_info *fs_info, } out: if (ret) - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); return ret; } @@ -1873,7 +1877,7 @@ int btrfs_remove_qgroup(struct btrfs_trans_handle *trans, u64 qgroupid) * very frequently. */ if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_FULL && - !(fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) { + !test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags)) { if (unlikely(qgroup->rfer || qgroup->excl || qgroup->rfer_cmpr || qgroup->excl_cmpr)) { DEBUG_WARN(); @@ -2118,7 +2122,7 @@ int btrfs_qgroup_trace_extent_post(struct btrfs_trans_handle *trans, */ ASSERT(trans != NULL); - if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING) + if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags)) return 0; ret = btrfs_find_all_roots(&ctx, true); @@ -2959,7 +2963,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, * we can't just exit here. */ if (!btrfs_qgroup_full_accounting(fs_info) || - fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING) + test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags)) goto out_free; if (new_roots) { @@ -2981,7 +2985,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, num_bytes, nr_old_roots, nr_new_roots); mutex_lock(&fs_info->qgroup_rescan_lock); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { if (fs_info->qgroup_rescan_progress.objectid <= bytenr) { mutex_unlock(&fs_info->qgroup_rescan_lock); ret = 0; @@ -3042,8 +3046,8 @@ int btrfs_qgroup_account_extents(struct btrfs_trans_handle *trans) num_dirty_extents++; trace_btrfs_qgroup_account_extents(fs_info, record, bytenr); - if (!ret && !(fs_info->qgroup_flags & - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING)) { + if (!ret && !test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, + &fs_info->qgroup_flags)) { struct btrfs_backref_walk_ctx ctx = { 0 }; ctx.bytenr = bytenr; @@ -3150,9 +3154,9 @@ int btrfs_run_qgroups(struct btrfs_trans_handle *trans) spin_lock(&fs_info->qgroup_lock); } if (btrfs_qgroup_enabled(fs_info)) - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_ON; + set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); else - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON; + clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); spin_unlock(&fs_info->qgroup_lock); ret = update_qgroup_status_item(trans); @@ -3842,7 +3846,7 @@ static bool rescan_should_stop(struct btrfs_fs_info *fs_info) return true; if (!btrfs_qgroup_enabled(fs_info)) return true; - if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) + if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) return true; return false; } @@ -3892,12 +3896,10 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) btrfs_free_path(path); mutex_lock(&fs_info->qgroup_rescan_lock); - if (ret > 0 && - fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT) { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; - } else if (ret < 0 || stopped) { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; - } + if (ret > 0) + clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + else if (ret < 0 || stopped) + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); mutex_unlock(&fs_info->qgroup_rescan_lock); /* @@ -3921,9 +3923,9 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) } mutex_lock(&fs_info->qgroup_rescan_lock); - if (!stopped || - fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + if (!stopped || test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, + &fs_info->qgroup_flags)) + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); if (trans) { int ret2 = update_qgroup_status_item(trans); @@ -3933,7 +3935,7 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) } } fs_info->qgroup_rescan_running = false; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN; + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); complete_all(&fs_info->qgroup_rescan_completion); mutex_unlock(&fs_info->qgroup_rescan_lock); @@ -3944,7 +3946,7 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) if (stopped) { btrfs_info(fs_info, "qgroup scan paused"); - } else if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) { + } else if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) { btrfs_info(fs_info, "qgroup scan cancelled"); } else if (ret >= 0) { btrfs_info(fs_info, "qgroup scan completed%s", @@ -3971,13 +3973,11 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, if (!init_flags) { /* we're resuming qgroup rescan at mount time */ - if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_RESCAN)) { + if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags))) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup rescan is not queued"); ret = -EINVAL; - } else if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_ON)) { + } else if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags))) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup is not enabled"); ret = -ENOTCONN; @@ -3990,10 +3990,9 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_lock(&fs_info->qgroup_rescan_lock); if (init_flags) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { ret = -EINPROGRESS; - } else if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_ON)) { + } else if (!test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup is not enabled"); ret = -ENOTCONN; @@ -4006,13 +4005,13 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_unlock(&fs_info->qgroup_rescan_lock); return ret; } - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_RESCAN; + set_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); } memset(&fs_info->qgroup_rescan_progress, 0, sizeof(fs_info->qgroup_rescan_progress)); - fs_info->qgroup_flags &= ~(BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN | - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); fs_info->qgroup_rescan_progress.objectid = progress_objectid; init_completion(&fs_info->qgroup_rescan_completion); mutex_unlock(&fs_info->qgroup_rescan_lock); @@ -4063,7 +4062,7 @@ btrfs_qgroup_rescan(struct btrfs_fs_info *fs_info) ret = btrfs_commit_current_transaction(fs_info->fs_root); if (ret) { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); return ret; } @@ -4116,7 +4115,7 @@ int btrfs_qgroup_wait_for_completion(struct btrfs_fs_info *fs_info, void btrfs_qgroup_rescan_resume(struct btrfs_fs_info *fs_info) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { mutex_lock(&fs_info->qgroup_rescan_lock); fs_info->qgroup_rescan_running = true; btrfs_queue_work(fs_info->qgroup_rescan_workers, diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index 80dd2dacd56db4..b3aaad5e617d51 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -121,8 +121,8 @@ struct btrfs_qgroup_swapped_blocks; * To minimize the chance of collision with new persisted status flags, these * count backwards from the MSB. */ -#define BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN (1ULL << 63) -#define BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING (1ULL << 62) +#define BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN (BITS_PER_LONG - 1) +#define BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING (BITS_PER_LONG - 2) #define BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT (3) diff --git a/fs/btrfs/sysfs.c b/fs/btrfs/sysfs.c index 39cb01ee441ab8..1df6340a71234f 100644 --- a/fs/btrfs/sysfs.c +++ b/fs/btrfs/sysfs.c @@ -2359,9 +2359,7 @@ static ssize_t qgroup_enabled_show(struct kobject *qgroups_kobj, struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent); bool enabled; - spin_lock(&fs_info->qgroup_lock); - enabled = fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON; - spin_unlock(&fs_info->qgroup_lock); + enabled = test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); return sysfs_emit(buf, "%d\n", enabled); } @@ -2401,9 +2399,7 @@ static ssize_t qgroup_inconsistent_show(struct kobject *qgroups_kobj, struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent); bool inconsistent; - spin_lock(&fs_info->qgroup_lock); - inconsistent = (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT); - spin_unlock(&fs_info->qgroup_lock); + inconsistent = test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); return sysfs_emit(buf, "%d\n", inconsistent); } diff --git a/include/uapi/linux/btrfs_tree.h b/include/uapi/linux/btrfs_tree.h index cc3b9f7dccafa2..b6ccaf848e4b3e 100644 --- a/include/uapi/linux/btrfs_tree.h +++ b/include/uapi/linux/btrfs_tree.h @@ -1255,13 +1255,16 @@ static inline __u16 btrfs_qgroup_level(__u64 qgroupid) } /* - * is subvolume quota turned on? - */ -#define BTRFS_QGROUP_STATUS_FLAG_ON (1ULL << 0) -/* - * RESCAN is set during the initialization phase + * The following BTRFS_QGROUP_STATUS_BIT_* are for * btrfs_qgroup_status_item::flags. + * + * Is subvolume quota turned on? */ -#define BTRFS_QGROUP_STATUS_FLAG_RESCAN (1ULL << 1) +#define BTRFS_QGROUP_STATUS_BIT_ON (0) +#define BTRFS_QGROUP_STATUS_FLAG_ON (1UL << BTRFS_QGROUP_STATUS_BIT_ON) + +/* RESCAN is set during the initialization phase */ +#define BTRFS_QGROUP_STATUS_BIT_RESCAN (1) +#define BTRFS_QGROUP_STATUS_FLAG_RESCAN (1UL << BTRFS_QGROUP_STATUS_BIT_RESCAN) /* * Some qgroup entries are known to be out of date, * either because the configuration has changed in a way that @@ -1269,14 +1272,16 @@ static inline __u16 btrfs_qgroup_level(__u64 qgroupid) * with a non-qgroup-aware version. * Turning qouta off and on again makes it inconsistent, too. */ -#define BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT (1ULL << 2) +#define BTRFS_QGROUP_STATUS_BIT_INCONSISTENT (2) +#define BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT (1UL << BTRFS_QGROUP_STATUS_BIT_INCONSISTENT) /* * Whether or not this filesystem is using simple quotas. Not exactly the * incompat bit, because we support using simple quotas, disabling it, then * going back to full qgroup quotas. */ -#define BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE (1ULL << 3) +#define BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE (3) +#define BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE (1UL << BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE) #define BTRFS_QGROUP_STATUS_FLAGS_MASK (BTRFS_QGROUP_STATUS_FLAG_ON | \ BTRFS_QGROUP_STATUS_FLAG_RESCAN | \ From 2ee2a56ea952b31be466373f65811e5f98e4a010 Mon Sep 17 00:00:00 2001 From: FAN YE Date: Fri, 21 Aug 2026 17:50:09 +0000 Subject: [PATCH 481/857] btrfs: zstd: fix lost wakeup when waiting for a workspace A writer can sleep forever in zstd_get_workspace() even though a workspace is free. When zstd_alloc_workspace() fails, the task is queued on zwsm->wait and schedules unconditionally, never re-testing the pool. zstd_put_workspace() publishes the workspace and then calls cond_wake_up(), which only wakes when a sleeper is already visible, so a workspace returned between the failed allocation and prepare_to_wait() wakes nobody. The window is wide: zstd_alloc_workspace() goes through kvmalloc() and may enter reclaim. Only a max level workspace triggers the wakeup and one is deliberately kept allocated as the fallback every waiter waits for, so once its wakeup is lost the writer stays in TASK_UNINTERRUPTIBLE until some other task happens to return one. Re-check the pool after prepare_to_wait() has published the waiter, and use the workspace if one turned up. Fixes: 3f93aef535c8 ("btrfs: add zstd compression level support") Assisted-by: Claude:claude-opus-5 Reviewed-by: Qu Wenruo Signed-off-by: FAN YE Signed-off-by: David Sterba --- fs/btrfs/zstd.c | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/fs/btrfs/zstd.c b/fs/btrfs/zstd.c index 86919293fd546b..58d9ff76fe07bb 100644 --- a/fs/btrfs/zstd.c +++ b/fs/btrfs/zstd.c @@ -307,8 +307,17 @@ struct list_head *zstd_get_workspace(struct btrfs_fs_info *fs_info, int level) DEFINE_WAIT(wait); prepare_to_wait(&zwsm->wait, &wait, TASK_UNINTERRUPTIBLE); - schedule(); + /* + * Re-check after being queued: zstd_put_workspace() only wakes + * a queue that already has a sleeper, so a workspace returned + * since the failed allocation woke nobody. + */ + ws = zstd_find_workspace(fs_info, level); + if (!ws) + schedule(); finish_wait(&zwsm->wait, &wait); + if (ws) + return ws; goto again; } From 24023b632ac3a246f965f2bf822beecd452d6ee2 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 27 Aug 2026 16:25:29 +0930 Subject: [PATCH 482/857] btrfs: reject new qgroup rescan during subvolume dropping Commit 011b46c30476 ("btrfs: skip subtree scan if it's too high to avoid low stall in btrfs_commit_transaction()") introduced a threshold to skip huge subtree scan during subvolume dropping. But that's not covering all cases, e.g. rescan can still be started immediately after that huge subtree skipping. This will cause rescan to do the same accounting for that subtree anyway, still causing a long stall during transaction commit. Introduce a new runtime qgroup flag, BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, so that during cleanup of a subvolume, no new qgroup rescan can be initiated. The rejection uses the same -EINPROGRESS, as if there is already a running qgroup rescan. And since we have the extra bit, we can no longer allow plain assignment in btrfs_quota_enable(), as the plain assignment will override the REJECT_RESCAN bit. To co-operate this new flag: - Make btrfs_quota_enable() to only set BTRFS_QGROUP_STATUS_BIT_ON So it won't override the existing BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN bit. - Make btrfs_quota_disable() to clear every non-rescan bit This includes: * BTRFS_QGROUP_STATUS_BIT_ON * BTRFS_QGROUP_STATUS_BIT_INCONSISTENT * BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING For rescan related bits, they are either cleared by the rescan thread, or by the caller who rejects rescan. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 2 ++ fs/btrfs/qgroup.c | 13 +++++++++++-- fs/btrfs/qgroup.h | 11 +++++++++++ 3 files changed, 24 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 466fadb1815a81..a1d83ad9a4c00c 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -1496,7 +1496,9 @@ static int cleaner_kthread(void *arg) btrfs_run_delayed_iputs(fs_info); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags); again = btrfs_clean_one_deleted_snapshot(fs_info); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags); mutex_unlock(&fs_info->cleaner_mutex); /* diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index 2c2ac0f16b1e5b..e01b31aa0b1bf0 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -1103,7 +1103,7 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, struct btrfs_qgroup_status_item); btrfs_set_qgroup_status_generation(leaf, ptr, trans->transid); btrfs_set_qgroup_status_version(leaf, ptr, BTRFS_QGROUP_STATUS_VERSION); - fs_info->qgroup_flags = (1UL << BTRFS_QGROUP_STATUS_BIT_ON); + set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); if (simple) { set_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); btrfs_set_fs_incompat(fs_info, SIMPLE_QUOTA); @@ -1405,8 +1405,14 @@ int btrfs_quota_disable(struct btrfs_fs_info *fs_info) spin_lock(&fs_info->qgroup_lock); quota_root = fs_info->quota_root; fs_info->quota_root = NULL; + /* + * Clear all on-disk and runtime bits, except RESCAN related ones, that + * are either handled by rescan thread, or the caller who rejects rescan. + */ clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); clear_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); fs_info->qgroup_drop_subtree_thres = BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT; spin_unlock(&fs_info->qgroup_lock); @@ -3990,7 +3996,10 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_lock(&fs_info->qgroup_rescan_lock); if (init_flags) { - if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, + &fs_info->qgroup_flags) || + test_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, + &fs_info->qgroup_flags)) { ret = -EINPROGRESS; } else if (!test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) { btrfs_debug(fs_info, diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index b3aaad5e617d51..090ba536787269 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -124,6 +124,17 @@ struct btrfs_qgroup_swapped_blocks; #define BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN (BITS_PER_LONG - 1) #define BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING (BITS_PER_LONG - 2) +/* + * No new rescan allowed when set. + * + * During huge subtree dropping, qgroup will be marked inconsistent, and skip + * all future accounting to avoid long stall. But, an immediate rescan will + * re-enable qgroup and still stall the system. + * + * This bit is to avoid such rescan during the duration of a subvolume dropping. + */ +#define BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN (BITS_PER_LONG - 3) + #define BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT (3) /* From 4284a1aa264f3805b4d6ca319ee1a5930d212cca Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 27 Aug 2026 16:25:30 +0930 Subject: [PATCH 483/857] btrfs: avoid long stall when dropping a non-shared large subvolume Commit 011b46c30476 ("btrfs: skip subtree scan if it's too high to avoid low stall in btrfs_commit_transaction()") introduced a mechanism to skip large subtree during snapshot dropping. But even for a subvolume without any shared tree blocks, we can still queue quite a lot of qgroup records into one transaction, and cause a long qgroup related stall. So also add a check against the subvolume root level, to determine if we need to mark qgroup inconsistent. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/extent-tree.c | 10 ++++++++++ fs/btrfs/qgroup.c | 18 ++++++++++++++++++ fs/btrfs/qgroup.h | 1 + 3 files changed, 29 insertions(+) diff --git a/fs/btrfs/extent-tree.c b/fs/btrfs/extent-tree.c index d6a4390ee34ac9..a0d5ab03aae264 100644 --- a/fs/btrfs/extent-tree.c +++ b/fs/btrfs/extent-tree.c @@ -6315,6 +6315,16 @@ int btrfs_drop_snapshot(struct btrfs_root *root, bool update_ref, bool for_reloc set_bit(BTRFS_ROOT_DELETING, &root->state); unfinished_drop = test_bit(BTRFS_ROOT_UNFINISHED_DROP, &root->state); + /* + * For subvolume dropping, check if the subvolume is large enough so + * that we need to mark qgroup inconsistent to avoid long qgroup stall. + * + * Even for a subvolume without any snapshot, there can still be + * a lot of qgroup records queued into one transaction. + */ + if (!for_reloc) + btrfs_qgroup_check_tree_drop(fs_info, rootid, + btrfs_header_level(root->node)); if (btrfs_disk_key_objectid(&root_item->drop_progress) == 0) { level = btrfs_header_level(root->node); path->nodes[level] = btrfs_lock_root_node(root); diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index e01b31aa0b1bf0..05e35eb126dc5b 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -2748,6 +2748,24 @@ int btrfs_qgroup_trace_subtree(struct btrfs_trans_handle *trans, return 0; } +void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level) +{ + u8 drop_subtree_thres; + + if (btrfs_qgroup_mode(fs_info) != BTRFS_QGROUP_MODE_FULL) + return; + + if (!btrfs_is_fstree(rootid)) + return; + + spin_lock(&fs_info->qgroup_lock); + drop_subtree_thres = fs_info->qgroup_drop_subtree_thres; + spin_unlock(&fs_info->qgroup_lock); + + if (level >= drop_subtree_thres) + qgroup_mark_inconsistent(fs_info, "subtree level reached threshold"); +} + static void qgroup_iterator_nested_add(struct list_head *head, struct btrfs_qgroup *qgroup) { if (!list_empty(&qgroup->nested_iterator)) diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index 090ba536787269..c64b26b09c22f2 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -376,6 +376,7 @@ int btrfs_qgroup_trace_leaf_items(struct btrfs_trans_handle *trans, int btrfs_qgroup_trace_subtree(struct btrfs_trans_handle *trans, struct extent_buffer *root_eb, u64 root_gen, int root_level); +void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level); int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, u64 num_bytes, struct ulist *old_roots, struct ulist *new_roots); From 3d77edf2e18063e17dc7f9953466d892cc7a3b08 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 11 Aug 2026 15:31:49 +0930 Subject: [PATCH 484/857] btrfs: tests: do not touch page cache if root/inode allocation failed Inside test_find_delalloc() of extent-io-tests.c, if we fail to allocate a dummy root or the test inode, we go to out label to clean up. But at that stage, @inode is still NULL and we will call process_page_range() to access the page cache of the inode, this will cause NULL pointer dereference. This is a very minor bug, as it only affects selftests which are not compiled in by default for most distros, and very hard to trigger. Fix it by adding a new out_root_info label to handle root and inode allocation failure. This is a pre-existing bug reported by Sashiko while reviewing another patch. Link: https://sashiko.dev/#/patchset/cover.1786095309.git.wqu%40suse.com Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tests/extent-io-tests.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c index b2aacf846c8b75..23459cd4e50388 100644 --- a/fs/btrfs/tests/extent-io-tests.c +++ b/fs/btrfs/tests/extent-io-tests.c @@ -133,14 +133,14 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) if (IS_ERR(root)) { test_std_err(TEST_ALLOC_ROOT); ret = PTR_ERR(root); - goto out; + goto out_root_info; } inode = btrfs_new_test_inode(); if (!inode) { test_std_err(TEST_ALLOC_INODE); ret = -ENOMEM; - goto out; + goto out_root_info; } tmp = &BTRFS_I(inode)->io_tree; BTRFS_I(inode)->root = root; @@ -333,6 +333,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) process_page_range(inode, 0, total_dirty - 1, PROCESS_UNLOCK | PROCESS_RELEASE); iput(inode); +out_root_info: btrfs_free_dummy_root(root); btrfs_free_dummy_fs_info(fs_info); return ret; From b329bf20cf03bd2af3c910e9621ac700e001c794 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Fri, 21 Aug 2026 19:42:10 +0930 Subject: [PATCH 485/857] btrfs: remove runtime tweakable feature sysfs interface There are 2 features that are marked runtime tweakable inside /sys/fs/btrfs/features/ - acl Which is a mount option, and it will not show up in /sys/fs/btrfs//features/ directory anyway. - extended_iref This feature can only be enabled, but not disabled at runtime. Furthermore it's already the default behavior since 3.12. So it means this feature is always enabled and cannot be disabled for modern btrfs. So there is no need to maintain the ability to modify btrfs' runtime features through sysfs. And furthermore, the existing btrfs_feature_attr_store() is race-prone, it relies on fs_info->transaction_kthread, but our sysfs interfaces are enabled before transaction_kthread. Meaning at mount time a sysfs write can trigger NULL pointer dereference if the transaction_kthread is not yet initialized. The opposite is also possible during unmount. Thankfully that race is not possible in the real world, as the only supported feature is already enabled. But it also means we do not really need to keep the race-prone infrastructure, so just remove it completely, and make the per-module and per-mount features files to be completely read-only. Even with the sysfs tweakable features removed, we can still enable extended_iref feature through ioctl. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/sysfs.c | 121 ++--------------------------------------------- 1 file changed, 4 insertions(+), 117 deletions(-) diff --git a/fs/btrfs/sysfs.c b/fs/btrfs/sysfs.c index 1df6340a71234f..c5bb1c7eac6afe 100644 --- a/fs/btrfs/sysfs.c +++ b/fs/btrfs/sysfs.c @@ -83,8 +83,7 @@ struct raid_kobject { #define BTRFS_FEAT_ATTR(_name, _feature_set, _feature_prefix, _feature_bit) \ static struct btrfs_feature_attr btrfs_attr_features_##_name = { \ .kobj_attr = __INIT_KOBJ_ATTR(_name, S_IRUGO, \ - btrfs_feature_attr_show, \ - btrfs_feature_attr_store), \ + btrfs_feature_attr_show, NULL), \ .feature_set = _feature_set, \ .feature_bit = _feature_prefix ##_## _feature_bit, \ } @@ -130,130 +129,20 @@ static u64 get_features(struct btrfs_fs_info *fs_info, return btrfs_super_incompat_flags(disk_super); } -static void set_features(struct btrfs_fs_info *fs_info, - enum btrfs_feature_set set, u64 features) -{ - struct btrfs_super_block *disk_super = fs_info->super_copy; - if (set == FEAT_COMPAT) - btrfs_set_super_compat_flags(disk_super, features); - else if (set == FEAT_COMPAT_RO) - btrfs_set_super_compat_ro_flags(disk_super, features); - else - btrfs_set_super_incompat_flags(disk_super, features); -} - -static int can_modify_feature(struct btrfs_feature_attr *fa) -{ - int val = 0; - u64 set, clear; - switch (fa->feature_set) { - case FEAT_COMPAT: - set = BTRFS_FEATURE_COMPAT_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR; - break; - case FEAT_COMPAT_RO: - set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR; - break; - case FEAT_INCOMPAT: - set = BTRFS_FEATURE_INCOMPAT_SAFE_SET; - clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR; - break; - default: - btrfs_warn(NULL, "sysfs: unknown feature set %d", fa->feature_set); - return 0; - } - - if (set & fa->feature_bit) - val |= 1; - if (clear & fa->feature_bit) - val |= 2; - - return val; -} - static ssize_t btrfs_feature_attr_show(struct kobject *kobj, struct kobj_attribute *a, char *buf) { int val = 0; struct btrfs_fs_info *fs_info = to_fs_info(kobj); struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a); + if (fs_info) { u64 features = get_features(fs_info, fa->feature_set); if (features & fa->feature_bit) val = 1; - } else - val = can_modify_feature(fa); - - return sysfs_emit(buf, "%d\n", val); -} - -static ssize_t btrfs_feature_attr_store(struct kobject *kobj, - struct kobj_attribute *a, - const char *buf, size_t count) -{ - struct btrfs_fs_info *fs_info; - struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a); - u64 features, set, clear; - unsigned long val; - int ret; - - fs_info = to_fs_info(kobj); - if (!fs_info) - return -EPERM; - - if (sb_rdonly(fs_info->sb)) - return -EROFS; - - ret = kstrtoul(skip_spaces(buf), 0, &val); - if (ret) - return ret; - - if (fa->feature_set == FEAT_COMPAT) { - set = BTRFS_FEATURE_COMPAT_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR; - } else if (fa->feature_set == FEAT_COMPAT_RO) { - set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR; - } else { - set = BTRFS_FEATURE_INCOMPAT_SAFE_SET; - clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR; } - features = get_features(fs_info, fa->feature_set); - - /* Nothing to do */ - if ((val && (features & fa->feature_bit)) || - (!val && !(features & fa->feature_bit))) - return count; - - if ((val && !(set & fa->feature_bit)) || - (!val && !(clear & fa->feature_bit))) { - btrfs_info(fs_info, - "%sabling feature %s on mounted fs is not supported.", - val ? "En" : "Dis", fa->kobj_attr.attr.name); - return -EPERM; - } - - btrfs_info(fs_info, "%s %s feature flag", - val ? "Setting" : "Clearing", fa->kobj_attr.attr.name); - - spin_lock(&fs_info->super_lock); - features = get_features(fs_info, fa->feature_set); - if (val) - features |= fa->feature_bit; - else - features &= ~fa->feature_bit; - set_features(fs_info, fa->feature_set, features); - spin_unlock(&fs_info->super_lock); - - /* - * We don't want to do full transaction commit from inside sysfs - */ - set_bit(BTRFS_FS_NEED_TRANS_COMMIT, &fs_info->flags); - wake_up_process(fs_info->transaction_kthread); - - return count; + return sysfs_emit(buf, "%d\n", val); } static umode_t btrfs_feature_visible(struct kobject *kobj, @@ -269,9 +158,7 @@ static umode_t btrfs_feature_visible(struct kobject *kobj, fa = attr_to_btrfs_feature_attr(attr); features = get_features(fs_info, fa->feature_set); - if (can_modify_feature(fa)) - mode |= S_IWUSR; - else if (!(features & fa->feature_bit)) + if (!(features & fa->feature_bit)) mode = 0; } From b846814b43e90a59e9c1eed876518ecaa3383ae3 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 1 Sep 2026 09:31:01 +0930 Subject: [PATCH 486/857] btrfs: use ordered extent to grab the logical address for submission In submit_one_sector() we call btrfs_get_extent() to grab the IO extent map so that we know where the logical location to submit the block. However there is no guarantee that there is an IO extent map for the block, and if there is no IO extent map nor ordered extent, btrfs_get_extent() can grab the file extent from on-disk metadata. That's why we have ASSERT()s to reject holes and compressed file extents. On the other hand, for the write range we should have both an IO extent map and an ordered extent, so there is no reason not to grab the ordered extent instead. There is some minor advantages: - No hole ordered extent So no need to rely on ASSERT()s to reject hole extents. And the ASSERT()s are depending on the kernel config, without CONFIG_BTRFS_ASSERT those ASSERT()s won't even trigger. - No IO errors Unlike btrfs_get_extent() which can return IO error when doing the metadata tree search, btrfs_lookup_ordered_extent() will either return an OE or not found. - Cached OE in bio_ctrl->bbio At bbio allocation we have already did an OE lookup, and we have a high chance that the current block also belongs to that OE. Use that cached OE can reduce the frequency to do an rb-tree search. - Smaller rb-tree Unlike extent-map-tree, which can contain cached extent maps, the life span of ordered extents are much shorter, they get removed from the ordered tree after the file extent item is inserted into the subvolume tree. So doing ordered extent tree search can be a tiny faster. And since we're here, also address some minor points: - Add error message for every EUCLEAN error - Remove a dead comment on btrfs_folio_clear_dirty() We no longer call folio_clear_dirty_for_io() since commit 095be159f3eb ("btrfs: unify folio dirty flag clearing"), so the folio flag is still dirty, and the folio dirty flag will be cleared by the last dirty block. Reviewed-by: Boris Burkov Reviewed-by: Johannes Thumshirn Reviewed-by: Daniel Vacek Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/extent_io.c | 64 +++++++++++++++++++++++++++----------------- 1 file changed, 40 insertions(+), 24 deletions(-) diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c index d7600e5fa3d95d..a221b63bdb205d 100644 --- a/fs/btrfs/extent_io.c +++ b/fs/btrfs/extent_io.c @@ -1808,6 +1808,22 @@ static noinline_for_stack int writepage_delalloc(struct btrfs_inode *inode, return 0; } +static struct btrfs_ordered_extent *get_oe_from_bbio(const struct btrfs_bio *bbio, + u64 filepos) +{ + struct btrfs_ordered_extent *oe; + + if (!bbio || !bbio->ordered) + return NULL; + + oe = bbio->ordered; + if (!in_range(filepos, oe->file_offset, oe->num_bytes)) + return NULL; + + refcount_inc(&oe->refs); + return oe; +} + /* * Return 0 if we have submitted or queued the sector for submission. * Return <0 for critical errors, and the involved sector will be cleaned up. @@ -1820,11 +1836,10 @@ static int submit_one_sector(struct btrfs_inode *inode, loff_t i_size) { struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct extent_map *em; + struct btrfs_ordered_extent *oe; u64 block_start; u64 disk_bytenr; u64 extent_offset; - u64 em_end; const u32 sectorsize = fs_info->sectorsize; unsigned int queued; @@ -1833,8 +1848,11 @@ static int submit_one_sector(struct btrfs_inode *inode, /* @filepos >= i_size case should be handled by the caller. */ ASSERT(filepos < i_size); - em = btrfs_get_extent(inode, NULL, filepos, sectorsize); - if (IS_ERR(em)) { + /* Try to reuse the existing OE from bbio first. */ + oe = get_oe_from_bbio(bio_ctrl->bbio, filepos); + if (!oe) + oe = btrfs_lookup_ordered_extent(inode, filepos); + if (unlikely(!oe)) { /* * bio_ctrl may contain a bio crossing several folios. * Submit it immediately so that the bio has a chance @@ -1857,31 +1875,25 @@ static int submit_one_sector(struct btrfs_inode *inode, */ btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); - return PTR_ERR(em); + btrfs_err_rl(fs_info, + "no ordered extent for root %lld ino %llu filepos %llu", + btrfs_root_id(inode->root), btrfs_ino(inode), + filepos); + return -EUCLEAN; } - extent_offset = filepos - em->start; - em_end = btrfs_extent_map_end(em); - ASSERT(filepos <= em_end); - ASSERT(IS_ALIGNED(em->start, sectorsize)); - ASSERT(IS_ALIGNED(em->len, sectorsize)); - - block_start = btrfs_extent_map_block_start(em); - disk_bytenr = btrfs_extent_map_block_start(em) + extent_offset; + extent_offset = filepos - oe->file_offset; + ASSERT(filepos < oe->file_offset + oe->num_bytes); + ASSERT(IS_ALIGNED(oe->file_offset, sectorsize)); + ASSERT(IS_ALIGNED(oe->num_bytes, sectorsize)); + ASSERT(oe->compress_type == BTRFS_COMPRESS_NONE); + ASSERT(!test_bit(BTRFS_ORDERED_COMPRESSED, &oe->flags)); - ASSERT(!btrfs_extent_map_is_compressed(em)); - ASSERT(block_start != EXTENT_MAP_HOLE); - ASSERT(block_start != EXTENT_MAP_INLINE); + block_start = oe->disk_bytenr + oe->offset; + disk_bytenr = block_start + extent_offset; - btrfs_free_extent_map(em); - em = NULL; + btrfs_put_ordered_extent(oe); - /* - * Although the PageDirty bit is cleared before entering this - * function, subpage dirty bit is not cleared. - * So clear subpage dirty bit here so next time we won't submit - * a folio for a range already written to disk. - */ btrfs_folio_clear_dirty(fs_info, folio, filepos, sectorsize); btrfs_folio_set_writeback(fs_info, folio, filepos, sectorsize); /* @@ -1898,6 +1910,10 @@ static int submit_one_sector(struct btrfs_inode *inode, btrfs_folio_clear_writeback(fs_info, folio, filepos, sectorsize); btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); + btrfs_err_rl(fs_info, + "failed to queue sector for root %lld ino %llu filepos %llu", + btrfs_root_id(inode->root), + btrfs_ino(inode), filepos); return -EUCLEAN; } return 0; From c1c424942d75bfea1a34f203d6c1ef34069ef5d7 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 1 Sep 2026 10:01:33 +0930 Subject: [PATCH 487/857] btrfs: tree-checker: reject file extent items for special files File extent items are only utilized by regular files or symlinks, other files like directory/char/block/FIFO/sock files should not have any file extent item. Previously we were unable to reject such cases, as the inode item may not be in the same leaf. But we already have @prev_key in check_leaf_item(), this means we just need a new way to pass the mode of the previously hit inode item, then we can detect such problems. Introduce a new helper structure, saved_inode_info, to record the inode number and its mode hit in the same leaf, and keep it across the whole leaf. Then if we hit a file extent item, and the inode item is in the same leaf, we can refer to that to determine if we need to reject the file extent item. Now with the following corrupted fs tree, the kernel can safely reject the leaf: item 0 key (256 INODE_ITEM 0) itemoff 16123 itemsize 160 generation 3 transid 9 size 12 nbytes 16384 block group 0 mode 40755 links 1 uid 0 gid 0 rdev 0 sequence 1 flags 0x0(none) item 1 key (256 INODE_REF 256) itemoff 16111 itemsize 12 index 0 namelen 2 name: .. item 2 key (256 DIR_ITEM 496027801) itemoff 16075 itemsize 36 location key (257 INODE_ITEM 0) type FILE transid 9 data_len 0 name_len 6 name: foobar item 3 key (256 DIR_INDEX 2) itemoff 16039 itemsize 36 location key (257 INODE_ITEM 0) type FILE transid 9 data_len 0 name_len 6 name: foobar item 4 key (257 INODE_ITEM 0) itemoff 15879 itemsize 160 generation 9 transid 9 size 8192 nbytes 8192 block group 0 mode 60600 links 1 uid 0 gid 0 rdev 0 ^^ This is BLK type, not REG. sequence 2 flags 0x0(none) item 5 key (257 INODE_REF 256) itemoff 15863 itemsize 16 index 2 namelen 6 name: foobar item 6 key (257 EXTENT_DATA 0) itemoff 15810 itemsize 53 generation 9 type 1 (regular) extent data disk byte 13631488 nr 8192 extent data offset 0 nr 8192 ram 8192 extent compression 0 (none) extent encryption 0 With the patch, kernel will reject it with the following tree-checker errors: BTRFS critical (device loop0): corrupt leaf: root=5 block=30408704 slot=6 ino=257 file_offset=0, unexpected file extent item type 1 for inode mode 060600 BTRFS error (device loop0): read time tree block corruption detected on logical 30408704 mirror 1 Reported-by: ZhengYuan Huang Link: https://lore.kernel.org/linux-btrfs/20260817132051.267646-1-gality369@gmail.com/ Assisted-by: LLM (for generating the corrupted image) Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tree-checker.c | 66 ++++++++++++++++++++++++++++++++++------- 1 file changed, 55 insertions(+), 11 deletions(-) diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index 0ce91396b517f5..622334ffd24015 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -163,6 +163,12 @@ static void dir_item_err(const struct extent_buffer *eb, int slot, va_end(args); } +/* Record info for the last hit inode. */ +struct saved_inode_info { + u64 ino; + u32 mode; +}; + /* * This functions checks prev_key->objectid, to ensure current key and prev_key * share the same objectid as inode number. @@ -204,15 +210,41 @@ static bool check_prev_ino(struct extent_buffer *leaf, prev_key->objectid, key->objectid); return false; } + +static bool can_have_extent_data(struct extent_buffer *leaf, + struct btrfs_key *key, int slot, u8 fi_type, + const struct saved_inode_info *inode_info) +{ + /* No inode item in this leaf. */ + if (inode_info->ino != key->objectid) + return true; + if (S_ISREG(inode_info->mode)) + return true; + if (S_ISLNK(inode_info->mode)) { + /* For a symlink, the file extent item should always be inlined. */ + if (unlikely(fi_type != BTRFS_FILE_EXTENT_INLINE)) + return false; + return true; + } + + /* + * The rest are special files, e.g. block/FIFO files, which cannnot + * have any file extent. + */ + return false; +} + static int check_extent_data_item(struct extent_buffer *leaf, struct btrfs_key *key, int slot, - struct btrfs_key *prev_key) + struct btrfs_key *prev_key, + const struct saved_inode_info *inode_info) { struct btrfs_fs_info *fs_info = leaf->fs_info; struct btrfs_file_extent_item *fi; u32 sectorsize = fs_info->sectorsize; u32 item_size = btrfs_item_size(leaf, slot); u64 extent_end; + u8 fi_type; if (unlikely(!IS_ALIGNED(key->offset, sectorsize))) { file_extent_err(leaf, slot, @@ -243,12 +275,18 @@ static int check_extent_data_item(struct extent_buffer *leaf, SZ_4K); return -EUCLEAN; } - if (unlikely(btrfs_file_extent_type(leaf, fi) >= - BTRFS_NR_FILE_EXTENT_TYPES)) { + fi_type = btrfs_file_extent_type(leaf, fi); + if (unlikely(fi_type >= BTRFS_NR_FILE_EXTENT_TYPES)) { file_extent_err(leaf, slot, "invalid type for file extent, have %u expect range [0, %u]", - btrfs_file_extent_type(leaf, fi), - BTRFS_NR_FILE_EXTENT_TYPES - 1); + fi_type, BTRFS_NR_FILE_EXTENT_TYPES - 1); + return -EUCLEAN; + } + + if (unlikely(!can_have_extent_data(leaf, key, slot, fi_type, inode_info))) { + file_extent_err(leaf, slot, + "unexpected file extent item type %u for inode mode 0%o", + fi_type, inode_info->mode); return -EUCLEAN; } @@ -270,7 +308,8 @@ static int check_extent_data_item(struct extent_buffer *leaf, btrfs_file_extent_encryption(leaf, fi)); return -EUCLEAN; } - if (btrfs_file_extent_type(leaf, fi) == BTRFS_FILE_EXTENT_INLINE) { + + if (fi_type == BTRFS_FILE_EXTENT_INLINE) { /* Inline extent must have 0 as key offset */ if (unlikely(key->offset)) { file_extent_err(leaf, slot, @@ -1206,7 +1245,8 @@ static int check_dev_item(struct extent_buffer *leaf, } static int check_inode_item(struct extent_buffer *leaf, - struct btrfs_key *key, int slot) + struct btrfs_key *key, int slot, + struct saved_inode_info *inode_info) { struct btrfs_fs_info *fs_info = leaf->fs_info; struct btrfs_inode_item *iitem; @@ -1291,6 +1331,8 @@ static int check_inode_item(struct extent_buffer *leaf, ro_flags); return -EUCLEAN; } + inode_info->ino = key->objectid; + inode_info->mode = mode; return 0; } @@ -2319,14 +2361,15 @@ static int check_free_space_bitmap(struct extent_buffer *leaf, static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, struct btrfs_key *key, int slot, - struct btrfs_key *prev_key) + struct btrfs_key *prev_key, + struct saved_inode_info *inode_info) { int ret = 0; struct btrfs_chunk *chunk; switch (key->type) { case BTRFS_EXTENT_DATA_KEY: - ret = check_extent_data_item(leaf, key, slot, prev_key); + ret = check_extent_data_item(leaf, key, slot, prev_key, inode_info); break; case BTRFS_EXTENT_CSUM_KEY: ret = check_csum_item(leaf, key, slot, prev_key); @@ -2356,7 +2399,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, ret = check_dev_extent_item(leaf, key, slot, prev_key); break; case BTRFS_INODE_ITEM_KEY: - ret = check_inode_item(leaf, key, slot); + ret = check_inode_item(leaf, key, slot, inode_info); break; case BTRFS_ROOT_ITEM_KEY: ret = check_root_item(leaf, key, slot); @@ -2404,6 +2447,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf) { struct btrfs_fs_info *fs_info = leaf->fs_info; + struct saved_inode_info inode_info = { 0 }; /* No valid key type is 0, so all key should be larger than this key */ struct btrfs_key prev_key = {0, 0, 0}; struct btrfs_key key; @@ -2539,7 +2583,7 @@ enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf) } /* Check if the item size and content meet other criteria. */ - ret = check_leaf_item(leaf, &key, slot, &prev_key); + ret = check_leaf_item(leaf, &key, slot, &prev_key, &inode_info); if (unlikely(ret != BTRFS_TREE_BLOCK_CLEAN)) return ret; From 6a807cecbaa42e1461478219c734371473f0dc2b Mon Sep 17 00:00:00 2001 From: Johannes Thumshirn Date: Mon, 24 Aug 2026 18:19:10 +0200 Subject: [PATCH 488/857] btrfs: zoned: handle RAID profiles in btrfs_can_activate_zone() btrfs_can_activate_zone() only accounts for the single and DUP profiles. For a RAID0, RAID1, RAID1C3, RAID1C4 or RAID10 block group the profile switch matches no case, so 'ret' stays false and the function reports that no zone can be activated, even when the devices have plenty of active zones left. As a side effect BTRFS_FS_NEED_ZONE_FINISH gets set and, since btrfs_can_activate_zone() bails out early once that bit is set, data allocations will fail permanently: writers loop on -EAGAIN and hang in btrfs_new_extent_direct() waiting for the bit to clear. Each of these profiles needs one active zone per device, just like single, so handle them the same way. Reviewed-by: Boris Burkov Signed-off-by: Johannes Thumshirn Signed-off-by: David Sterba --- fs/btrfs/zoned.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index 9cc2c9c1a606b6..08a15465a0877d 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -2688,6 +2688,11 @@ bool btrfs_can_activate_zone(struct btrfs_fs_devices *fs_devices, u64 flags) switch (flags & BTRFS_BLOCK_GROUP_PROFILE_MASK) { case 0: /* single */ + case BTRFS_BLOCK_GROUP_RAID0: + case BTRFS_BLOCK_GROUP_RAID1: + case BTRFS_BLOCK_GROUP_RAID1C3: + case BTRFS_BLOCK_GROUP_RAID1C4: + case BTRFS_BLOCK_GROUP_RAID10: ret = (atomic_read(&zinfo->active_zones_left) >= (1 + reserved)); break; case BTRFS_BLOCK_GROUP_DUP: From c28c90360d2377c2d75827361e199e161442684f Mon Sep 17 00:00:00 2001 From: Johannes Thumshirn Date: Mon, 24 Aug 2026 18:47:58 +0200 Subject: [PATCH 489/857] btrfs: set space_info before adding new free space in btrfs_make_block_group() btrfs_make_block_group() calls btrfs_add_new_free_space() before assigning cache->space_info. On a zoned filesystem that ends up in __btrfs_add_free_space_zoned(), which dereferences block_group->space_info and thus hits a NULL pointer dereference when a non-initial free space range is added (e.g. during relocation). Assign cache->space_info before the btrfs_add_new_free_space() call. Reviewed-by: Boris Burkov Signed-off-by: Johannes Thumshirn Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 18 +++++++++++------- 1 file changed, 11 insertions(+), 7 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 830460a40e8655..ee182369254c08 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -3074,21 +3074,25 @@ struct btrfs_block_group *btrfs_make_block_group(struct btrfs_trans_handle *tran return ERR_PTR(ret); } - ret = btrfs_add_new_free_space(cache, chunk_offset, chunk_offset + size, NULL); - btrfs_free_excluded_extents(cache); - if (ret) { - btrfs_put_block_group(cache); - return ERR_PTR(ret); - } - /* * Ensure the corresponding space_info object is created and * assigned to our block group. We want our bg to be added to the rbtree * with its ->space_info set. + * + * On a zoned filesystem btrfs_add_new_free_space() ends up in + * __btrfs_add_free_space_zoned(), which dereferences + * block_group->space_info, so it has to be set beforehand. */ cache->space_info = space_info; ASSERT(cache->space_info); + ret = btrfs_add_new_free_space(cache, chunk_offset, chunk_offset + size, NULL); + btrfs_free_excluded_extents(cache); + if (ret) { + btrfs_put_block_group(cache); + return ERR_PTR(ret); + } + ret = btrfs_add_block_group_cache(cache); if (ret) { btrfs_remove_free_space_cache(cache); From 9f142de3245381643299494ac249db762d8231b9 Mon Sep 17 00:00:00 2001 From: Chris Mason Date: Thu, 27 Aug 2026 12:29:16 -0700 Subject: [PATCH 490/857] MAINTAINERS: update Chris Mason's email address David Sterba has been doing the Btrfs maintainership work for years, and my email update to mason@kernel.org seems like a good time to make the MAINTAINERS file a little more accurate. Link: https://lore.kernel.org/all/20260827193032.786461-1-clm@meta.com/ Signed-off-by: Chris Mason Signed-off-by: David Sterba --- MAINTAINERS | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/MAINTAINERS b/MAINTAINERS index 3a19da74d00c9d..bb32a819a8dbb8 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -5671,8 +5671,8 @@ W: http://bu3sch.de/btgpio.php F: drivers/gpio/gpio-bt8xx.c BTRFS FILE SYSTEM -M: Chris Mason M: David Sterba +R: Chris Mason L: linux-btrfs@vger.kernel.org S: Maintained W: https://btrfs.readthedocs.io From de58225840540aae609c4f0971109ea1094590e6 Mon Sep 17 00:00:00 2001 From: David Sterba Date: Wed, 21 Feb 2024 15:50:10 +0100 Subject: [PATCH 491/857] btrfs: === misc-next on b-for-next === Any commits after this one are for testing and evaluation only. Signed-off-by: David Sterba --- fs/btrfs/Kconfig | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/btrfs/Kconfig b/fs/btrfs/Kconfig index 4b10d78ed99b16..0b8d8905e38e25 100644 --- a/fs/btrfs/Kconfig +++ b/fs/btrfs/Kconfig @@ -1,4 +1,5 @@ # SPDX-License-Identifier: GPL-2.0 +# misc-next marker config BTRFS_FS tristate "Btrfs filesystem support" From d30371051efda3d7bfec6aa9bcca314e910d3717 Mon Sep 17 00:00:00 2001 From: Guenter Roeck Date: Thu, 13 Aug 2026 14:14:33 -0700 Subject: [PATCH 492/857] hwmon: Add support for currX_emergency and inX_[l]emergency attributes Some hardware monitoring chips support three alarm levels for current and voltage high limits, and three alarm levels for voltage low limits. Add support for currX_emergency, inX_emergency, and inX_lemergency attributes together with the appropriate alarm attributes to support such chips. Cc: Manaf Meethalavalappu Pallikunhi Tested-by: Manaf Meethalavalappu Pallikunhi Signed-off-by: Guenter Roeck Link: https://patch.msgid.link/20260813211433.879638-1-linux@roeck-us.net Signed-off-by: Guenter Roeck --- Documentation/ABI/testing/sysfs-class-hwmon | 36 +++++++++++++++++++++ Documentation/hwmon/sysfs-interface.rst | 12 +++++++ drivers/hwmon/hwmon.c | 6 ++++ include/linux/hwmon.h | 12 +++++++ 4 files changed, 66 insertions(+) diff --git a/Documentation/ABI/testing/sysfs-class-hwmon b/Documentation/ABI/testing/sysfs-class-hwmon index b185bdfc7186a7..34df6d7bd5f85e 100644 --- a/Documentation/ABI/testing/sysfs-class-hwmon +++ b/Documentation/ABI/testing/sysfs-class-hwmon @@ -61,6 +61,18 @@ Description: take drastic action such as power down or reset. At the very least, it should report a fault. +What: /sys/class/hwmon/hwmonX/inY_lemergency +Description: + Voltage emergency min value. + + Unit: millivolt + + RW + + If voltage drops to or below this limit, the system is + expected to take drastic action such as immediate power + down or reset. At the very least, it should report a fault. + What: /sys/class/hwmon/hwmonX/inY_max Description: Voltage max value. @@ -81,6 +93,18 @@ Description: take drastic action such as power down or reset. At the very least, it should report a fault. +What: /sys/class/hwmon/hwmonX/inY_emergency +Description: + Voltage emergency max value. + + Unit: millivolt + + RW + + If voltage reaches or exceeds this limit, the system is expected + to take drastic action such as immediate power down or reset. + At the very least, it should report a fault. + What: /sys/class/hwmon/hwmonX/inY_input Description: Voltage input value. @@ -647,6 +671,18 @@ Description: RW +What: /sys/class/hwmon/hwmonX/currY_emergency +Description: + Current emergency high value. + + Unit: milliampere + + RW + + If a current reaches or exceeds this limit, the system is + expected to take drastic action such as immediate power down + or reset. At the very least, it should report a fault. + What: /sys/class/hwmon/hwmonX/currY_input Description: Current input value diff --git a/Documentation/hwmon/sysfs-interface.rst b/Documentation/hwmon/sysfs-interface.rst index 94e1bbce172a33..f7ab5d5c4d3a12 100644 --- a/Documentation/hwmon/sysfs-interface.rst +++ b/Documentation/hwmon/sysfs-interface.rst @@ -121,12 +121,18 @@ Voltages `in[0-*]_lcrit` Voltage critical min value. +`in[0-*]_lemergency` + Voltage emergency min value. + `in[0-*]_max` Voltage max value. `in[0-*]_crit` Voltage critical max value. +`in[0-*]_emergency` + Voltage emergency max value. + `in[0-*]_input` Voltage input value. @@ -332,6 +338,9 @@ Currents `curr[1-*]_crit` Current critical high value. +`curr[1-*]_emergency` + Current emergency high value. + `curr[1-*]_input` Current input value. @@ -527,12 +536,15 @@ implementation. +-------------------------------+-----------------------+ | **`in[0-*]_min_alarm`, | Limit alarm | | `in[0-*]_max_alarm`, | | +| `in[0-*]_lemergency_alarm`, | | | `in[0-*]_lcrit_alarm`, | - 0: no alarm | | `in[0-*]_crit_alarm`, | - 1: alarm | +| `in[0-*]_emergency_alarm`, | | | `curr[1-*]_min_alarm`, | | | `curr[1-*]_max_alarm`, | RO | | `curr[1-*]_lcrit_alarm`, | | | `curr[1-*]_crit_alarm`, | | +| `curr[1-*]_emergency_alarm`, | | | `power[1-*]_cap_alarm`, | | | `power[1-*]_max_alarm`, | | | `power[1-*]_crit_alarm`, | | diff --git a/drivers/hwmon/hwmon.c b/drivers/hwmon/hwmon.c index 10d2df3efdfaee..3f32b0e83b532a 100644 --- a/drivers/hwmon/hwmon.c +++ b/drivers/hwmon/hwmon.c @@ -625,6 +625,8 @@ static const char * const hwmon_in_attr_templates[] = { [hwmon_in_max] = "in%d_max", [hwmon_in_lcrit] = "in%d_lcrit", [hwmon_in_crit] = "in%d_crit", + [hwmon_in_lemergency] = "in%d_lemergency", + [hwmon_in_emergency] = "in%d_emergency", [hwmon_in_average] = "in%d_average", [hwmon_in_lowest] = "in%d_lowest", [hwmon_in_highest] = "in%d_highest", @@ -635,6 +637,8 @@ static const char * const hwmon_in_attr_templates[] = { [hwmon_in_max_alarm] = "in%d_max_alarm", [hwmon_in_lcrit_alarm] = "in%d_lcrit_alarm", [hwmon_in_crit_alarm] = "in%d_crit_alarm", + [hwmon_in_lemergency_alarm] = "in%d_lemergency_alarm", + [hwmon_in_emergency_alarm] = "in%d_emergency_alarm", [hwmon_in_rated_min] = "in%d_rated_min", [hwmon_in_rated_max] = "in%d_rated_max", [hwmon_in_beep] = "in%d_beep", @@ -648,6 +652,7 @@ static const char * const hwmon_curr_attr_templates[] = { [hwmon_curr_max] = "curr%d_max", [hwmon_curr_lcrit] = "curr%d_lcrit", [hwmon_curr_crit] = "curr%d_crit", + [hwmon_curr_emergency] = "curr%d_emergency", [hwmon_curr_average] = "curr%d_average", [hwmon_curr_lowest] = "curr%d_lowest", [hwmon_curr_highest] = "curr%d_highest", @@ -658,6 +663,7 @@ static const char * const hwmon_curr_attr_templates[] = { [hwmon_curr_max_alarm] = "curr%d_max_alarm", [hwmon_curr_lcrit_alarm] = "curr%d_lcrit_alarm", [hwmon_curr_crit_alarm] = "curr%d_crit_alarm", + [hwmon_curr_emergency_alarm] = "curr%d_emergency_alarm", [hwmon_curr_rated_min] = "curr%d_rated_min", [hwmon_curr_rated_max] = "curr%d_rated_max", [hwmon_curr_beep] = "curr%d_beep", diff --git a/include/linux/hwmon.h b/include/linux/hwmon.h index dd713e193d0c3a..a3a7d27f3b5ffd 100644 --- a/include/linux/hwmon.h +++ b/include/linux/hwmon.h @@ -134,6 +134,8 @@ enum hwmon_in_attributes { hwmon_in_max, hwmon_in_lcrit, hwmon_in_crit, + hwmon_in_lemergency, + hwmon_in_emergency, hwmon_in_average, hwmon_in_lowest, hwmon_in_highest, @@ -144,6 +146,8 @@ enum hwmon_in_attributes { hwmon_in_max_alarm, hwmon_in_lcrit_alarm, hwmon_in_crit_alarm, + hwmon_in_lemergency_alarm, + hwmon_in_emergency_alarm, hwmon_in_rated_min, hwmon_in_rated_max, hwmon_in_beep, @@ -156,6 +160,8 @@ enum hwmon_in_attributes { #define HWMON_I_MAX BIT(hwmon_in_max) #define HWMON_I_LCRIT BIT(hwmon_in_lcrit) #define HWMON_I_CRIT BIT(hwmon_in_crit) +#define HWMON_I_LEMERGENCY BIT(hwmon_in_lemergency) +#define HWMON_I_EMERGENCY BIT(hwmon_in_emergency) #define HWMON_I_AVERAGE BIT(hwmon_in_average) #define HWMON_I_LOWEST BIT(hwmon_in_lowest) #define HWMON_I_HIGHEST BIT(hwmon_in_highest) @@ -166,6 +172,8 @@ enum hwmon_in_attributes { #define HWMON_I_MAX_ALARM BIT(hwmon_in_max_alarm) #define HWMON_I_LCRIT_ALARM BIT(hwmon_in_lcrit_alarm) #define HWMON_I_CRIT_ALARM BIT(hwmon_in_crit_alarm) +#define HWMON_I_LEMERGENCY_ALARM BIT(hwmon_in_lemergency_alarm) +#define HWMON_I_EMERGENCY_ALARM BIT(hwmon_in_emergency_alarm) #define HWMON_I_RATED_MIN BIT(hwmon_in_rated_min) #define HWMON_I_RATED_MAX BIT(hwmon_in_rated_max) #define HWMON_I_BEEP BIT(hwmon_in_beep) @@ -178,6 +186,7 @@ enum hwmon_curr_attributes { hwmon_curr_max, hwmon_curr_lcrit, hwmon_curr_crit, + hwmon_curr_emergency, hwmon_curr_average, hwmon_curr_lowest, hwmon_curr_highest, @@ -188,6 +197,7 @@ enum hwmon_curr_attributes { hwmon_curr_max_alarm, hwmon_curr_lcrit_alarm, hwmon_curr_crit_alarm, + hwmon_curr_emergency_alarm, hwmon_curr_rated_min, hwmon_curr_rated_max, hwmon_curr_beep, @@ -199,6 +209,7 @@ enum hwmon_curr_attributes { #define HWMON_C_MAX BIT(hwmon_curr_max) #define HWMON_C_LCRIT BIT(hwmon_curr_lcrit) #define HWMON_C_CRIT BIT(hwmon_curr_crit) +#define HWMON_C_EMERGENCY BIT(hwmon_curr_emergency) #define HWMON_C_AVERAGE BIT(hwmon_curr_average) #define HWMON_C_LOWEST BIT(hwmon_curr_lowest) #define HWMON_C_HIGHEST BIT(hwmon_curr_highest) @@ -209,6 +220,7 @@ enum hwmon_curr_attributes { #define HWMON_C_MAX_ALARM BIT(hwmon_curr_max_alarm) #define HWMON_C_LCRIT_ALARM BIT(hwmon_curr_lcrit_alarm) #define HWMON_C_CRIT_ALARM BIT(hwmon_curr_crit_alarm) +#define HWMON_C_EMERGENCY_ALARM BIT(hwmon_curr_emergency_alarm) #define HWMON_C_RATED_MIN BIT(hwmon_curr_rated_min) #define HWMON_C_RATED_MAX BIT(hwmon_curr_rated_max) #define HWMON_C_BEEP BIT(hwmon_curr_beep) From 06023c5fa6f33c99eebf4c32d560ab579a7dfc6a Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Wed, 12 Aug 2026 21:09:44 +0200 Subject: [PATCH 493/857] dt-bindings: hwmon: tmp102: Document TMP110 The TMP110 is register compatible with TMP102, document it using a fallback compatible. Signed-off-by: Marek Vasut Reviewed-by: Krzysztof Kozlowski Link: https://patch.msgid.link/20260812191021.65304-1-marex@nabladev.com Signed-off-by: Guenter Roeck --- .../devicetree/bindings/hwmon/ti,tmp102.yaml | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml b/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml index 96b2e4969f78a1..f0dbe9f07b939a 100644 --- a/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml +++ b/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml @@ -4,15 +4,19 @@ $id: http://devicetree.org/schemas/hwmon/ti,tmp102.yaml# $schema: http://devicetree.org/meta-schemas/core.yaml# -title: TMP102 temperature sensor +title: TMP102 and TMP110 temperature sensor maintainers: - Krzysztof Kozlowski properties: compatible: - enum: - - ti,tmp102 + oneOf: + - items: + - const: ti,tmp110 + - const: ti,tmp102 + - enum: + - ti,tmp102 interrupts: maxItems: 1 @@ -28,7 +32,7 @@ properties: const: 1 vcc-supply: - description: Power supply for tmp102 + description: Power supply for the sensor required: - compatible From 9ad068b8a1453bfab4363e96ebd759d66370b4d1 Mon Sep 17 00:00:00 2001 From: rahlquist Date: Thu, 13 Aug 2026 12:47:46 -0400 Subject: [PATCH 494/857] hwmon: (yogafan) Add Lenovo Yoga Pro 9 16IMH9 The Lenovo Yoga Pro 9 16IMH9 (83DN) exposes its fan tachometers at ACPI paths different from the generic Yoga configuration. Add a model-specific two-fan configuration for the PC00.LPCB.EC0 namespace and document the corrected mapping. Tested on a Lenovo Yoga Pro 9 16IMH9 (83DN) with BIOS NKCN35WW: the patched module registers fan1_input and fan2_input, both reporting 1800 RPM at idle. Signed-off-by: Richard Ahlquist Link: https://patch.msgid.link/20260813164746.105154-1-rahlquist@gmail.com Signed-off-by: Guenter Roeck --- Documentation/hwmon/yogafan.rst | 3 ++- drivers/hwmon/yogafan.c | 19 +++++++++++++++++++ 2 files changed, 21 insertions(+), 1 deletion(-) diff --git a/Documentation/hwmon/yogafan.rst b/Documentation/hwmon/yogafan.rst index 6395c94f4a6b06..9ff5db5dc08ce4 100644 --- a/Documentation/hwmon/yogafan.rst +++ b/Documentation/hwmon/yogafan.rst @@ -89,7 +89,8 @@ immediately to ensure the user knows the fan has stopped. ---------------------------------------------------------------------------------------------------- 82N7 | Yoga 14cACN | 0x06 | \_SB.PCI0.LPC0.EC0.FANS | 8-bit | 100 80V2 / 81C3 | Yoga 710/720 | 0x06 | \_SB.PCI0.LPC0.EC0.FAN0 | 8-bit | 100 - 83E2 / 83DN | Yoga Pro 7/9 | 0xFE | \_SB.PCI0.LPC0.EC0.FANS | 8-bit | 100 + 83E2 | Yoga Pro 7 | 0xFE | \_SB.PCI0.LPC0.EC0.FANS | 8-bit | 100 + 83DN | Yoga Pro 9 16IMH9 | 0x06/0xFE | \_SB.PC00.LPCB.EC0.FANS/FA2S | 8-bit | 100 82A2 / 82A3 | Yoga Slim 7 | 0x06 | \_SB.PCI0.LPC0.EC0.FANS | 8-bit | 100 81YM / 82FG | IdeaPad 5 | 0x06 | \_SB.PCI0.LPC0.EC0.FAN0 | 8-bit | 100 82JW / 82JU | Legion 5 (AMD) | 0xFE/0xFF | \_SB.PCI0.LPC0.EC0.FANS (Fan1) | 16-bit | 1 diff --git a/drivers/hwmon/yogafan.c b/drivers/hwmon/yogafan.c index 278cb089b0fd17..413ddc721a113f 100644 --- a/drivers/hwmon/yogafan.c +++ b/drivers/hwmon/yogafan.c @@ -90,6 +90,17 @@ static const struct yogafan_config yoga_pro_7_14iah10_cfg = { .paths = { "\\_SB.PC00.LPCB.EC0.FANS", NULL } }; +/* + * Lenovo Yoga Pro 9 16IMH9 (83DN) uses the PC00 namespace and has two + * 8-bit fan tachometer fields. + */ +static const struct yogafan_config yoga_pro_83dn_cfg = { + .multiplier = 100, + .fan_count = 2, + .paths = { "\\_SB.PC00.LPCB.EC0.FANS", + "\\_SB.PC00.LPCB.EC0.FA2S" } +}; + static void apply_rllag_filter(struct yoga_fan_data *data, int idx, long raw_rpm) { ktime_t now = ktime_get_boottime(); @@ -237,6 +248,14 @@ static const struct dmi_system_id yogafan_quirks[] = { }, .driver_data = (void *)&xiaoxin_8bit_dual_cfg, }, + { + .ident = "Lenovo Yoga Pro 9 16IMH9 (83DN)", + .matches = { + DMI_MATCH(DMI_SYS_VENDOR, "LENOVO"), + DMI_MATCH(DMI_PRODUCT_NAME, "83DN"), + }, + .driver_data = (void *)&yoga_pro_83dn_cfg, + }, { .ident = "Lenovo Yoga", .matches = { From 5478ea53d871abd782952b0503f2a19bdf5c748e Mon Sep 17 00:00:00 2001 From: Nikita Dubrovskih Date: Sun, 16 Aug 2026 02:40:41 +0300 Subject: [PATCH 495/857] hwmon: Add fan monitoring support for HONOR FMI-XX The HONOR FMI-XX firmware exposes a serialized \\GFNS ACPI method. It returns a status byte and a 16-bit fan speed in RPM for either of two firmware channels. Add a DMI-restricted, read-only hwmon driver using that firmware interface. The driver deliberately exposes no fan control or direct Embedded Controller access. The interface was validated on firmware 1.09 with fan channel 0 reporting approximately 2500-2800 RPM. Channel 1 is readable and remained at 0 RPM during idle and a short CPU load. Signed-off-by: Nikita Dubrovskih Link: https://patch.msgid.link/20260815234041.2262291-1-testname142@gmail.com Signed-off-by: Guenter Roeck --- Documentation/hwmon/honor-fmi.rst | 32 ++++++ Documentation/hwmon/index.rst | 1 + MAINTAINERS | 7 ++ drivers/hwmon/Kconfig | 10 ++ drivers/hwmon/Makefile | 1 + drivers/hwmon/honor-fmi.c | 178 ++++++++++++++++++++++++++++++ 6 files changed, 229 insertions(+) create mode 100644 Documentation/hwmon/honor-fmi.rst create mode 100644 drivers/hwmon/honor-fmi.c diff --git a/Documentation/hwmon/honor-fmi.rst b/Documentation/hwmon/honor-fmi.rst new file mode 100644 index 00000000000000..a42a1dd532e86b --- /dev/null +++ b/Documentation/hwmon/honor-fmi.rst @@ -0,0 +1,32 @@ +.. SPDX-License-Identifier: GPL-2.0-only + +Kernel driver honor-fmi +======================= + +Supported systems: + + * HONOR FMI-XX + +Author: Nikita Dubrovskih + +Description +----------- + +The driver provides read-only monitoring of the fan speed on the HONOR FMI-XX. +The system firmware implements a ``GFNS`` ACPI method which returns the speed +of one of two firmware fan channels in RPM. Embedded Controller access and +serialization are handled by the firmware method. + +The driver does not expose fan control or direct Embedded Controller access. + +Sysfs entries +------------- + +The following attributes are supported: + +======================= ======= ============================================= +Name Perm Description +======================= ======= ============================================= +``fan1_input`` RO Fan channel 0 speed in RPM +``fan2_input`` RO Fan channel 1 speed in RPM +======================= ======= ============================================= diff --git a/Documentation/hwmon/index.rst b/Documentation/hwmon/index.rst index 9955a525436ae6..06a7992ce5daa0 100644 --- a/Documentation/hwmon/index.rst +++ b/Documentation/hwmon/index.rst @@ -90,6 +90,7 @@ Hardware Monitoring Kernel Drivers gxp-fan-ctrl hac300s hih6130 + honor-fmi hp-wmi-sensors hs3001 htu31 diff --git a/MAINTAINERS b/MAINTAINERS index 3a19da74d00c9d..92c16ba8c78e78 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -11963,6 +11963,13 @@ S: Maintained F: Documentation/devicetree/bindings/iio/pressure/honeywell,mprls0025pa.yaml F: drivers/iio/pressure/mprls0025pa* +HONOR FMI-XX HARDWARE MONITOR DRIVER +M: Nikita Dubrovskih +L: linux-hwmon@vger.kernel.org +S: Maintained +F: Documentation/hwmon/honor-fmi.rst +F: drivers/hwmon/honor-fmi.c + HP BIOSCFG DRIVER M: Jorge Lopez L: platform-driver-x86@vger.kernel.org diff --git a/drivers/hwmon/Kconfig b/drivers/hwmon/Kconfig index fecff8610ea8ba..b19aa16986af6e 100644 --- a/drivers/hwmon/Kconfig +++ b/drivers/hwmon/Kconfig @@ -2827,6 +2827,16 @@ config SENSORS_ASUS_EC This driver can also be built as a module. If so, the module will be called asus_ec_sensors. +config SENSORS_HONOR_FMI + tristate "HONOR FMI-XX fan monitor" + depends on X86 && ACPI + help + If you say yes here, you get support for fan speed monitoring on + the HONOR FMI-XX laptop through its firmware ACPI method. + + This driver can also be built as a module. If so, the module + will be called honor-fmi. + config SENSORS_HP_WMI tristate "HP WMI Sensors" depends on ACPI_WMI diff --git a/drivers/hwmon/Makefile b/drivers/hwmon/Makefile index 1229b6b3996d79..646813b7aaa852 100644 --- a/drivers/hwmon/Makefile +++ b/drivers/hwmon/Makefile @@ -11,6 +11,7 @@ obj-$(CONFIG_SENSORS_ACPI_POWER) += acpi_power_meter.o obj-$(CONFIG_SENSORS_ATK0110) += asus_atk0110.o obj-$(CONFIG_SENSORS_ASUS_EC) += asus-ec-sensors.o obj-$(CONFIG_SENSORS_ASUS_WMI) += asus_wmi_sensors.o +obj-$(CONFIG_SENSORS_HONOR_FMI) += honor-fmi.o obj-$(CONFIG_SENSORS_HP_WMI) += hp-wmi-sensors.o # Native drivers diff --git a/drivers/hwmon/honor-fmi.c b/drivers/hwmon/honor-fmi.c new file mode 100644 index 00000000000000..3421e40a132f83 --- /dev/null +++ b/drivers/hwmon/honor-fmi.c @@ -0,0 +1,178 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Read-only fan monitoring for the HONOR FMI-XX. + * + * The firmware-provided \GFNS ACPI method accepts a three-byte buffer. + * Byte 2 selects fan 0 or 1. It returns a status byte followed by a + * little-endian 16-bit fan speed in RPM. The method owns all Embedded + * Controller access and serialization; this driver deliberately exposes no + * fan control interface. + */ + +#include +#include +#include +#include +#include +#include + +#define HONOR_FMI_GFNS_RESULT_SIZE 3 + +struct honor_fmi_data { + acpi_handle gfns; +}; + +static const struct dmi_system_id honor_fmi_dmi_table[] = { + { + .matches = { + DMI_MATCH(DMI_SYS_VENDOR, "HONOR"), + DMI_EXACT_MATCH(DMI_PRODUCT_NAME, "FMI-XX"), + }, + }, + {} +}; +MODULE_DEVICE_TABLE(dmi, honor_fmi_dmi_table); + +static int honor_fmi_read_rpm(struct honor_fmi_data *data, int channel, + long *rpm) +{ + union acpi_object input = { + .buffer = { + .type = ACPI_TYPE_BUFFER, + .length = 3, + }, + }; + struct acpi_object_list arguments = { + .count = 1, + .pointer = &input, + }; + struct acpi_buffer output = { ACPI_ALLOCATE_BUFFER, NULL }; + union acpi_object *result; + u8 input_bytes[3] = { 0, 0, channel }; + acpi_status status; + int ret = 0; + + input.buffer.pointer = input_bytes; + + status = acpi_evaluate_object(data->gfns, NULL, &arguments, &output); + if (ACPI_FAILURE(status)) + return -EIO; + + result = output.pointer; + if (!result || result->type != ACPI_TYPE_BUFFER || + result->buffer.length < HONOR_FMI_GFNS_RESULT_SIZE) { + ret = -EPROTO; + goto out_free; + } + + if (result->buffer.pointer[0]) { + ret = -EIO; + goto out_free; + } + + *rpm = result->buffer.pointer[1] | + (result->buffer.pointer[2] << 8); + +out_free: + kfree(output.pointer); + return ret; +} + +static umode_t honor_fmi_is_visible(const void *data, + enum hwmon_sensor_types type, u32 attr, + int channel) +{ + return 0444; +} + +static int honor_fmi_read(struct device *dev, enum hwmon_sensor_types type, + u32 attr, int channel, long *value) +{ + struct honor_fmi_data *data = dev_get_drvdata(dev); + + return honor_fmi_read_rpm(data, channel, value); +} + +static const struct hwmon_ops honor_fmi_hwmon_ops = { + .is_visible = honor_fmi_is_visible, + .read = honor_fmi_read, +}; + +static const struct hwmon_channel_info * const honor_fmi_hwmon_info[] = { + HWMON_CHANNEL_INFO(fan, HWMON_F_INPUT, HWMON_F_INPUT), + NULL +}; + +static const struct hwmon_chip_info honor_fmi_chip_info = { + .ops = &honor_fmi_hwmon_ops, + .info = honor_fmi_hwmon_info, +}; + +static int honor_fmi_probe(struct platform_device *pdev) +{ + struct honor_fmi_data *data; + struct device *hwmon_dev; + acpi_status status; + + if (!dmi_check_system(honor_fmi_dmi_table)) + return -ENODEV; + + data = devm_kzalloc(&pdev->dev, sizeof(*data), GFP_KERNEL); + if (!data) + return -ENOMEM; + + status = acpi_get_handle(NULL, "\\GFNS", &data->gfns); + if (ACPI_FAILURE(status)) + return dev_err_probe(&pdev->dev, -ENODEV, + "firmware does not provide \\GFNS\n"); + + hwmon_dev = devm_hwmon_device_register_with_info(&pdev->dev, "honor_fmi", + data, + &honor_fmi_chip_info, + NULL); + return PTR_ERR_OR_ZERO(hwmon_dev); +} + +static struct platform_driver honor_fmi_driver = { + .probe = honor_fmi_probe, + .driver = { + .name = "honor-fmi-hwmon", + }, +}; + +static struct platform_device *honor_fmi_device; + +static int __init honor_fmi_init(void) +{ + int ret; + + if (!dmi_check_system(honor_fmi_dmi_table)) + return -ENODEV; + + ret = platform_driver_register(&honor_fmi_driver); + if (ret) + return ret; + + honor_fmi_device = platform_device_register_simple("honor-fmi-hwmon", + PLATFORM_DEVID_NONE, + NULL, 0); + if (IS_ERR(honor_fmi_device)) { + platform_driver_unregister(&honor_fmi_driver); + return PTR_ERR(honor_fmi_device); + } + + return 0; +} + +static void __exit honor_fmi_exit(void) +{ + platform_device_unregister(honor_fmi_device); + platform_driver_unregister(&honor_fmi_driver); +} + +module_init(honor_fmi_init); +module_exit(honor_fmi_exit); + +MODULE_AUTHOR("Nikita Dubrovskih "); +MODULE_DESCRIPTION("HONOR FMI-XX fan speed monitor"); +MODULE_LICENSE("GPL"); From 5190cb12625b54362f91a9c2fe38804bf3e98f43 Mon Sep 17 00:00:00 2001 From: Stoyan Bogdanov Date: Mon, 17 Aug 2026 13:14:43 +0300 Subject: [PATCH 496/857] hwmon: (pmbus/tps25990): Rework driver for multi-device support Rework existing implementation to allow adding support for new devices to the existing driver. chip_id is used to identify the current device and differentiate logic where needed. Changes include: - Add an enum listing supported chips - Add a structure to hold per-device m, b, R coefficients Signed-off-by: Stoyan Bogdanov Link: https://patch.msgid.link/20260817101455.3526260-2-sbogdanov@baylibre.com Signed-off-by: Guenter Roeck --- drivers/hwmon/pmbus/tps25990.c | 123 +++++++++++++++++++-------------- 1 file changed, 70 insertions(+), 53 deletions(-) diff --git a/drivers/hwmon/pmbus/tps25990.c b/drivers/hwmon/pmbus/tps25990.c index 9d318e6509abf6..7634ac743025df 100644 --- a/drivers/hwmon/pmbus/tps25990.c +++ b/drivers/hwmon/pmbus/tps25990.c @@ -47,6 +47,15 @@ PK_MIN_AVG_RST_AVG | \ PK_MIN_AVG_RST_MIN) +enum chips { + tps25990, +}; + +struct tps25990_data { + struct pmbus_driver_info info; + enum chips chip_id; +}; + /* * Arbitrary default Rimon value: 1kOhm * This correspond to an overcurrent limit of 55A, close to the specified limit @@ -337,63 +346,65 @@ static const struct regulator_desc tps25990_reg_desc[] = { }; #endif -static const struct pmbus_driver_info tps25990_base_info = { - .pages = 1, - .format[PSC_VOLTAGE_IN] = direct, - .m[PSC_VOLTAGE_IN] = 5251, - .b[PSC_VOLTAGE_IN] = 0, - .R[PSC_VOLTAGE_IN] = -2, - .format[PSC_VOLTAGE_OUT] = direct, - .m[PSC_VOLTAGE_OUT] = 5251, - .b[PSC_VOLTAGE_OUT] = 0, - .R[PSC_VOLTAGE_OUT] = -2, - .format[PSC_TEMPERATURE] = direct, - .m[PSC_TEMPERATURE] = 140, - .b[PSC_TEMPERATURE] = 32100, - .R[PSC_TEMPERATURE] = -2, - /* - * Current and Power measurement depends on the ohm value - * of Rimon. m is multiplied by 1000 below to have an integer - * and -3 is added to R to compensate. - */ - .format[PSC_CURRENT_IN] = direct, - .m[PSC_CURRENT_IN] = 9538, - .b[PSC_CURRENT_IN] = 0, - .R[PSC_CURRENT_IN] = -6, - .format[PSC_POWER] = direct, - .m[PSC_POWER] = 4901, - .b[PSC_POWER] = 0, - .R[PSC_POWER] = -7, - .func[0] = (PMBUS_HAVE_VIN | - PMBUS_HAVE_VOUT | - PMBUS_HAVE_VMON | - PMBUS_HAVE_IIN | - PMBUS_HAVE_PIN | - PMBUS_HAVE_TEMP | - PMBUS_HAVE_STATUS_VOUT | - PMBUS_HAVE_STATUS_IOUT | - PMBUS_HAVE_STATUS_INPUT | - PMBUS_HAVE_STATUS_TEMP | - PMBUS_HAVE_SAMPLES), - .read_word_data = tps25990_read_word_data, - .write_word_data = tps25990_write_word_data, - .read_byte_data = tps25990_read_byte_data, - .write_byte_data = tps25990_write_byte_data, +static const struct pmbus_driver_info tps25990_base_info[] = { + [tps25990] = { + .pages = 1, + .format[PSC_VOLTAGE_IN] = direct, + .m[PSC_VOLTAGE_IN] = 5251, + .b[PSC_VOLTAGE_IN] = 0, + .R[PSC_VOLTAGE_IN] = -2, + .format[PSC_VOLTAGE_OUT] = direct, + .m[PSC_VOLTAGE_OUT] = 5251, + .b[PSC_VOLTAGE_OUT] = 0, + .R[PSC_VOLTAGE_OUT] = -2, + .format[PSC_TEMPERATURE] = direct, + .m[PSC_TEMPERATURE] = 140, + .b[PSC_TEMPERATURE] = 32100, + .R[PSC_TEMPERATURE] = -2, + /* + * Current and Power measurement depends on the ohm value + * of Rimon. m is multiplied by 1000 below to have an integer + * and -3 is added to R to compensate. + */ + .format[PSC_CURRENT_IN] = direct, + .m[PSC_CURRENT_IN] = 9538, + .b[PSC_CURRENT_IN] = 0, + .R[PSC_CURRENT_IN] = -6, + .format[PSC_POWER] = direct, + .m[PSC_POWER] = 4901, + .b[PSC_POWER] = 0, + .R[PSC_POWER] = -7, + .func[0] = (PMBUS_HAVE_VIN | + PMBUS_HAVE_VOUT | + PMBUS_HAVE_VMON | + PMBUS_HAVE_IIN | + PMBUS_HAVE_PIN | + PMBUS_HAVE_TEMP | + PMBUS_HAVE_STATUS_VOUT | + PMBUS_HAVE_STATUS_IOUT | + PMBUS_HAVE_STATUS_INPUT | + PMBUS_HAVE_STATUS_TEMP | + PMBUS_HAVE_SAMPLES), + .read_word_data = tps25990_read_word_data, + .write_word_data = tps25990_write_word_data, + .read_byte_data = tps25990_read_byte_data, + .write_byte_data = tps25990_write_byte_data, #if IS_ENABLED(CONFIG_SENSORS_TPS25990_REGULATOR) - .reg_desc = tps25990_reg_desc, - .num_regulators = ARRAY_SIZE(tps25990_reg_desc), + .reg_desc = tps25990_reg_desc, + .num_regulators = ARRAY_SIZE(tps25990_reg_desc), #endif + }, }; static const struct i2c_device_id tps25990_i2c_id[] = { - { .name = "tps25990" }, - { } + { .name = "tps25990", .driver_data = tps25990 }, + {} }; MODULE_DEVICE_TABLE(i2c, tps25990_i2c_id); static const struct of_device_id tps25990_of_match[] = { - { .compatible = "ti,tps25990" }, + { .compatible = "ti,tps25990", .data = (void *)tps25990 }, {} }; MODULE_DEVICE_TABLE(of, tps25990_of_match); @@ -401,8 +412,9 @@ MODULE_DEVICE_TABLE(of, tps25990_of_match); static int tps25990_probe(struct i2c_client *client) { struct device *dev = &client->dev; - struct pmbus_driver_info *info; + struct tps25990_data *data; const char *propname; + enum chips chip_id; u32 rimon; int ret; @@ -415,15 +427,20 @@ static int tps25990_probe(struct i2c_client *client) rimon = TPS25990_DEFAULT_RIMON; } - info = devm_kmemdup(dev, &tps25990_base_info, sizeof(*info), GFP_KERNEL); - if (!info) + chip_id = (enum chips)(unsigned long)i2c_get_match_data(client); + + data = devm_kzalloc(dev, sizeof(struct tps25990_data), GFP_KERNEL); + if (!data) return -ENOMEM; + data->info = tps25990_base_info[chip_id]; + data->chip_id = chip_id; + /* Adapt the current and power scale for each instance */ - tps25990_set_m(&info->m[PSC_CURRENT_IN], rimon); - tps25990_set_m(&info->m[PSC_POWER], rimon); + tps25990_set_m(&data->info.m[PSC_CURRENT_IN], rimon); + tps25990_set_m(&data->info.m[PSC_POWER], rimon); - return pmbus_do_probe(client, info); + return pmbus_do_probe(client, &data->info); } static struct i2c_driver tps25990_driver = { From 86978419df8257b8d452b90a9924cb967855dd99 Mon Sep 17 00:00:00 2001 From: Stoyan Bogdanov Date: Mon, 17 Aug 2026 13:14:44 +0300 Subject: [PATCH 497/857] dt-bindings: hwmon: pmbus/tps25990: Add TPS1689 Add device compatible support for TPS1689 Signed-off-by: Stoyan Bogdanov Acked-by: Krzysztof Kozlowski Link: https://patch.msgid.link/20260817101455.3526260-3-sbogdanov@baylibre.com Signed-off-by: Guenter Roeck --- .../devicetree/bindings/hwmon/pmbus/ti,tps25990.yaml | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/Documentation/devicetree/bindings/hwmon/pmbus/ti,tps25990.yaml b/Documentation/devicetree/bindings/hwmon/pmbus/ti,tps25990.yaml index f4115870e45094..63ccb67576df8b 100644 --- a/Documentation/devicetree/bindings/hwmon/pmbus/ti,tps25990.yaml +++ b/Documentation/devicetree/bindings/hwmon/pmbus/ti,tps25990.yaml @@ -5,18 +5,20 @@ $id: http://devicetree.org/schemas/hwmon/pmbus/ti,tps25990.yaml# $schema: http://devicetree.org/meta-schemas/core.yaml# -title: Texas Instruments TPS25990 Stackable eFuse +title: Texas Instruments Stackable eFuses maintainers: - Jerome Brunet description: - The TI TPS25990 is an integrated, high-current circuit + The TI TPS25990 and TPS1689 are integrated, high-current circuit protection and power management device with PMBUS interface properties: compatible: - const: ti,tps25990 + enum: + - ti,tps1689 + - ti,tps25990 reg: maxItems: 1 From 1d28682c4e3219a2500cf68e739f87ffb7552c45 Mon Sep 17 00:00:00 2001 From: Stoyan Bogdanov Date: Mon, 17 Aug 2026 13:14:45 +0300 Subject: [PATCH 498/857] hwmon: (pmbus/tps25990): Add TPS1689 support Extend the existing TPS25990 driver to support the TPS1689 eFuse, as both devices share the same command interface and functionality. Update the documentation to include TPS1689 support. Signed-off-by: Stoyan Bogdanov Link: https://patch.msgid.link/20260817101455.3526260-4-sbogdanov@baylibre.com Signed-off-by: Guenter Roeck --- Documentation/hwmon/tps25990.rst | 15 ++-- drivers/hwmon/pmbus/tps25990.c | 126 ++++++++++++++++++++++++++++--- 2 files changed, 124 insertions(+), 17 deletions(-) diff --git a/Documentation/hwmon/tps25990.rst b/Documentation/hwmon/tps25990.rst index 04faec780d2628..e8bc9a550bda39 100644 --- a/Documentation/hwmon/tps25990.rst +++ b/Documentation/hwmon/tps25990.rst @@ -9,26 +9,31 @@ Supported chips: Prefix: 'tps25990' - * Datasheet + Datasheet: Publicly available at Texas Instruments website: https://www.ti.com/lit/gpn/tps25990 - Publicly available at Texas Instruments website: https://www.ti.com/lit/gpn/tps25990 + * TI TPS1689 + + Prefix: 'tps1689' + + Datasheet: Publicly available at Texas Instruments website: https://www.ti.com/lit/gpn/tps1689 Author: Jerome Brunet + Stoyan Bogdanov Description ----------- -This driver implements support for TI TPS25990 eFuse. +This driver implements support for TI TPS25990 and TI TPS1689 eFuse chips. This is an integrated, high-current circuit protection and power management device with PMBUS interface -Device compliant with: +Devices are compliant with: - PMBus rev 1.3 interface. -Device supports direct format for reading input voltages, +Devices supports direct format for reading input voltages, output voltage, input current, input power and temperature. Due to the specificities of the chip, all history reset attributes diff --git a/drivers/hwmon/pmbus/tps25990.c b/drivers/hwmon/pmbus/tps25990.c index 7634ac743025df..0d5c99053f23f1 100644 --- a/drivers/hwmon/pmbus/tps25990.c +++ b/drivers/hwmon/pmbus/tps25990.c @@ -47,7 +47,14 @@ PK_MIN_AVG_RST_AVG | \ PK_MIN_AVG_RST_MIN) +#define TPS1689_VIN_OV_RANGE_SEL_MASK GENMASK(7, 6) +#define TPS1689_VIN_VOV_MASK GENMASK(5, 0) +#define TPS1689_VIN_SCALING 251 +#define TPS1689_VIN_VOV_STEP_MV 250 +#define TPS1689_VIN_RANGE_SPAN_MV 16000 + enum chips { + tps1689, tps25990, }; @@ -105,6 +112,8 @@ static int tps25990_mfr_write_protect_get(struct i2c_client *client) static int tps25990_read_word_data(struct i2c_client *client, int page, int phase, int reg) { + const struct pmbus_driver_info *info = pmbus_get_driver_info(client); + struct tps25990_data *data = container_of(info, struct tps25990_data, info); int ret; switch (reg) { @@ -193,9 +202,18 @@ static int tps25990_read_word_data(struct i2c_client *client, ret = pmbus_read_word_data(client, page, phase, reg); if (ret < 0) break; - ret = DIV_ROUND_CLOSEST(ret * TPS25990_VIN_OVF_NUM, - TPS25990_VIN_OVF_DIV); - ret += TPS25990_VIN_OVF_OFF; + if (data->chip_id == tps25990) { + ret = DIV_ROUND_CLOSEST(ret * TPS25990_VIN_OVF_NUM, + TPS25990_VIN_OVF_DIV); + ret += TPS25990_VIN_OVF_OFF; + } else if (data->chip_id == tps1689) { + int rng = (FIELD_GET(TPS1689_VIN_OV_RANGE_SEL_MASK, ret) + 1) * + TPS1689_VIN_RANGE_SPAN_MV; + int vov = FIELD_GET(TPS1689_VIN_VOV_MASK, ret) * TPS1689_VIN_VOV_STEP_MV; + + ret = DIV_ROUND_CLOSEST(rng + vov - TPS1689_VIN_RANGE_SPAN_MV, + TPS1689_VIN_SCALING); + } break; case PMBUS_IIN_OC_FAULT_LIMIT: @@ -238,6 +256,8 @@ static int tps25990_read_word_data(struct i2c_client *client, static int tps25990_write_word_data(struct i2c_client *client, int page, int reg, u16 value) { + const struct pmbus_driver_info *info = pmbus_get_driver_info(client); + struct tps25990_data *data = container_of(info, struct tps25990_data, info); int ret; switch (reg) { @@ -249,26 +269,52 @@ static int tps25990_write_word_data(struct i2c_client *client, case PMBUS_OT_WARN_LIMIT: case PMBUS_OT_FAULT_LIMIT: case PMBUS_PIN_OP_WARN_LIMIT: - value >>= TPS25990_8B_SHIFT; + value = clamp_val((s16)value, 0, S16_MAX) >> TPS25990_8B_SHIFT; value = clamp_val(value, 0, 0xff); ret = pmbus_write_word_data(client, page, reg, value); break; case PMBUS_VIN_OV_FAULT_LIMIT: - value -= TPS25990_VIN_OVF_OFF; - value = DIV_ROUND_CLOSEST(((unsigned int)value) * TPS25990_VIN_OVF_DIV, - TPS25990_VIN_OVF_NUM); - value = clamp_val(value, 0, 0xf); + if ((s16)value < 0) + return -EINVAL; + + if (data->chip_id == tps25990) { + int tmp = (int)value - TPS25990_VIN_OVF_OFF; + + tmp = clamp_val(tmp, 0, INT_MAX); + value = DIV_ROUND_CLOSEST((unsigned int)tmp * TPS25990_VIN_OVF_DIV, + TPS25990_VIN_OVF_NUM); + value = clamp_val(value, 0, 0xf); + } else if (data->chip_id == tps1689) { + u32 scaled_value = value * TPS1689_VIN_SCALING + TPS1689_VIN_RANGE_SPAN_MV; + u32 rng_idx = scaled_value / TPS1689_VIN_RANGE_SPAN_MV; + u32 ov_set; + + rng_idx = clamp_val(rng_idx, 1, + FIELD_MAX(TPS1689_VIN_OV_RANGE_SEL_MASK) + 1); + ov_set = scaled_value - (TPS1689_VIN_RANGE_SPAN_MV * rng_idx); + ov_set = min_t(u32, ov_set / TPS1689_VIN_VOV_STEP_MV, + FIELD_MAX(TPS1689_VIN_VOV_MASK)); + value = FIELD_PREP(TPS1689_VIN_OV_RANGE_SEL_MASK, rng_idx - 1) | + FIELD_PREP(TPS1689_VIN_VOV_MASK, ov_set); + } ret = pmbus_write_word_data(client, page, reg, value); break; - case PMBUS_IIN_OC_FAULT_LIMIT: - value -= TPS25990_IIN_OCF_OFF; - value = DIV_ROUND_CLOSEST(((unsigned int)value) * TPS25990_IIN_OCF_DIV, + case PMBUS_IIN_OC_FAULT_LIMIT: { + int tmp; + + if ((s16)value < 0) + return -EINVAL; + + tmp = (int)value - TPS25990_IIN_OCF_OFF; + tmp = clamp_val(tmp, 0, INT_MAX); + value = DIV_ROUND_CLOSEST((unsigned int)tmp * TPS25990_IIN_OCF_DIV, TPS25990_IIN_OCF_NUM); value = clamp_val(value, 0, 0x3f); ret = pmbus_write_byte_data(client, page, TPS25990_VIREF, value); break; + } case PMBUS_VIRT_SAMPLES: value = clamp_val(value, 1, 1 << PK_MIN_AVG_AVG_CNT); @@ -347,6 +393,60 @@ static const struct regulator_desc tps25990_reg_desc[] = { #endif static const struct pmbus_driver_info tps25990_base_info[] = { + [tps1689] = { + .pages = 1, + .format[PSC_VOLTAGE_IN] = direct, + .m[PSC_VOLTAGE_IN] = 3984, + .b[PSC_VOLTAGE_IN] = -63750, + .R[PSC_VOLTAGE_IN] = -3, + .format[PSC_VOLTAGE_OUT] = direct, + .m[PSC_VOLTAGE_OUT] = 1166, + .b[PSC_VOLTAGE_OUT] = 0, + .R[PSC_VOLTAGE_OUT] = -2, + .format[PSC_TEMPERATURE] = direct, + .m[PSC_TEMPERATURE] = 140, + .b[PSC_TEMPERATURE] = 32103, + .R[PSC_TEMPERATURE] = -2, + /* + * Current and Power measurement depends on the ohm value + * of Rimon. m is multiplied by 1000 below to have an integer + * and -3 is added to R to compensate. + */ + .format[PSC_CURRENT_IN] = direct, + .m[PSC_CURRENT_IN] = 9548, + .b[PSC_CURRENT_IN] = 0, + .R[PSC_CURRENT_IN] = -6, + .format[PSC_CURRENT_OUT] = direct, + .m[PSC_CURRENT_OUT] = 24347, + .b[PSC_CURRENT_OUT] = 0, + .R[PSC_CURRENT_OUT] = -3, + .format[PSC_POWER] = direct, + .m[PSC_POWER] = 2775, + .b[PSC_POWER] = 0, + .R[PSC_POWER] = -4, + .func[0] = (PMBUS_HAVE_VIN | + PMBUS_HAVE_VOUT | + PMBUS_HAVE_VMON | + PMBUS_HAVE_IIN | + PMBUS_HAVE_IOUT | + PMBUS_HAVE_PIN | + PMBUS_HAVE_TEMP | + PMBUS_HAVE_STATUS_VOUT | + PMBUS_HAVE_STATUS_IOUT | + PMBUS_HAVE_STATUS_INPUT | + PMBUS_HAVE_STATUS_TEMP | + PMBUS_HAVE_SAMPLES), + + .read_word_data = tps25990_read_word_data, + .write_word_data = tps25990_write_word_data, + .read_byte_data = tps25990_read_byte_data, + .write_byte_data = tps25990_write_byte_data, + +#if IS_ENABLED(CONFIG_SENSORS_TPS25990_REGULATOR) + .reg_desc = tps25990_reg_desc, + .num_regulators = ARRAY_SIZE(tps25990_reg_desc), +#endif + }, [tps25990] = { .pages = 1, .format[PSC_VOLTAGE_IN] = direct, @@ -389,7 +489,6 @@ static const struct pmbus_driver_info tps25990_base_info[] = { .write_word_data = tps25990_write_word_data, .read_byte_data = tps25990_read_byte_data, .write_byte_data = tps25990_write_byte_data, - #if IS_ENABLED(CONFIG_SENSORS_TPS25990_REGULATOR) .reg_desc = tps25990_reg_desc, .num_regulators = ARRAY_SIZE(tps25990_reg_desc), @@ -398,12 +497,14 @@ static const struct pmbus_driver_info tps25990_base_info[] = { }; static const struct i2c_device_id tps25990_i2c_id[] = { + { .name = "tps1689", .driver_data = tps1689 }, { .name = "tps25990", .driver_data = tps25990 }, {} }; MODULE_DEVICE_TABLE(i2c, tps25990_i2c_id); static const struct of_device_id tps25990_of_match[] = { + { .compatible = "ti,tps1689", .data = (void *)tps1689 }, { .compatible = "ti,tps25990", .data = (void *)tps25990 }, {} }; @@ -438,6 +539,7 @@ static int tps25990_probe(struct i2c_client *client) /* Adapt the current and power scale for each instance */ tps25990_set_m(&data->info.m[PSC_CURRENT_IN], rimon); + tps25990_set_m(&data->info.m[PSC_CURRENT_OUT], rimon); tps25990_set_m(&data->info.m[PSC_POWER], rimon); return pmbus_do_probe(client, &data->info); From e73cf6da9fd79820c164a186c8692cac12ad80d3 Mon Sep 17 00:00:00 2001 From: Alessandro Zini Date: Fri, 21 Aug 2026 11:19:23 +0200 Subject: [PATCH 499/857] dt-bindings: trivial-devices: Add Sensirion STS4x series Add "sensirion,sts40" as compatible for the Sensirion STS4x/STS4xA series of digital temperature sensors. Link: https://sensirion.com/resource/datasheet/sts4x Acked-by: Conor Dooley Signed-off-by: Alessandro Zini Link: https://patch.msgid.link/20260821091924.18975-2-alessandro.zini@siemens.com Signed-off-by: Guenter Roeck --- Documentation/devicetree/bindings/trivial-devices.yaml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/Documentation/devicetree/bindings/trivial-devices.yaml b/Documentation/devicetree/bindings/trivial-devices.yaml index 2de8eb09cb7d14..a82460e2416fd7 100644 --- a/Documentation/devicetree/bindings/trivial-devices.yaml +++ b/Documentation/devicetree/bindings/trivial-devices.yaml @@ -427,6 +427,8 @@ properties: - sensirion,sht21 - sensirion,sht25 - sensirion,sht4x + # Sensirion temperature sensor with I2C interface + - sensirion,sts40 # Sensortek 3 axis accelerometer - sensortek,stk8312 # Sensortek 3 axis accelerometer From bd9e720b8a7a041733187606e7e81d01446ebf22 Mon Sep 17 00:00:00 2001 From: Alessandro Zini Date: Fri, 21 Aug 2026 11:19:24 +0200 Subject: [PATCH 500/857] hwmon: (sht4x): Add support for Sensirion STS4x temperature sensors The Sensirion STS4x series is the temperature-only variant of the SHT4x family. It shares the same I2C command set, conversion formulas, CRC checksum, and timing with the SHT4x, but only returns temperature data (3 bytes: 2 data bytes + 1 CRC byte). Add support for the STS4x series by dynamically adjusting the read response length, suppressing humidity channel attributes when probed as STS4x, and omitting heater sysfs attributes. Link: https://sensirion.com/resource/datasheet/sts4x Signed-off-by: Alessandro Zini Link: https://patch.msgid.link/20260821091924.18975-3-alessandro.zini@siemens.com Signed-off-by: Guenter Roeck --- Documentation/hwmon/sht4x.rst | 17 ++++++++-- drivers/hwmon/sht4x.c | 62 +++++++++++++++++++++++++---------- 2 files changed, 59 insertions(+), 20 deletions(-) diff --git a/Documentation/hwmon/sht4x.rst b/Documentation/hwmon/sht4x.rst index ba094ad0e2816f..b9564632a1be2d 100644 --- a/Documentation/hwmon/sht4x.rst +++ b/Documentation/hwmon/sht4x.rst @@ -15,6 +15,16 @@ Supported Chips: English: https://www.sensirion.com/fileadmin/user_upload/customers/sensirion/Dokumente/2_Humidity_Sensors/Datasheets/Sensirion_Humidity_Sensors_SHT4x_Datasheet.pdf + * Sensirion STS4X + + Prefix: 'sts4x' + + Addresses scanned: None + + Datasheet: + + English: https://sensirion.com/resource/datasheet/sts4x + Author: Navin Sankar Velliangiri @@ -22,9 +32,10 @@ Description ----------- This driver implements support for the Sensirion SHT4x chip, a humidity -and temperature sensor. Temperature is measured in degree celsius, relative -humidity is expressed as a percentage. In sysfs interface, all values are -scaled by 1000, i.e. the value for 31.5 degrees celsius is 31500. +and temperature sensor, and the Sensirion STS4x chip, a temperature sensor. +Temperature is measured in degree celsius, relative humidity is expressed as a +percentage (on SHT4x only). In sysfs interface, all values are scaled by 1000, +i.e. the value for 31.5 degrees celsius is 31500. Usage Notes ----------- diff --git a/drivers/hwmon/sht4x.c b/drivers/hwmon/sht4x.c index a97dda9e92dc5d..fd1f950448f064 100644 --- a/drivers/hwmon/sht4x.c +++ b/drivers/hwmon/sht4x.c @@ -43,6 +43,7 @@ #define SHT4X_CRC8_LEN 1 #define SHT4X_WORD_LEN 2 #define SHT4X_RESPONSE_LENGTH 6 +#define STS4X_RESPONSE_LENGTH 3 #define SHT4X_CRC8_POLYNOMIAL 0x31 #define SHT4X_CRC8_INIT 0xff #define SHT4X_MIN_TEMPERATURE -45000 @@ -52,9 +53,15 @@ DECLARE_CRC8_TABLE(sht4x_crc8_table); +enum sht4x_chips { + sht4x, + sts4x, +}; + /** * struct sht4x_data - All the data required to operate an SHT4X chip * @client: the i2c client associated with the SHT4X + * @chip_id: the chip type (sht4x or sts4x) * @heating_complete: the time that the last heating finished * @data_pending: true if and only if there are measurements to retrieve after heating * @heater_power: the power at which the heater will be started @@ -67,6 +74,7 @@ DECLARE_CRC8_TABLE(sht4x_crc8_table); */ struct sht4x_data { struct i2c_client *client; + enum sht4x_chips chip_id; unsigned long heating_complete; /* in jiffies */ bool data_pending; u32 heater_power; /* in milli-watts */ @@ -92,11 +100,15 @@ static int sht4x_read_values(struct sht4x_data *data) u8 crc; u8 cmd[SHT4X_CMD_LEN] = {SHT4X_CMD_MEASURE_HPM}; u8 raw_data[SHT4X_RESPONSE_LENGTH]; + size_t response_length = data->chip_id == sts4x ? + STS4X_RESPONSE_LENGTH : SHT4X_RESPONSE_LENGTH; unsigned long curr_jiffies; - curr_jiffies = jiffies; - if (time_before(curr_jiffies, data->heating_complete)) - msleep(jiffies_to_msecs(data->heating_complete - curr_jiffies)); + if (data->chip_id != sts4x) { + curr_jiffies = jiffies; + if (time_before(curr_jiffies, data->heating_complete)) + msleep(jiffies_to_msecs(data->heating_complete - curr_jiffies)); + } if (data->data_pending && time_before(jiffies, data->heating_complete + data->update_interval)) { @@ -115,15 +127,14 @@ static int sht4x_read_values(struct sht4x_data *data) usleep_range(SHT4X_MEAS_DELAY_HPM, SHT4X_MEAS_DELAY_HPM + SHT4X_DELAY_EXTRA); } - ret = i2c_master_recv(client, raw_data, SHT4X_RESPONSE_LENGTH); - if (ret != SHT4X_RESPONSE_LENGTH) { + ret = i2c_master_recv(client, raw_data, response_length); + if (ret != response_length) { if (ret >= 0) ret = -ENODATA; return ret; } t_ticks = raw_data[0] << 8 | raw_data[1]; - rh_ticks = raw_data[3] << 8 | raw_data[4]; crc = crc8(sht4x_crc8_table, &raw_data[0], SHT4X_WORD_LEN, CRC8_INIT_VALUE); if (crc != raw_data[2]) { @@ -131,14 +142,19 @@ static int sht4x_read_values(struct sht4x_data *data) return -EIO; } - crc = crc8(sht4x_crc8_table, &raw_data[3], SHT4X_WORD_LEN, CRC8_INIT_VALUE); - if (crc != raw_data[5]) { - dev_err(&client->dev, "data integrity check failed\n"); - return -EIO; + data->temperature = ((21875 * (int32_t)t_ticks) >> 13) - 45000; + + if (data->chip_id != sts4x) { + rh_ticks = raw_data[3] << 8 | raw_data[4]; + crc = crc8(sht4x_crc8_table, &raw_data[3], SHT4X_WORD_LEN, CRC8_INIT_VALUE); + if (crc != raw_data[5]) { + dev_err(&client->dev, "data integrity check failed\n"); + return -EIO; + } + + data->humidity = ((15625 * (int32_t)rh_ticks) >> 13) - 6000; } - data->temperature = ((21875 * (int32_t)t_ticks) >> 13) - 45000; - data->humidity = ((15625 * (int32_t)rh_ticks) >> 13) - 6000; data->last_updated = jiffies; data->valid = true; return 0; @@ -190,9 +206,14 @@ static umode_t sht4x_hwmon_visible(const void *data, enum hwmon_sensor_types type, u32 attr, int channel) { + const struct sht4x_data *chip_data = data; + switch (type) { case hwmon_temp: + return 0444; case hwmon_humidity: + if (chip_data->chip_id == sts4x) + return 0; return 0444; case hwmon_chip: return 0644; @@ -388,6 +409,7 @@ static const struct hwmon_chip_info sht4x_chip_info = { static int sht4x_probe(struct i2c_client *client) { + const struct attribute_group **groups = NULL; struct device *device = &client->dev; struct device *hwmon_dev; struct sht4x_data *data; @@ -406,11 +428,15 @@ static int sht4x_probe(struct i2c_client *client) if (!data) return -ENOMEM; + data->chip_id = (uintptr_t)i2c_get_match_data(client); data->update_interval = SHT4X_MIN_POLL_INTERVAL; data->client = client; - data->heater_power = 200; - data->heater_time = 1000; data->heating_complete = jiffies; + if (data->chip_id != sts4x) { + data->heater_power = 200; + data->heater_time = 1000; + groups = sht4x_groups; + } crc8_populate_msb(sht4x_crc8_table, SHT4X_CRC8_POLYNOMIAL); @@ -424,19 +450,21 @@ static int sht4x_probe(struct i2c_client *client) client->name, data, &sht4x_chip_info, - sht4x_groups); + groups); return PTR_ERR_OR_ZERO(hwmon_dev); } static const struct i2c_device_id sht4x_id[] = { - { .name = "sht4x" }, + { .name = "sht4x", .driver_data = sht4x }, + { .name = "sts4x", .driver_data = sts4x }, { } }; MODULE_DEVICE_TABLE(i2c, sht4x_id); static const struct of_device_id sht4x_of_match[] = { - { .compatible = "sensirion,sht4x" }, + { .compatible = "sensirion,sht4x", .data = (void *)sht4x }, + { .compatible = "sensirion,sts40", .data = (void *)sts4x }, { } }; MODULE_DEVICE_TABLE(of, sht4x_of_match); From c2e0011ee7eee1ac166c9e77202e8ae32a996f81 Mon Sep 17 00:00:00 2001 From: "benoit.masson" Date: Sun, 30 Aug 2026 00:01:47 +0200 Subject: [PATCH 501/857] hwmon: it87: describe per-chip PWM temperature maps Add a per-chip count for PWM temperature mapping sources and use it when reporting and validating mappings. Keep existing chips on their previous three-source defaults. This prepares the driver for chips with a different number of mapping sources. Signed-off-by: benoit.masson Signed-off-by: Jerome Tollet Link: https://patch.msgid.link/619425df92463b3c3f9e90f00c527c4009599ea8.1788040385.git.jerome.tollet@gmail.com Signed-off-by: Guenter Roeck --- drivers/hwmon/it87.c | 46 ++++++++++++++++++++++++++++++++++++-------- 1 file changed, 38 insertions(+), 8 deletions(-) diff --git a/drivers/hwmon/it87.c b/drivers/hwmon/it87.c index 87edb1b6048bb5..5cb1c002904146 100644 --- a/drivers/hwmon/it87.c +++ b/drivers/hwmon/it87.c @@ -292,6 +292,7 @@ struct it87_devices { const char *name; const char * const model; u32 features; + u8 num_temp_map; u8 peci_mask; u8 old_peci_mask; u8 smbus_bitmap; /* SMBus enable bits in extra config register */ @@ -335,12 +336,14 @@ static const struct it87_devices it87_devices[] = { .model = "IT87F", .features = FEAT_OLD_AUTOPWM | FEAT_FANCTL_ONOFF, /* may need to overwrite */ + .num_temp_map = 3, }, [it8712] = { .name = "it8712", .model = "IT8712F", .features = FEAT_OLD_AUTOPWM | FEAT_VID | FEAT_FANCTL_ONOFF, /* may need to overwrite */ + .num_temp_map = 3, }, [it8716] = { .name = "it8716", @@ -348,6 +351,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_VID | FEAT_FAN16_CONFIG | FEAT_FIVE_FANS | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, }, [it8718] = { .name = "it8718", @@ -355,6 +359,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_VID | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_FIVE_FANS | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .old_peci_mask = 0x4, }, [it8720] = { @@ -363,6 +368,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_VID | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_FIVE_FANS | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .old_peci_mask = 0x4, }, [it8721] = { @@ -372,6 +378,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_TEMP_PECI | FEAT_FAN16_CONFIG | FEAT_FIVE_FANS | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .peci_mask = 0x05, .old_peci_mask = 0x02, /* Actually reports PCH */ }, @@ -382,6 +389,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_FIVE_FANS | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .peci_mask = 0x07, }, [it8732] = { @@ -391,6 +399,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_TEMP_PECI | FEAT_10_9MV_ADC | FEAT_IN7_INTERNAL | FEAT_FOUR_FANS | FEAT_FOUR_PWM | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .peci_mask = 0x07, .old_peci_mask = 0x02, /* Actually reports PCH */ }, @@ -404,6 +413,7 @@ static const struct it87_devices it87_devices[] = { /* 12mV ADC (OHM) */ /* 16 bit fans (OHM) */ /* three fans, always 16 bit (guesswork) */ + .num_temp_map = 3, .peci_mask = 0x07, }, [it8772] = { @@ -416,6 +426,7 @@ static const struct it87_devices it87_devices[] = { /* 12mV ADC (HWSensors4, OHM) */ /* 16 bit fans (HWSensors4, OHM) */ /* three fans, always 16 bit (datasheet) */ + .num_temp_map = 3, .peci_mask = 0x07, }, [it8781] = { @@ -424,6 +435,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .old_peci_mask = 0x4, }, [it8782] = { @@ -432,6 +444,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .old_peci_mask = 0x4, }, [it8783] = { @@ -440,6 +453,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .old_peci_mask = 0x4, }, [it8786] = { @@ -448,6 +462,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_NEWER_AUTOPWM | FEAT_12MV_ADC | FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .peci_mask = 0x07, }, [it8790] = { @@ -456,6 +471,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_NEWER_AUTOPWM | FEAT_12MV_ADC | FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF | FEAT_NOCONF, + .num_temp_map = 3, .peci_mask = 0x07, }, [it8792] = { @@ -465,6 +481,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_TEMP_PECI | FEAT_10_9MV_ADC | FEAT_IN7_INTERNAL | FEAT_FANCTL_ONOFF | FEAT_NOCONF, + .num_temp_map = 3, .peci_mask = 0x07, .old_peci_mask = 0x02, /* Actually reports PCH */ }, @@ -474,6 +491,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_NEWER_AUTOPWM | FEAT_12MV_ADC | FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_IN7_INTERNAL | FEAT_AVCC3 | FEAT_PWM_FREQ2, + .num_temp_map = 3, .peci_mask = 0x07, }, [it8620] = { @@ -483,6 +501,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_SIX_FANS | FEAT_IN7_INTERNAL | FEAT_SIX_PWM | FEAT_PWM_FREQ2 | FEAT_SIX_TEMP | FEAT_VIN3_5V | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .peci_mask = 0x07, }, [it8622] = { @@ -492,6 +511,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_FIVE_FANS | FEAT_FIVE_PWM | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_AVCC3 | FEAT_VIN3_5V | FEAT_FOUR_TEMP, + .num_temp_map = 3, .peci_mask = 0x07, .smbus_bitmap = BIT(1) | BIT(2), }, @@ -502,6 +522,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_SIX_FANS | FEAT_IN7_INTERNAL | FEAT_SIX_PWM | FEAT_PWM_FREQ2 | FEAT_SIX_TEMP | FEAT_VIN3_5V | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .peci_mask = 0x07, }, [it8689] = { @@ -511,6 +532,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_SIX_FANS | FEAT_IN7_INTERNAL | FEAT_SIX_PWM | FEAT_PWM_FREQ2 | FEAT_SIX_TEMP | FEAT_AVCC3 | FEAT_FANCTL_ONOFF, + .num_temp_map = 3, .smbus_bitmap = BIT(1) | BIT(2), }, [it87952] = { @@ -520,6 +542,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_TEMP_PECI | FEAT_10_9MV_ADC | FEAT_IN7_INTERNAL | FEAT_FANCTL_ONOFF | FEAT_NOCONF, + .num_temp_map = 3, .peci_mask = 0x07, .old_peci_mask = 0x02, /* Actually reports PCH */ }, @@ -589,6 +612,7 @@ struct it87_data { int sioaddr; enum chips type; u32 features; + u8 num_temp_map; u8 peci_mask; u8 old_peci_mask; @@ -1693,16 +1717,18 @@ static ssize_t show_pwm_temp_map(struct device *dev, struct sensor_device_attribute *sensor_attr = to_sensor_dev_attr(attr); struct it87_data *data = it87_update_device(dev); int nr = sensor_attr->index; + u8 num_map; int map; if (IS_ERR(data)) return PTR_ERR(data); + num_map = data->num_temp_map; map = data->pwm_temp_map[nr]; - if (map >= 3) + if (map >= num_map) map = 0; /* Should never happen */ - if (nr >= 3) /* pwm channels 3..6 map to temp4..6 */ - map += 3; + if (nr >= num_map) /* pwm channels 3..6 map to temp4..6 */ + map += num_map; return sprintf(buf, "%d\n", (int)BIT(map)); } @@ -1714,6 +1740,7 @@ static ssize_t set_pwm_temp_map(struct device *dev, struct sensor_device_attribute *sensor_attr = to_sensor_dev_attr(attr); struct it87_data *data = dev_get_drvdata(dev); int nr = sensor_attr->index; + u8 num_map = data->num_temp_map; long val; int err; u8 reg; @@ -1721,8 +1748,8 @@ static ssize_t set_pwm_temp_map(struct device *dev, if (kstrtol(buf, 10, &val) < 0) return -EINVAL; - if (nr >= 3) - val -= 3; + if (nr >= num_map) + val -= num_map; switch (val) { case BIT(0): @@ -3461,6 +3488,7 @@ static int it87_probe(struct platform_device *pdev) struct resource *res; struct device *dev = &pdev->dev; struct it87_sio_data *sio_data = dev_get_platdata(dev); + const struct it87_devices *chip; int enable_pwm_interface; struct device *hwmon_dev; int err; @@ -3483,9 +3511,11 @@ static int it87_probe(struct platform_device *pdev) data->type = sio_data->type; data->smbus_bitmap = sio_data->smbus_bitmap; data->ec_special_config = sio_data->ec_special_config; - data->features = it87_devices[sio_data->type].features; - data->peci_mask = it87_devices[sio_data->type].peci_mask; - data->old_peci_mask = it87_devices[sio_data->type].old_peci_mask; + chip = &it87_devices[sio_data->type]; + data->features = chip->features; + data->peci_mask = chip->peci_mask; + data->old_peci_mask = chip->old_peci_mask; + data->num_temp_map = chip->num_temp_map; /* * IT8705F Datasheet 0.4.1, 3h == Version G. * IT8712F Datasheet 0.9.1, section 8.3.5 indicates 8h == Version J. From 01dc25baf5bb8032b3550f93b0a06d2022c5096e Mon Sep 17 00:00:00 2001 From: "benoit.masson" Date: Sun, 30 Aug 2026 00:01:48 +0200 Subject: [PATCH 502/857] hwmon: it87: prepare for extended PWM temp maps Introduce helper logic for PWM-to-temperature mappings so newer register layouts can be supported while retaining the legacy two groups of three temperature sources. Honor the four global temperature sources on IT8603E and IT8622E instead of applying the legacy grouping to those chips. Use per-chip masks and shifts for newer extended mappings. Newer controllers keep the duty cycle in a separate register, so write their temperature mapping in both manual and automatic mode. This keeps the selected mapping across cache refreshes and mode changes. On older controllers, defer mapping writes while in manual mode and apply the cached mapping when switching to automatic mode. Signed-off-by: benoit.masson Signed-off-by: Jerome Tollet Link: https://patch.msgid.link/7b4f2befc3d214b646c0582e6410f17e78748b4e.1788040385.git.jerome.tollet@gmail.com Signed-off-by: Guenter Roeck --- drivers/hwmon/it87.c | 219 ++++++++++++++++++++++++++++++++----------- 1 file changed, 166 insertions(+), 53 deletions(-) diff --git a/drivers/hwmon/it87.c b/drivers/hwmon/it87.c index 5cb1c002904146..232cc858625872 100644 --- a/drivers/hwmon/it87.c +++ b/drivers/hwmon/it87.c @@ -252,6 +252,7 @@ static const u8 IT87_REG_TEMP_OFFSET[] = { 0x56, 0x57, 0x59 }; #define IT87_REG_FAN_MAIN_CTRL 0x13 #define IT87_REG_FAN_CTL 0x14 static const u8 IT87_REG_PWM[] = { 0x15, 0x16, 0x17, 0x7f, 0xa7, 0xaf }; +static const u8 IT87_REG_PWM_8665[] = { 0x15, 0x16, 0x17, 0x1e, 0x1f, 0x92 }; static const u8 IT87_REG_PWM_DUTY[] = { 0x63, 0x6b, 0x73, 0x7b, 0xa3, 0xab }; static const u8 IT87_REG_VIN[] = { 0x20, 0x21, 0x22, 0x23, 0x24, 0x25, 0x26, @@ -283,6 +284,7 @@ static const u8 IT87_REG_AUTO_BASE[] = { 0x60, 0x68, 0x70, 0x78, 0xa0, 0xa8 }; #define NUM_TEMP 6 #define NUM_TEMP_OFFSET ARRAY_SIZE(IT87_REG_TEMP_OFFSET) #define NUM_TEMP_LIMIT 3 +#define IT87_PWM_OLD_NUM_TEMP 3 #define NUM_FAN ARRAY_SIZE(IT87_REG_FAN) #define NUM_FAN_DIV 3 #define NUM_PWM ARRAY_SIZE(IT87_REG_PWM) @@ -292,6 +294,7 @@ struct it87_devices { const char *name; const char * const model; u32 features; + const u8 *reg_pwm; u8 num_temp_map; u8 peci_mask; u8 old_peci_mask; @@ -329,6 +332,7 @@ struct it87_devices { #define FEAT_FOUR_PWM BIT(21) /* Supports four fan controls */ #define FEAT_FOUR_TEMP BIT(22) #define FEAT_FANCTL_ONOFF BIT(23) /* chip has FAN_CTL ON/OFF */ +#define FEAT_NEW_TEMPMAP BIT(24) /* PWM uses extended temp map */ static const struct it87_devices it87_devices[] = { [it87] = { @@ -336,6 +340,7 @@ static const struct it87_devices it87_devices[] = { .model = "IT87F", .features = FEAT_OLD_AUTOPWM | FEAT_FANCTL_ONOFF, /* may need to overwrite */ + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, }, [it8712] = { @@ -343,6 +348,7 @@ static const struct it87_devices it87_devices[] = { .model = "IT8712F", .features = FEAT_OLD_AUTOPWM | FEAT_VID | FEAT_FANCTL_ONOFF, /* may need to overwrite */ + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, }, [it8716] = { @@ -351,6 +357,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_VID | FEAT_FAN16_CONFIG | FEAT_FIVE_FANS | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, }, [it8718] = { @@ -359,6 +366,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_VID | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_FIVE_FANS | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .old_peci_mask = 0x4, }, @@ -368,6 +376,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_VID | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_FIVE_FANS | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .old_peci_mask = 0x4, }, @@ -378,6 +387,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_TEMP_PECI | FEAT_FAN16_CONFIG | FEAT_FIVE_FANS | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x05, .old_peci_mask = 0x02, /* Actually reports PCH */ @@ -389,6 +399,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_FIVE_FANS | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, }, @@ -399,6 +410,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_TEMP_PECI | FEAT_10_9MV_ADC | FEAT_IN7_INTERNAL | FEAT_FOUR_FANS | FEAT_FOUR_PWM | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, .old_peci_mask = 0x02, /* Actually reports PCH */ @@ -413,6 +425,7 @@ static const struct it87_devices it87_devices[] = { /* 12mV ADC (OHM) */ /* 16 bit fans (OHM) */ /* three fans, always 16 bit (guesswork) */ + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, }, @@ -426,6 +439,7 @@ static const struct it87_devices it87_devices[] = { /* 12mV ADC (HWSensors4, OHM) */ /* 16 bit fans (HWSensors4, OHM) */ /* three fans, always 16 bit (datasheet) */ + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, }, @@ -435,6 +449,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .old_peci_mask = 0x4, }, @@ -444,6 +459,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .old_peci_mask = 0x4, }, @@ -453,6 +469,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_FAN16_CONFIG | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .old_peci_mask = 0x4, }, @@ -462,6 +479,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_NEWER_AUTOPWM | FEAT_12MV_ADC | FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, }, @@ -471,6 +489,7 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_NEWER_AUTOPWM | FEAT_12MV_ADC | FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_FANCTL_ONOFF | FEAT_NOCONF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, }, @@ -481,6 +500,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_TEMP_PECI | FEAT_10_9MV_ADC | FEAT_IN7_INTERNAL | FEAT_FANCTL_ONOFF | FEAT_NOCONF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, .old_peci_mask = 0x02, /* Actually reports PCH */ @@ -491,7 +511,8 @@ static const struct it87_devices it87_devices[] = { .features = FEAT_NEWER_AUTOPWM | FEAT_12MV_ADC | FEAT_16BIT_FANS | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_IN7_INTERNAL | FEAT_AVCC3 | FEAT_PWM_FREQ2, - .num_temp_map = 3, + .reg_pwm = IT87_REG_PWM, + .num_temp_map = 4, .peci_mask = 0x07, }, [it8620] = { @@ -501,6 +522,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_SIX_FANS | FEAT_IN7_INTERNAL | FEAT_SIX_PWM | FEAT_PWM_FREQ2 | FEAT_SIX_TEMP | FEAT_VIN3_5V | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, }, @@ -511,7 +533,8 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_FIVE_FANS | FEAT_FIVE_PWM | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 | FEAT_AVCC3 | FEAT_VIN3_5V | FEAT_FOUR_TEMP, - .num_temp_map = 3, + .reg_pwm = IT87_REG_PWM_8665, + .num_temp_map = 4, .peci_mask = 0x07, .smbus_bitmap = BIT(1) | BIT(2), }, @@ -522,6 +545,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_SIX_FANS | FEAT_IN7_INTERNAL | FEAT_SIX_PWM | FEAT_PWM_FREQ2 | FEAT_SIX_TEMP | FEAT_VIN3_5V | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, }, @@ -532,6 +556,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_SIX_FANS | FEAT_IN7_INTERNAL | FEAT_SIX_PWM | FEAT_PWM_FREQ2 | FEAT_SIX_TEMP | FEAT_AVCC3 | FEAT_FANCTL_ONOFF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .smbus_bitmap = BIT(1) | BIT(2), }, @@ -542,6 +567,7 @@ static const struct it87_devices it87_devices[] = { | FEAT_TEMP_OFFSET | FEAT_TEMP_OLD_PECI | FEAT_TEMP_PECI | FEAT_10_9MV_ADC | FEAT_IN7_INTERNAL | FEAT_FANCTL_ONOFF | FEAT_NOCONF, + .reg_pwm = IT87_REG_PWM, .num_temp_map = 3, .peci_mask = 0x07, .old_peci_mask = 0x02, /* Actually reports PCH */ @@ -583,6 +609,7 @@ static const struct it87_devices it87_devices[] = { #define has_scaling(data) ((data)->features & (FEAT_12MV_ADC | \ FEAT_10_9MV_ADC)) #define has_fanctl_onoff(data) ((data)->features & FEAT_FANCTL_ONOFF) +#define has_new_tempmap(data) ((data)->features & FEAT_NEW_TEMPMAP) struct it87_sio_data { int sioaddr; @@ -612,6 +639,7 @@ struct it87_data { int sioaddr; enum chips type; u32 features; + const u8 *reg_pwm; u8 num_temp_map; u8 peci_mask; u8 old_peci_mask; @@ -659,7 +687,9 @@ struct it87_data { u8 has_pwm; /* Bitfield, pwm control enabled */ u8 pwm_ctrl[NUM_PWM]; /* Register value */ u8 pwm_duty[NUM_PWM]; /* Manual PWM value set by user */ - u8 pwm_temp_map[NUM_PWM];/* PWM to temp. chan. mapping (bits 1-0) */ + u8 pwm_temp_map[NUM_PWM];/* PWM to temp. chan. mapping */ + u8 pwm_temp_map_mask; + u8 pwm_temp_map_shift; /* Automatic fan speed control registers */ u8 auto_pwm[NUM_AUTO_PWM][4]; /* [nr][3] is hard-coded */ @@ -741,6 +771,77 @@ static int pwm_from_reg(const struct it87_data *data, u8 reg) return (reg & 0x7f) << 1; } +static inline u8 pwm_temp_map_get(const struct it87_data *data, u8 ctrl) +{ + return (ctrl >> data->pwm_temp_map_shift) & + data->pwm_temp_map_mask; +} + +static inline u8 pwm_temp_map_set(const struct it87_data *data, u8 ctrl, + u8 map) +{ + ctrl &= ~(data->pwm_temp_map_mask << data->pwm_temp_map_shift); + return ctrl | ((map & data->pwm_temp_map_mask) + << data->pwm_temp_map_shift); +} + +static inline u8 pwm_num_temp_map(const struct it87_data *data) +{ + return data->num_temp_map; +} + +static inline bool uses_global_temp_map(const struct it87_data *data) +{ + return has_new_tempmap(data) || + pwm_num_temp_map(data) != IT87_PWM_OLD_NUM_TEMP; +} + +static unsigned int pwm_temp_channel(const struct it87_data *data, + int nr, u8 map) +{ + if (uses_global_temp_map(data)) { + u8 num = pwm_num_temp_map(data); + + if (map >= num) + map = 0; + return map; + } + + if (map >= IT87_PWM_OLD_NUM_TEMP) + map = 0; + + if (nr >= IT87_PWM_OLD_NUM_TEMP) + map += IT87_PWM_OLD_NUM_TEMP; + + return map; +} + +static int pwm_temp_map_from_channel(const struct it87_data *data, int nr, + unsigned int channel, u8 *map) +{ + if (uses_global_temp_map(data)) { + u8 num = pwm_num_temp_map(data); + + if (channel >= num) + return -EINVAL; + *map = channel; + return 0; + } + + if (nr >= IT87_PWM_OLD_NUM_TEMP) { + if (channel < IT87_PWM_OLD_NUM_TEMP || + channel >= 2 * IT87_PWM_OLD_NUM_TEMP) + return -EINVAL; + channel -= IT87_PWM_OLD_NUM_TEMP; + } else { + if (channel >= IT87_PWM_OLD_NUM_TEMP) + return -EINVAL; + } + + *map = channel; + return 0; +} + static int DIV_TO_REG(int val) { int answer = 0; @@ -752,6 +853,11 @@ static int DIV_TO_REG(int val) #define DIV_FROM_REG(val) BIT(val) +static inline u16 it87_reg_pwm(const struct it87_data *data, int nr) +{ + return data->reg_pwm[nr]; +} + /* * PWM base frequencies. The frequency has to be divided by either 128 or 256, * depending on the chip type, to calculate the actual PWM frequency. @@ -832,16 +938,23 @@ static void it87_write_value(struct it87_data *data, u8 reg, u8 value) static void it87_update_pwm_ctrl(struct it87_data *data, int nr) { - data->pwm_ctrl[nr] = it87_read_value(data, IT87_REG_PWM[nr]); + data->pwm_ctrl[nr] = it87_read_value(data, it87_reg_pwm(data, nr)); if (has_newer_autopwm(data)) { - data->pwm_temp_map[nr] = data->pwm_ctrl[nr] & 0x03; + data->pwm_temp_map[nr] = + pwm_temp_map_get(data, data->pwm_ctrl[nr]); + if (uses_global_temp_map(data) && + data->pwm_temp_map[nr] >= pwm_num_temp_map(data)) + data->pwm_temp_map[nr] = 0; data->pwm_duty[nr] = it87_read_value(data, IT87_REG_PWM_DUTY[nr]); - } else { - if (data->pwm_ctrl[nr] & 0x80) /* Automatic mode */ - data->pwm_temp_map[nr] = data->pwm_ctrl[nr] & 0x03; - else /* Manual mode */ - data->pwm_duty[nr] = data->pwm_ctrl[nr] & 0x7f; + } else if (data->pwm_ctrl[nr] & 0x80) { /* Automatic mode */ + data->pwm_temp_map[nr] = + pwm_temp_map_get(data, data->pwm_ctrl[nr]); + if (uses_global_temp_map(data) && + data->pwm_temp_map[nr] >= pwm_num_temp_map(data)) + data->pwm_temp_map[nr] = 0; + } else { /* Manual mode */ + data->pwm_duty[nr] = data->pwm_ctrl[nr] & 0x7f; } if (has_old_autopwm(data)) { @@ -1591,27 +1704,32 @@ static ssize_t set_pwm_enable(struct device *dev, struct device_attribute *attr, data->pwm_duty[nr]); /* and set manual mode */ if (has_newer_autopwm(data)) { - ctrl = (data->pwm_ctrl[nr] & 0x7c) | - data->pwm_temp_map[nr]; + ctrl = pwm_temp_map_set(data, + data->pwm_ctrl[nr] & + ~0x80, + data->pwm_temp_map[nr]); } else { ctrl = data->pwm_duty[nr]; } data->pwm_ctrl[nr] = ctrl; - it87_write_value(data, IT87_REG_PWM[nr], ctrl); + it87_write_value(data, it87_reg_pwm(data, nr), ctrl); } } else { u8 ctrl; if (has_newer_autopwm(data)) { - ctrl = (data->pwm_ctrl[nr] & 0x7c) | - data->pwm_temp_map[nr]; + ctrl = pwm_temp_map_set(data, + data->pwm_ctrl[nr] & ~0x80, + data->pwm_temp_map[nr]); if (val != 1) ctrl |= 0x80; } else { - ctrl = (val == 1 ? data->pwm_duty[nr] : 0x80); + ctrl = val == 1 ? data->pwm_duty[nr] : + pwm_temp_map_set(data, 0x80, + data->pwm_temp_map[nr]); } data->pwm_ctrl[nr] = ctrl; - it87_write_value(data, IT87_REG_PWM[nr], ctrl); + it87_write_value(data, it87_reg_pwm(data, nr), ctrl); if (has_fanctl_onoff(data) && nr < 3) { /* set SmartGuardian mode */ @@ -1662,7 +1780,7 @@ static ssize_t set_pwm(struct device *dev, struct device_attribute *attr, */ if (!(data->pwm_ctrl[nr] & 0x80)) { data->pwm_ctrl[nr] = data->pwm_duty[nr]; - it87_write_value(data, IT87_REG_PWM[nr], + it87_write_value(data, it87_reg_pwm(data, nr), data->pwm_ctrl[nr]); } } @@ -1717,20 +1835,14 @@ static ssize_t show_pwm_temp_map(struct device *dev, struct sensor_device_attribute *sensor_attr = to_sensor_dev_attr(attr); struct it87_data *data = it87_update_device(dev); int nr = sensor_attr->index; - u8 num_map; - int map; + unsigned int channel; if (IS_ERR(data)) return PTR_ERR(data); - num_map = data->num_temp_map; - map = data->pwm_temp_map[nr]; - if (map >= num_map) - map = 0; /* Should never happen */ - if (nr >= num_map) /* pwm channels 3..6 map to temp4..6 */ - map += num_map; + channel = pwm_temp_channel(data, nr, data->pwm_temp_map[nr]); - return sprintf(buf, "%d\n", (int)BIT(map)); + return sprintf(buf, "%d\n", (int)BIT(channel)); } static ssize_t set_pwm_temp_map(struct device *dev, @@ -1740,45 +1852,35 @@ static ssize_t set_pwm_temp_map(struct device *dev, struct sensor_device_attribute *sensor_attr = to_sensor_dev_attr(attr); struct it87_data *data = dev_get_drvdata(dev); int nr = sensor_attr->index; - u8 num_map = data->num_temp_map; long val; int err; - u8 reg; + unsigned int channel; + u8 map; - if (kstrtol(buf, 10, &val) < 0) + if (kstrtol(buf, 10, &val) < 0 || val <= 0 || !is_power_of_2(val)) return -EINVAL; - if (nr >= num_map) - val -= num_map; - - switch (val) { - case BIT(0): - reg = 0x00; - break; - case BIT(1): - reg = 0x01; - break; - case BIT(2): - reg = 0x02; - break; - default: + channel = __ffs(val); + if (pwm_temp_map_from_channel(data, nr, channel, &map)) return -EINVAL; - } err = it87_lock(data); if (err) return err; it87_update_pwm_ctrl(data, nr); - data->pwm_temp_map[nr] = reg; + data->pwm_temp_map[nr] = map; /* - * If we are in automatic mode, write the temp mapping immediately; - * otherwise, just store it for later use. + * Newer controllers keep the duty cycle in a separate register, so + * their temperature mapping can be updated in any mode. On older + * controllers, defer the update until automatic mode is enabled. */ - if (data->pwm_ctrl[nr] & 0x80) { - data->pwm_ctrl[nr] = (data->pwm_ctrl[nr] & 0xfc) | - data->pwm_temp_map[nr]; - it87_write_value(data, IT87_REG_PWM[nr], data->pwm_ctrl[nr]); + if (has_newer_autopwm(data) || (data->pwm_ctrl[nr] & 0x80)) { + data->pwm_ctrl[nr] = pwm_temp_map_set(data, + data->pwm_ctrl[nr], + data->pwm_temp_map[nr]); + it87_write_value(data, it87_reg_pwm(data, nr), + data->pwm_ctrl[nr]); } it87_unlock(data); return count; @@ -3377,7 +3479,10 @@ static void it87_init_device(struct platform_device *pdev) * manual duty cycle. */ for (i = 0; i < NUM_AUTO_PWM; i++) { - data->pwm_temp_map[i] = i; + if (uses_global_temp_map(data)) + data->pwm_temp_map[i] = 0; + else + data->pwm_temp_map[i] = i % IT87_PWM_OLD_NUM_TEMP; data->pwm_duty[i] = 0x7f; /* Full speed */ data->auto_pwm[i][3] = 0x7f; /* Full speed, hard-coded */ } @@ -3513,9 +3618,17 @@ static int it87_probe(struct platform_device *pdev) data->ec_special_config = sio_data->ec_special_config; chip = &it87_devices[sio_data->type]; data->features = chip->features; + data->reg_pwm = chip->reg_pwm; data->peci_mask = chip->peci_mask; data->old_peci_mask = chip->old_peci_mask; data->num_temp_map = chip->num_temp_map; + if (has_new_tempmap(data)) { + data->pwm_temp_map_mask = 0x07; + data->pwm_temp_map_shift = 3; + } else { + data->pwm_temp_map_mask = 0x03; + data->pwm_temp_map_shift = 0; + } /* * IT8705F Datasheet 0.4.1, 3h == Version G. * IT8712F Datasheet 0.9.1, section 8.3.5 indicates 8h == Version J. From 2548df6c6008d7b91b1c21729f1001815e52f9bb Mon Sep 17 00:00:00 2001 From: "benoit.masson" Date: Sun, 30 Aug 2026 00:01:49 +0200 Subject: [PATCH 503/857] hwmon: it87: add IT8613E support Teach the Super I/O probe path to recognize IT8613E and add its hardware monitoring configuration. Add feature flags, 11 mV ADC scaling, the IT8665-style PWM register map, six PWM temperature mapping sources, and GPIO pin-mux checks. Only three temperature inputs are currently known, so retain the existing three temperature limit and offset resources. Document the chip in the hwmon guide. Signed-off-by: benoit.masson Signed-off-by: Jerome Tollet Link: https://patch.msgid.link/5afd336442307450f77467b2a749d405970a2099.1788040385.git.jerome.tollet@gmail.com Signed-off-by: Guenter Roeck --- Documentation/hwmon/it87.rst | 8 +++++ drivers/hwmon/it87.c | 63 ++++++++++++++++++++++++++++++++++-- 2 files changed, 69 insertions(+), 2 deletions(-) diff --git a/Documentation/hwmon/it87.rst b/Documentation/hwmon/it87.rst index fc1c90b023ae6e..c33ba8a0749a8a 100644 --- a/Documentation/hwmon/it87.rst +++ b/Documentation/hwmon/it87.rst @@ -11,6 +11,14 @@ Supported chips: Datasheet: Not publicly available + * IT8613E + + Prefix: 'it8613' + + Addresses scanned: from Super I/O config space (8 I/O ports) + + Datasheet: Not publicly available + * IT8620E Prefix: 'it8620' diff --git a/drivers/hwmon/it87.c b/drivers/hwmon/it87.c index 232cc858625872..2757d428a23dba 100644 --- a/drivers/hwmon/it87.c +++ b/drivers/hwmon/it87.c @@ -36,6 +36,7 @@ * IT8790E Super I/O chip w/LPC interface * IT8792E Super I/O chip w/LPC interface * IT87952E Super I/O chip w/LPC interface + * IT8613E Super I/O chip w/LPC interface * Sis950 A clone of the IT8705F * * Copyright (C) 2001 Chris Gauthron @@ -65,7 +66,7 @@ enum chips { it87, it8712, it8716, it8718, it8720, it8721, it8728, it8732, it8771, it8772, it8781, it8782, it8783, it8786, it8790, - it8792, it8603, it8620, it8622, it8628, it8689, it87952 }; + it8792, it8603, it8613, it8620, it8622, it8628, it8689, it87952 }; static struct platform_device *it87_pdev[2]; @@ -159,6 +160,7 @@ static inline void superio_exit(int ioreg, bool noexit) #define IT8786E_DEVID 0x8786 #define IT8790E_DEVID 0x8790 #define IT8603E_DEVID 0x8603 +#define IT8613E_DEVID 0x8613 #define IT8620E_DEVID 0x8620 #define IT8622E_DEVID 0x8622 #define IT8623E_DEVID 0x8623 @@ -333,6 +335,7 @@ struct it87_devices { #define FEAT_FOUR_TEMP BIT(22) #define FEAT_FANCTL_ONOFF BIT(23) /* chip has FAN_CTL ON/OFF */ #define FEAT_NEW_TEMPMAP BIT(24) /* PWM uses extended temp map */ +#define FEAT_11MV_ADC BIT(25) static const struct it87_devices it87_devices[] = { [it87] = { @@ -515,6 +518,18 @@ static const struct it87_devices it87_devices[] = { .num_temp_map = 4, .peci_mask = 0x07, }, + [it8613] = { + .name = "it8613", + .model = "IT8613E", + /* Only three temperature inputs are currently known. */ + .features = FEAT_NEWER_AUTOPWM | FEAT_11MV_ADC | FEAT_16BIT_FANS + | FEAT_TEMP_OFFSET | FEAT_TEMP_PECI | FEAT_FIVE_FANS + | FEAT_FIVE_PWM | FEAT_IN7_INTERNAL | FEAT_PWM_FREQ2 + | FEAT_AVCC3 | FEAT_NEW_TEMPMAP, + .reg_pwm = IT87_REG_PWM_8665, + .num_temp_map = 6, + .peci_mask = 0x07, + }, [it8620] = { .name = "it8620", .model = "IT8620E", @@ -577,6 +592,7 @@ static const struct it87_devices it87_devices[] = { #define has_16bit_fans(data) ((data)->features & FEAT_16BIT_FANS) #define has_12mv_adc(data) ((data)->features & FEAT_12MV_ADC) #define has_10_9mv_adc(data) ((data)->features & FEAT_10_9MV_ADC) +#define has_11mv_adc(data) ((data)->features & FEAT_11MV_ADC) #define has_newer_autopwm(data) ((data)->features & FEAT_NEWER_AUTOPWM) #define has_old_autopwm(data) ((data)->features & FEAT_OLD_AUTOPWM) #define has_temp_offset(data) ((data)->features & FEAT_TEMP_OFFSET) @@ -607,7 +623,8 @@ static const struct it87_devices it87_devices[] = { #define has_vin3_5v(data) ((data)->features & FEAT_VIN3_5V) #define has_noconf(data) ((data)->features & FEAT_NOCONF) #define has_scaling(data) ((data)->features & (FEAT_12MV_ADC | \ - FEAT_10_9MV_ADC)) + FEAT_10_9MV_ADC | \ + FEAT_11MV_ADC)) #define has_fanctl_onoff(data) ((data)->features & FEAT_FANCTL_ONOFF) #define has_new_tempmap(data) ((data)->features & FEAT_NEW_TEMPMAP) @@ -712,6 +729,8 @@ static int adc_lsb(const struct it87_data *data, int nr) lsb = 120; else if (has_10_9mv_adc(data)) lsb = 109; + else if (has_11mv_adc(data)) + lsb = 110; else lsb = 160; if (data->in_scaled & BIT(nr)) @@ -2919,6 +2938,9 @@ static int __init it87_find(int sioaddr, unsigned short *address, case IT8623E_DEVID: sio_data->type = it8603; break; + case IT8613E_DEVID: + sio_data->type = it8613; + break; case IT8620E_DEVID: sio_data->type = it8620; break; @@ -3096,6 +3118,43 @@ static int __init it87_find(int sioaddr, unsigned short *address, sio_data->skip_in |= BIT(5); /* No VIN5 */ sio_data->skip_in |= BIT(6); /* No VIN6 */ + sio_data->beep_pin = superio_inb(sioaddr, + IT87_SIO_BEEP_PIN_REG) & 0x3f; + } else if (sio_data->type == it8613) { + int reg27, reg29, reg2a; + + superio_select(sioaddr, GPIO); + + /* Check for pwm3, fan3, pwm5, fan5 */ + reg27 = superio_inb(sioaddr, IT87_SIO_GPIO3_REG); + if (!(reg27 & BIT(1))) + sio_data->skip_fan |= BIT(4); + if (reg27 & BIT(3)) + sio_data->skip_pwm |= BIT(4); + if (reg27 & BIT(6)) + sio_data->skip_pwm |= BIT(2); + if (reg27 & BIT(7)) + sio_data->skip_fan |= BIT(2); + + /* Check for pwm2, fan2 */ + reg29 = superio_inb(sioaddr, IT87_SIO_GPIO5_REG); + if (reg29 & BIT(1)) + sio_data->skip_pwm |= BIT(1); + if (reg29 & BIT(2)) + sio_data->skip_fan |= BIT(1); + + /* Check for pwm4, fan4 */ + reg2a = superio_inb(sioaddr, IT87_SIO_PINX1_REG); + if (!(reg2a & BIT(0)) || (reg29 & BIT(7))) { + sio_data->skip_fan |= BIT(3); + sio_data->skip_pwm |= BIT(3); + } + + sio_data->skip_pwm |= BIT(0); /* No pwm1 */ + sio_data->skip_fan |= BIT(0); /* No fan1 */ + sio_data->skip_in |= BIT(3); /* No VIN3 */ + sio_data->skip_in |= BIT(6); /* No VIN6 */ + sio_data->beep_pin = superio_inb(sioaddr, IT87_SIO_BEEP_PIN_REG) & 0x3f; } else if (sio_data->type == it8620 || sio_data->type == it8628) { From abb0f9fe447b8cf3a45f69c2e9d69697ef004973 Mon Sep 17 00:00:00 2001 From: Flaviu Nistor Date: Wed, 26 Aug 2026 21:47:48 +0300 Subject: [PATCH 504/857] dt-bindings: hwmon: national,lm90: Fix channel constraints for temperature offset Limit temperature-offset-millicelsius to remote channels only, since channel 0 is local and this property does not apply to it. Channel 2 is only valid on devices with two remote sensors, so reject channel 2 for compatibles that do not support a second remote channel. Signed-off-by: Flaviu Nistor Acked-by: Conor Dooley Link: https://patch.msgid.link/20260826184750.4798-2-flaviu.nistor@gmail.com Signed-off-by: Guenter Roeck --- .../bindings/hwmon/national,lm90.yaml | 28 +++++++++++++++++-- 1 file changed, 25 insertions(+), 3 deletions(-) diff --git a/Documentation/devicetree/bindings/hwmon/national,lm90.yaml b/Documentation/devicetree/bindings/hwmon/national,lm90.yaml index 164068ba069d73..a8b2a24501b354 100644 --- a/Documentation/devicetree/bindings/hwmon/national,lm90.yaml +++ b/Documentation/devicetree/bindings/hwmon/national,lm90.yaml @@ -113,6 +113,23 @@ allOf: properties: ti,extended-range-enable: false + - if: + not: + properties: + compatible: + contains: + enum: + - adi,adt7481 + - dallas,max6695 + - dallas,max6696 + then: + patternProperties: + "^channel@[0-1]$": + properties: + reg: + enum: [0, 1] + "channel@2": false + - if: properties: compatible: @@ -134,6 +151,11 @@ allOf: "^channel@([0-2])$": properties: temperature-offset-millicelsius: false + else: + patternProperties: + "channel@0": + properties: + temperature-offset-millicelsius: false - if: properties: @@ -149,7 +171,7 @@ allOf: - onnn,nct1008 then: patternProperties: - "^channel@([0-2])$": + "^channel@([1-2])$": properties: temperature-offset-millicelsius: maximum: 127750 @@ -172,7 +194,7 @@ allOf: - winbond,w83l771 then: patternProperties: - "^channel@([0-2])$": + "^channel@([1-2])$": properties: temperature-offset-millicelsius: maximum: 127875 @@ -186,7 +208,7 @@ allOf: - ti,tmp461 then: patternProperties: - "^channel@([0-2])$": + "^channel@([1-2])$": properties: temperature-offset-millicelsius: maximum: 127937 From bbeea487f5b7f94f89550005a142e15ddb869f15 Mon Sep 17 00:00:00 2001 From: Flaviu Nistor Date: Wed, 26 Aug 2026 21:47:49 +0300 Subject: [PATCH 505/857] hwmon: (lm90) Reject channel 2 on chips with only one remote sensor Validate firmware channel definitions against chip capabilities and return -EINVAL when channel 2 is configured on devices with 1 remote channel. Signed-off-by: Flaviu Nistor Link: https://patch.msgid.link/20260826184750.4798-3-flaviu.nistor@gmail.com Signed-off-by: Guenter Roeck --- drivers/hwmon/lm90.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/drivers/hwmon/lm90.c b/drivers/hwmon/lm90.c index 1c603272538aa1..3faeb2c6ab01b2 100644 --- a/drivers/hwmon/lm90.c +++ b/drivers/hwmon/lm90.c @@ -2712,6 +2712,11 @@ static int lm90_probe_channel(struct i2c_client *client, return -EINVAL; } + if (id == 2 && !(data->flags & LM90_HAVE_TEMP3)) { + dev_err(dev, "channel %d is not supported for this chip in %pfw\n", id, child); + return -EINVAL; + } + err = fwnode_property_read_string(child, "label", &data->channel_label[id]); if (err == -ENODATA || err == -EILSEQ) { dev_err(dev, "invalid label property in %pfw\n", child); From 0f7bc24d3905cc454ebd72fe076b1696baab522c Mon Sep 17 00:00:00 2001 From: Asai Neko Date: Tue, 1 Sep 2026 00:18:24 +0800 Subject: [PATCH 506/857] hwmon: (asus-ec-sensors) add ROG STRIX X670E-A GAMING WIFI The ROG STRIX X670E-A GAMING WIFI is missing from the driver's DMI table. Consequently, the board lookup fails with -ENODEV, asus_ec_sensors does not load, and no asusec hwmon device or EC temperature readings are available. The board uses the same EC sensors, access mutex, and AMD 600-series register layout as the ROG STRIX X670E-E GAMING WIFI. Add its DMI entry using the existing X670E-E board information and document the board as supported. Before the change, there was no asusec device under /sys/class/hwmon and there were no EC readings. After the change, the driver registered four sensors with representative readings of 57-61 C for CPU, 68-71 C for CPU package, 43-44 C for motherboard, and 49-52 C for VRM. The readings correlated with nct6775 and k10temp. Repeated polling with both hwmon drivers loaded produced no EC access, bank-switch, concurrent access, or locking errors. Tested on an ASUS ROG STRIX X670E-A GAMING WIFI with BIOS 2704. Signed-off-by: Asai Neko Reviewed-by: Eugene Shalygin Link: https://patch.msgid.link/20260901-asus-x670e-a-hwmon-fix-v2-1-759406c2a61a@sne.moe Signed-off-by: Guenter Roeck --- Documentation/hwmon/asus_ec_sensors.rst | 1 + drivers/hwmon/asus-ec-sensors.c | 2 ++ 2 files changed, 3 insertions(+) diff --git a/Documentation/hwmon/asus_ec_sensors.rst b/Documentation/hwmon/asus_ec_sensors.rst index 2c68a343a2edde..ebed9f8d2adb85 100644 --- a/Documentation/hwmon/asus_ec_sensors.rst +++ b/Documentation/hwmon/asus_ec_sensors.rst @@ -45,6 +45,7 @@ Supported boards: * ROG STRIX X570-E GAMING WIFI II * ROG STRIX X570-F GAMING * ROG STRIX X570-I GAMING + * ROG STRIX X670E-A GAMING WIFI * ROG STRIX X670E-E GAMING WIFI * ROG STRIX X670E-I GAMING WIFI * ROG STRIX X870-F GAMING WIFI diff --git a/drivers/hwmon/asus-ec-sensors.c b/drivers/hwmon/asus-ec-sensors.c index e43645e884fdb6..7293c48c72b1d6 100644 --- a/drivers/hwmon/asus-ec-sensors.c +++ b/drivers/hwmon/asus-ec-sensors.c @@ -962,6 +962,8 @@ static const struct dmi_system_id dmi_table[] = { &board_info_strix_x570_f_gaming), DMI_EXACT_MATCH_ASUS_BOARD_NAME("ROG STRIX X570-I GAMING", &board_info_strix_x570_i_gaming), + DMI_EXACT_MATCH_ASUS_BOARD_NAME("ROG STRIX X670E-A GAMING WIFI", + &board_info_strix_x670e_e_gaming_wifi), DMI_EXACT_MATCH_ASUS_BOARD_NAME("ROG STRIX X670E-E GAMING WIFI", &board_info_strix_x670e_e_gaming_wifi), DMI_EXACT_MATCH_ASUS_BOARD_NAME("ROG STRIX X670E-I GAMING WIFI", From 94056769c49ab2dcb86e58967611aac484d5806e Mon Sep 17 00:00:00 2001 From: Frank Li Date: Wed, 26 Aug 2026 17:14:19 -0400 Subject: [PATCH 507/857] dt-bindings: hwmon: tmp102: move ti,tmp103 out of trivial-devices.yaml Move ti,tmp103 binding from trivial-devices.yaml to ti,tmp102.yaml. Both devices are single temperature sensors and update "#thermal-sensor-cells" property to accept values 0 and 1 (passing no argument is equivalent to passing 0 as the first argument) Fix below CHECK_DTBS warnings: arch/arm/boot/dts/nxp/imx/imx6dl-plym2m.dtb: temperature-sensor@70 (ti,tmp103): '#thermal-sensor-cells' does not match any of the regexes: '^pinctrl-[0-9]+$' Squashed with: dt-bindings: hwmon: tmp102: Fix up TMP103 bindings The TMP102 and TMP110 are backward compatible, therefore the valid compatible strings are "ti,tmp110", "ti,tmp102" and "ti,tmp102" . The TMP103 is not backward compatible with TMP102, but it does share a schema now, therefore the only valid compatible string for the TMP103 should be "ti,tmp103" . Make exactly those three compatible strings valid and everything else rejected. Signed-off-by: Frank Li Reviewed-by: Krzysztof Kozlowski Signed-off-by: Marek Vasut Acked-by: Conor Dooley Link: https://patch.msgid.link/20260901174847.97261-1-marex@nabladev.com Signed-off-by: Guenter Roeck --- Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml | 8 ++++---- Documentation/devicetree/bindings/trivial-devices.yaml | 2 -- 2 files changed, 4 insertions(+), 6 deletions(-) diff --git a/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml b/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml index f0dbe9f07b939a..477f478b9dd0fb 100644 --- a/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml +++ b/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml @@ -4,7 +4,7 @@ $id: http://devicetree.org/schemas/hwmon/ti,tmp102.yaml# $schema: http://devicetree.org/meta-schemas/core.yaml# -title: TMP102 and TMP110 temperature sensor +title: TMP102/TMP103/TMP110 temperature sensor maintainers: - Krzysztof Kozlowski @@ -15,8 +15,8 @@ properties: - items: - const: ti,tmp110 - const: ti,tmp102 - - enum: - - ti,tmp102 + - const: ti,tmp102 + - const: ti,tmp103 interrupts: maxItems: 1 @@ -29,7 +29,7 @@ properties: A descriptive name for this channel, like "ambient" or "psu". "#thermal-sensor-cells": - const: 1 + enum: [0, 1] vcc-supply: description: Power supply for the sensor diff --git a/Documentation/devicetree/bindings/trivial-devices.yaml b/Documentation/devicetree/bindings/trivial-devices.yaml index a82460e2416fd7..6229a65163027b 100644 --- a/Documentation/devicetree/bindings/trivial-devices.yaml +++ b/Documentation/devicetree/bindings/trivial-devices.yaml @@ -506,8 +506,6 @@ properties: - ti,lm74 # Temperature sensor with integrated fan control - ti,lm96000 - # Low Power Digital Temperature Sensor with SMBUS/Two Wire Serial Interface - - ti,tmp103 # Thermometer with SPI interface - ti,tmp121 - ti,tmp122 From 5d5a03839258fc65e8281c2ac0dc61a57122352a Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Tue, 1 Sep 2026 19:47:55 +0200 Subject: [PATCH 508/857] dt-bindings: hwmon: tmp102: Document TMP113 The TMP113 temperature sensor part is register compatible with TMP102, document it using a fallback compatible. Unlike TMP102 and TMP110, the TMP113 does have additional unique ID registers, it is up to the driver to handle those. Signed-off-by: Marek Vasut Acked-by: Conor Dooley Link: https://patch.msgid.link/20260901174847.97261-2-marex@nabladev.com Signed-off-by: Guenter Roeck --- Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml b/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml index 477f478b9dd0fb..c0b747a9ce0de4 100644 --- a/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml +++ b/Documentation/devicetree/bindings/hwmon/ti,tmp102.yaml @@ -4,7 +4,7 @@ $id: http://devicetree.org/schemas/hwmon/ti,tmp102.yaml# $schema: http://devicetree.org/meta-schemas/core.yaml# -title: TMP102/TMP103/TMP110 temperature sensor +title: TMP102/TMP103/TMP110/TMP113 temperature sensor maintainers: - Krzysztof Kozlowski @@ -13,7 +13,9 @@ properties: compatible: oneOf: - items: - - const: ti,tmp110 + - enum: + - ti,tmp110 + - ti,tmp113 - const: ti,tmp102 - const: ti,tmp102 - const: ti,tmp103 From 82fee53b6a700f7ef429c0c16ce4978927ef441d Mon Sep 17 00:00:00 2001 From: Armin Wolf Date: Tue, 1 Sep 2026 20:18:48 +0200 Subject: [PATCH 509/857] hwmon: (dell-smm) Add Dell OptiPlex 7090 to fan control whitelist MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A user reported that the Dell OptiPlex 7090 needs to be whitelisted for the special SMM calls necessary for globally enabling/disabling BIOS fan control. Closes: https://github.com/Wer-Wolf/i8kutils/issues/18 Signed-off-by: Armin Wolf Link: https://patch.msgid.link/20260901181849.241776-1-W_Armin@gmx.de Acked-by: Pali Rohár Signed-off-by: Guenter Roeck --- drivers/hwmon/dell-smm-hwmon.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/drivers/hwmon/dell-smm-hwmon.c b/drivers/hwmon/dell-smm-hwmon.c index 20920d4bd81e27..a3aa398a9f8f52 100644 --- a/drivers/hwmon/dell-smm-hwmon.c +++ b/drivers/hwmon/dell-smm-hwmon.c @@ -1631,6 +1631,14 @@ static const struct dmi_system_id i8k_whitelist_fan_control[] __initconst = { }, .driver_data = (void *)&i8k_fan_control_data[I8K_FAN_34A3_35A3], }, + { + .ident = "Dell Optiplex 7090", + .matches = { + DMI_MATCH(DMI_SYS_VENDOR, "Dell Inc."), + DMI_EXACT_MATCH(DMI_PRODUCT_NAME, "OptiPlex 7090"), + }, + .driver_data = (void *)&i8k_fan_control_data[I8K_FAN_30A3_31A3], + }, { .ident = "Dell XPS 9315", .matches = { From f521140d699436b3c6cf29716320a8ef026d6db6 Mon Sep 17 00:00:00 2001 From: Armin Wolf Date: Tue, 1 Sep 2026 20:18:49 +0200 Subject: [PATCH 510/857] hwmon: (dell-smm) Add Latitude 5420 to fan control whitelist MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A user reported that the Dell Latitude 5420 needs to be whitelisted for the special SMM calls necessary for globally enabling/disabling BIOS fan control. Reported-by: pp12313124124@gmail.com Closes: https://bugzilla.kernel.org/show_bug.cgi?id=221935 Signed-off-by: Armin Wolf Link: https://patch.msgid.link/20260901181849.241776-2-W_Armin@gmx.de Acked-by: Pali Rohár Signed-off-by: Guenter Roeck --- drivers/hwmon/dell-smm-hwmon.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/drivers/hwmon/dell-smm-hwmon.c b/drivers/hwmon/dell-smm-hwmon.c index a3aa398a9f8f52..7ed94a3db62128 100644 --- a/drivers/hwmon/dell-smm-hwmon.c +++ b/drivers/hwmon/dell-smm-hwmon.c @@ -1543,6 +1543,14 @@ static const struct dmi_system_id i8k_whitelist_fan_control[] __initconst = { }, .driver_data = (void *)&i8k_fan_control_data[I8K_FAN_34A3_35A3], }, + { + .ident = "Dell Latitude 5420", + .matches = { + DMI_MATCH(DMI_SYS_VENDOR, "Dell Inc."), + DMI_EXACT_MATCH(DMI_PRODUCT_NAME, "Latitude 5420"), + }, + .driver_data = (void *)&i8k_fan_control_data[I8K_FAN_30A3_31A3], + }, { .ident = "Dell Latitude 5480", .matches = { From feb7cbb08666591fd5ae7a82809ac473b8ebda6c Mon Sep 17 00:00:00 2001 From: Armin Wolf Date: Tue, 1 Sep 2026 22:01:41 +0200 Subject: [PATCH 511/857] hwmon: (spd5118) Select page 0 unconditionally during probe Some Intel i2c controllers can be configured by the BIOS to reject writes to the SPD device. This often causes problems when the register page needs to be changed, usually during resume. Avoid probing on affected devices by unconditionally selecting page 0 by writing the SPD5118_REG_I2C_LEGACY_MODE register during probe. This will fail on affected controllers and thus prevent the driver from probing. Signed-off-by: Armin Wolf Link: https://patch.msgid.link/20260901200142.495319-1-W_Armin@gmx.de Signed-off-by: Guenter Roeck --- drivers/hwmon/spd5118.c | 65 +++++++++++++++++++++-------------------- 1 file changed, 34 insertions(+), 31 deletions(-) diff --git a/drivers/hwmon/spd5118.c b/drivers/hwmon/spd5118.c index 9724cf70b61d4d..f79b46085eccb5 100644 --- a/drivers/hwmon/spd5118.c +++ b/drivers/hwmon/spd5118.c @@ -637,44 +637,47 @@ static int spd5118_i2c_init(struct i2c_client *client) I2C_FUNC_SMBUS_WORD_DATA)) return -ENODEV; + /* Early check to avoid obviously unsupported I2C devices */ regval = i2c_smbus_read_word_swapped(client, SPD5118_REG_TYPE); - if (regval < 0 || (regval && regval != 0x5118)) + if (regval < 0) + return regval; + + /* + * Some SPD5118 devices report 0x0 when page 0 is not selected, + * so we only fail here if the register value is not 0x0 or 0x5118. + */ + if (regval && regval != 0x5118) return -ENODEV; /* - * If the device type registers return 0, it is possible that the chip - * has a non-zero page selected and takes the specification literally, + * We must select page 0 to ensure that we can reliably read the volatile + * registers on chips that take the specification literally, * i.e. disables access to volatile registers besides the page register * if the page is not 0. The Renesas/ITD SPD5118 Hub Controller is known - * to show this behavior. Try to identify such chips. + * to show this behavior. + * + * We must also perform an unconditional register write to detect if + * the i2c controller blocks write accesses to the SPD device. Some Intel + * controllers might be configured by the BIOS to do this. */ - if (!regval) { - /* Vendor ID registers must also be 0 */ - regval = i2c_smbus_read_word_data(client, SPD5118_REG_VENDOR); - if (regval) - return -ENODEV; - - /* The selected page in MR11 must not be 0 */ - mode = i2c_smbus_read_byte_data(client, SPD5118_REG_I2C_LEGACY_MODE); - if (mode < 0 || (mode & ~SPD5118_LEGACY_MODE_MASK) || - !(mode & SPD5118_LEGACY_PAGE_MASK)) - return -ENODEV; - - err = i2c_smbus_write_byte_data(client, SPD5118_REG_I2C_LEGACY_MODE, - mode & SPD5118_LEGACY_MODE_ADDR); - if (err) - return -ENODEV; - - /* - * If the device type registers are still bad after selecting - * page 0, this is not a SPD5118 device. Restore original - * legacy mode register value and abort. - */ - regval = i2c_smbus_read_word_swapped(client, SPD5118_REG_TYPE); - if (regval != 0x5118) { - i2c_smbus_write_byte_data(client, SPD5118_REG_I2C_LEGACY_MODE, mode); - return -ENODEV; - } + mode = i2c_smbus_read_byte_data(client, SPD5118_REG_I2C_LEGACY_MODE); + if (mode < 0) + return mode; + + err = i2c_smbus_write_byte_data(client, SPD5118_REG_I2C_LEGACY_MODE, + mode & ~SPD5118_LEGACY_PAGE_MASK); + if (err < 0) + return err; + + /* We only need to access SPD5118_REG_TYPE again if regval was 0x0 */ + if (regval == 0x5118) + return 0; + + regval = i2c_smbus_read_word_swapped(client, SPD5118_REG_TYPE); + if (regval != 0x5118) { + /* Restore original register content */ + i2c_smbus_write_byte_data(client, SPD5118_REG_I2C_LEGACY_MODE, mode); + return -ENODEV; } /* We are reasonably sure that this is really a SPD5118 hub controller */ From ac3a727e61d97da79789c2de6712acb0a82325d6 Mon Sep 17 00:00:00 2001 From: Armin Wolf Date: Tue, 1 Sep 2026 22:01:42 +0200 Subject: [PATCH 512/857] hwmon: (spd5118) Avoid probing when 16-bit addressing is enabled Support for 16-bit addressing was removed when support for i3c was added to the driver. Switching between 8-bit and 16-bit addressing might confuse the system firmware, so we are forced to bail out if 16-bit addressing was configured during boot. Signed-off-by: Armin Wolf Link: https://patch.msgid.link/20260901200142.495319-2-W_Armin@gmx.de Signed-off-by: Guenter Roeck --- drivers/hwmon/spd5118.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/drivers/hwmon/spd5118.c b/drivers/hwmon/spd5118.c index f79b46085eccb5..52f35448934116 100644 --- a/drivers/hwmon/spd5118.c +++ b/drivers/hwmon/spd5118.c @@ -16,6 +16,7 @@ #include #include +#include #include #include #include @@ -664,6 +665,13 @@ static int spd5118_i2c_init(struct i2c_client *client) if (mode < 0) return mode; + /* 16-bit addressing is not supported */ + if (mode & SPD5118_LEGACY_MODE_ADDR) { + dev_notice(&client->dev, + "Unable to access device due to 16-bit addressing being enabled\n"); + return -ENODEV; + } + err = i2c_smbus_write_byte_data(client, SPD5118_REG_I2C_LEGACY_MODE, mode & ~SPD5118_LEGACY_PAGE_MASK); if (err < 0) From 5f9acc1c21dfb60e4d2f31cbbb431969a044f306 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sebasti=C3=A1n=20Peyrott?= Date: Tue, 1 Sep 2026 21:05:09 -0300 Subject: [PATCH 513/857] hwmon: Add Minisforum UM780 XTX EC monitoring and fan control MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add a DMI-gated hwmon driver for the embedded controller used by the Minisforum UM780 XTX with board revision 1.1 and BIOS 1.06. Expose the CPU and system fan control temperatures and tachometers. The tachometer protocol returns one byte per OEM command, so serialize transactions and use high-low-high sampling to reject torn values. Allow selecting either complete OEM CPU fan profile through pwm1_enable and changing the two validated system fan transition temperatures through standard automatic-curve attributes. Cache coherent settings and restore them after the firmware reloads defaults following resume. Signed-off-by: Sebastián Peyrott Link: https://patch.msgid.link/20260902000509.191115-1-speyrott@gmail.com Signed-off-by: Guenter Roeck --- Documentation/hwmon/index.rst | 1 + Documentation/hwmon/minisforum-um780xtx.rst | 68 +++ MAINTAINERS | 7 + drivers/hwmon/Kconfig | 16 + drivers/hwmon/Makefile | 1 + drivers/hwmon/minisforum-um780xtx.c | 501 ++++++++++++++++++++ 6 files changed, 594 insertions(+) create mode 100644 Documentation/hwmon/minisforum-um780xtx.rst create mode 100644 drivers/hwmon/minisforum-um780xtx.c diff --git a/Documentation/hwmon/index.rst b/Documentation/hwmon/index.rst index 06a7992ce5daa0..f07977a20271d3 100644 --- a/Documentation/hwmon/index.rst +++ b/Documentation/hwmon/index.rst @@ -187,6 +187,7 @@ Hardware Monitoring Kernel Drivers mcp3021 mcp9982 menf21bmc + minisforum-um780xtx mlxreg-fan mp2856 mp2869 diff --git a/Documentation/hwmon/minisforum-um780xtx.rst b/Documentation/hwmon/minisforum-um780xtx.rst new file mode 100644 index 00000000000000..8237d525a9c7d3 --- /dev/null +++ b/Documentation/hwmon/minisforum-um780xtx.rst @@ -0,0 +1,68 @@ +.. SPDX-License-Identifier: GPL-2.0-only + +==================================== +Kernel driver minisforum-um780xtx +==================================== + +Supported systems: + + * Minisforum UM780 XTX + + * DMI product name: ``Venus series`` + * Mainboard: ``F7BSD``, revision ``1.1`` + * BIOS version: ``1.06`` + +Author: Sebastián Peyrott + +Description +----------- + +This driver exposes hardware monitoring and fan-control data cached by the +IT5571E embedded controller. It uses the ACPI EC transport and only binds to +the exact system and firmware identity listed above. + +The two temperature channels are the values used by the EC's fan-control +loops. Their physical sensor placement is not known. The CPU channel is the +filtered AMD SB-TSI temperature, while the system channel is an external +thermistor input. + +The fan tachometers are read through OEM EC commands. Since each byte is +returned by a separate command, the driver uses a high-low-high sequence and +retries if the high byte changes. + +Sysfs entries +------------- + +========================== ============================================== +``temp1_input`` CPU-fan control temperature +``temp2_input`` System-fan control temperature +``fan1_input`` CPU fan speed in RPM +``fan2_input`` System fan speed in RPM +``pwm1_enable`` CPU profile: 2 is OEM B1, 3 is OEM B2 +``pwm2_auto_point1_temp`` Off-to-low system-fan transition temperature +``pwm2_auto_point2_temp`` Low-to-high system-fan transition temperature +========================== ============================================== + +CPU fan profiles +---------------- + +Values 2 and 3 of ``pwm1_enable`` select the two complete automatic profiles +implemented by the firmware. Writing either value reloads the whole CPU fan +curve, including firmware state which is not visible through the ACPI EC +window. Other values are rejected. + +System fan thresholds +--------------------- + +The two writable system-fan temperatures are rounded to whole degrees Celsius +and clamped to preserve strict ordering between both visible thresholds and +the firmware's internal third threshold. The third threshold does not select +a new PWM target and is therefore not exposed as another auto point. + +Resume state preservation +------------------------- + +Fan settings are held in EC RAM. On the supported firmware, initialization +during s2idle resume reloads the factory system-fan curve. The driver retains +the last coherent CPU profile and system-fan thresholds selected through its +interface and restores them synchronously at resume. diff --git a/MAINTAINERS b/MAINTAINERS index 92c16ba8c78e78..73290af7e7a546 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -18189,6 +18189,13 @@ F: include/linux/min_heap.h F: lib/min_heap.c F: lib/tests/min_heap_kunit.c +MINISFORUM UM780 XTX EC HARDWARE MONITOR DRIVER +M: Sebastián Peyrott +L: linux-hwmon@vger.kernel.org +S: Maintained +F: Documentation/hwmon/minisforum-um780xtx.rst +F: drivers/hwmon/minisforum-um780xtx.c + MIPI CCS, SMIA AND SMIA++ IMAGE SENSOR DRIVER M: Sakari Ailus L: linux-media@vger.kernel.org diff --git a/drivers/hwmon/Kconfig b/drivers/hwmon/Kconfig index b19aa16986af6e..c1d72af52e8f85 100644 --- a/drivers/hwmon/Kconfig +++ b/drivers/hwmon/Kconfig @@ -1490,6 +1490,22 @@ config SENSORS_MENF21BMC_HWMON This driver can also be built as a module. If so the module will be called menf21bmc_hwmon. +config SENSORS_MINISFORUM_UM780XTX + tristate "Minisforum UM780 XTX EC hardware monitoring" + depends on ACPI_EC && DMI && X86 + help + If you say yes here you get support for the hardware monitoring + features of the embedded controller in the Minisforum UM780 XTX. + This includes CPU and system fan speeds, their control temperatures, + two CPU fan profiles, system fan temperature thresholds, and + preservation of the selected settings across suspend and resume. + + This driver only binds to explicitly supported board and firmware + versions. + + This driver can also be built as a module. If so, the module + will be called minisforum-um780xtx. + config SENSORS_MR75203 tristate "Moortec Semiconductor MR75203 PVT Controller" select REGMAP_MMIO diff --git a/drivers/hwmon/Makefile b/drivers/hwmon/Makefile index 646813b7aaa852..0243cb62c058a8 100644 --- a/drivers/hwmon/Makefile +++ b/drivers/hwmon/Makefile @@ -181,6 +181,7 @@ obj-$(CONFIG_SENSORS_TC654) += tc654.o obj-$(CONFIG_SENSORS_TPS23861) += tps23861.o obj-$(CONFIG_SENSORS_MLXREG_FAN) += mlxreg-fan.o obj-$(CONFIG_SENSORS_MENF21BMC_HWMON) += menf21bmc_hwmon.o +obj-$(CONFIG_SENSORS_MINISFORUM_UM780XTX) += minisforum-um780xtx.o obj-$(CONFIG_SENSORS_MR75203) += mr75203.o obj-$(CONFIG_SENSORS_NCT6683) += nct6683.o obj-$(CONFIG_SENSORS_NCT6694) += nct6694-hwmon.o diff --git a/drivers/hwmon/minisforum-um780xtx.c b/drivers/hwmon/minisforum-um780xtx.c new file mode 100644 index 00000000000000..f5d6e18cb651f5 --- /dev/null +++ b/drivers/hwmon/minisforum-um780xtx.c @@ -0,0 +1,501 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Minisforum UM780 XTX (F7BSD) embedded-controller hwmon driver. + * + * Copyright (C) 2026 Sebastián Peyrott + */ + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#define UM780XTX_EC_TEMP_SYS 0x05 +#define UM780XTX_EC_TEMP_CPU 0x09 +#define UM780XTX_EC_CPU_PROFILE 0x2f +#define UM780XTX_EC_SYS_POINT1 0x31 +#define UM780XTX_EC_SYS_POINT2 0x34 +#define UM780XTX_EC_SYS_POINT3 0x37 + +#define UM780XTX_EC_PROFILE_B1 0xb1 +#define UM780XTX_EC_PROFILE_B2 0xb2 + +#define UM780XTX_EC_FAN1_LO 0xb6 +#define UM780XTX_EC_FAN1_HI 0xb7 +#define UM780XTX_EC_FAN2_LO 0xb9 +#define UM780XTX_EC_FAN2_HI 0xba + +#define UM780XTX_EC_MAX_RPM 9000 +#define UM780XTX_EC_RPM_RETRIES 3 + +static struct platform_device *um780xtx_pdev; + +struct um780xtx_data { + struct device *hwmon_dev; + u8 saved_profile; + u8 saved_sys_point1; + u8 saved_sys_point2; + u8 sys_point3; +}; + +static const struct dmi_system_id um780xtx_dmi_table[] = { + { + .matches = { + DMI_EXACT_MATCH(DMI_SYS_VENDOR, + "Micro Computer (HK) Tech Limited"), + DMI_EXACT_MATCH(DMI_PRODUCT_NAME, "Venus series"), + DMI_EXACT_MATCH(DMI_BOARD_VENDOR, + "Shenzhen Meigao Electronic Equipment Co.,Ltd"), + DMI_EXACT_MATCH(DMI_BOARD_NAME, "F7BSD"), + }, + }, + { } +}; +MODULE_DEVICE_TABLE(dmi, um780xtx_dmi_table); + +static bool um780xtx_firmware_match(void) +{ + const char *board_version = dmi_get_system_info(DMI_BOARD_VERSION); + const char *bios_version = dmi_get_system_info(DMI_BIOS_VERSION); + + return board_version && bios_version && + !strcmp(board_version, "1.1") && !strcmp(bios_version, "1.06"); +} + +static int um780xtx_oem_read(u8 command, u8 *value) +{ + return ec_transaction(command, NULL, 0, value, 1); +} + +static int um780xtx_read_rpm(u8 command_hi, u8 command_lo, long *rpm) +{ + u8 hi_before; + u8 hi_after; + u8 lo; + unsigned int value; + int attempt; + int ret; + + for (attempt = 0; attempt < UM780XTX_EC_RPM_RETRIES; attempt++) { + ret = um780xtx_oem_read(command_hi, &hi_before); + if (ret) + return ret; + ret = um780xtx_oem_read(command_lo, &lo); + if (ret) + return ret; + ret = um780xtx_oem_read(command_hi, &hi_after); + if (ret) + return ret; + if (hi_before != hi_after) + continue; + + value = (hi_after << 8) | lo; + if (value > UM780XTX_EC_MAX_RPM) + continue; + + *rpm = value; + return 0; + } + + return -EIO; +} + +static int um780xtx_read_profile(long *mode) +{ + u8 profile; + int ret; + + ret = ec_read(UM780XTX_EC_CPU_PROFILE, &profile); + if (ret) + return ret; + if (profile == UM780XTX_EC_PROFILE_B1) { + *mode = 2; + return 0; + } + if (profile == UM780XTX_EC_PROFILE_B2) { + *mode = 3; + return 0; + } + + return -ENODATA; +} + +static int um780xtx_write_profile(struct um780xtx_data *data, long mode) +{ + u8 expected; + u8 profile; + int ret; + + if (mode != 2 && mode != 3) + return -EINVAL; + expected = mode == 2 ? UM780XTX_EC_PROFILE_B1 : UM780XTX_EC_PROFILE_B2; + + ret = ec_transaction(expected, NULL, 0, NULL, 0); + if (ret) + return ret; + ret = ec_read(UM780XTX_EC_CPU_PROFILE, &profile); + if (ret) + return ret; + if (profile != expected) + return -EIO; + + data->saved_profile = profile; + return 0; +} + +static const u8 um780xtx_sys_point_offsets[] = { + UM780XTX_EC_SYS_POINT1, + UM780XTX_EC_SYS_POINT2, +}; + +static ssize_t um780xtx_sys_point_temp_show(struct device *dev, + struct device_attribute *attr, + char *buf) +{ + struct sensor_device_attribute *sattr = to_sensor_dev_attr(attr); + u8 value; + int ret; + + guard(hwmon_lock)(dev); + ret = ec_read(um780xtx_sys_point_offsets[sattr->index], &value); + if (ret) + return ret; + + return sysfs_emit(buf, "%u\n", value * 1000); +} + +static ssize_t um780xtx_sys_point_temp_store(struct device *dev, + struct device_attribute *attr, + const char *buf, size_t count) +{ + struct um780xtx_data *data = dev_get_drvdata(dev); + struct sensor_device_attribute *sattr = to_sensor_dev_attr(attr); + u8 points[2]; + u8 readback; + long value; + int ret; + + ret = kstrtol(buf, 10, &value); + if (ret) + return ret; + + guard(hwmon_lock)(dev); + /* Validate and clamp against the live peer threshold. */ + ret = ec_read(UM780XTX_EC_SYS_POINT1, &points[0]); + if (ret) + return ret; + ret = ec_read(UM780XTX_EC_SYS_POINT2, &points[1]); + if (ret) + return ret; + if (points[0] >= points[1] || points[1] >= data->sys_point3) + return -EIO; + + value = DIV_ROUND_CLOSEST(value, 1000); + if (!sattr->index) + value = clamp_val(value, 0, points[1] - 1); + else + value = clamp_val(value, points[0] + 1, + data->sys_point3 - 1); + points[sattr->index] = value; + + ret = ec_write(um780xtx_sys_point_offsets[sattr->index], + points[sattr->index]); + if (ret) + return ret; + ret = ec_read(um780xtx_sys_point_offsets[sattr->index], &readback); + if (ret) + return ret; + if (readback != points[sattr->index]) + return -EIO; + + data->saved_sys_point1 = points[0]; + data->saved_sys_point2 = points[1]; + return count; +} + +static SENSOR_DEVICE_ATTR_RW(pwm2_auto_point1_temp, + um780xtx_sys_point_temp, 0); +static SENSOR_DEVICE_ATTR_RW(pwm2_auto_point2_temp, + um780xtx_sys_point_temp, 1); + +static struct attribute *um780xtx_extra_attrs[] = { + &sensor_dev_attr_pwm2_auto_point1_temp.dev_attr.attr, + &sensor_dev_attr_pwm2_auto_point2_temp.dev_attr.attr, + NULL +}; + +static const struct attribute_group um780xtx_extra_group = { + .attrs = um780xtx_extra_attrs, +}; + +static const struct attribute_group *um780xtx_extra_groups[] = { + &um780xtx_extra_group, + NULL +}; + +static umode_t um780xtx_is_visible(const void *data, + enum hwmon_sensor_types type, + u32 attr, int channel) +{ + if (type == hwmon_temp || type == hwmon_fan) + return 0444; + if (type == hwmon_pwm && attr == hwmon_pwm_enable) + return 0644; + return 0; +} + +static int um780xtx_read(struct device *dev, enum hwmon_sensor_types type, + u32 attr, int channel, long *value) +{ + u8 raw; + int ret; + + if (type == hwmon_temp) { + ret = ec_read(channel ? UM780XTX_EC_TEMP_SYS : + UM780XTX_EC_TEMP_CPU, &raw); + if (ret) + return ret; + *value = raw * 1000L; + return 0; + } + if (type == hwmon_fan) { + if (!channel) + ret = um780xtx_read_rpm(UM780XTX_EC_FAN1_HI, + UM780XTX_EC_FAN1_LO, value); + else + ret = um780xtx_read_rpm(UM780XTX_EC_FAN2_HI, + UM780XTX_EC_FAN2_LO, value); + return ret; + } + if (type == hwmon_pwm) + return um780xtx_read_profile(value); + + return -EOPNOTSUPP; +} + +static int um780xtx_write(struct device *dev, enum hwmon_sensor_types type, + u32 attr, int channel, long value) +{ + struct um780xtx_data *data = dev_get_drvdata(dev); + + if (type == hwmon_pwm && attr == hwmon_pwm_enable) + return um780xtx_write_profile(data, value); + + return -EOPNOTSUPP; +} + +static int um780xtx_read_string(struct device *dev, + enum hwmon_sensor_types type, + u32 attr, int channel, const char **str) +{ + static const char * const labels[] = { "CPU", "SYS" }; + + if (type == hwmon_temp && attr == hwmon_temp_label) + *str = labels[channel]; + else if (type == hwmon_fan && attr == hwmon_fan_label) + *str = labels[channel]; + else + return -EOPNOTSUPP; + return 0; +} + +static const struct hwmon_ops um780xtx_hwmon_ops = { + .is_visible = um780xtx_is_visible, + .read = um780xtx_read, + .write = um780xtx_write, + .read_string = um780xtx_read_string, +}; + +static const struct hwmon_channel_info * const um780xtx_info[] = { + HWMON_CHANNEL_INFO(temp, HWMON_T_INPUT | HWMON_T_LABEL, + HWMON_T_INPUT | HWMON_T_LABEL), + HWMON_CHANNEL_INFO(fan, HWMON_F_INPUT | HWMON_F_LABEL, + HWMON_F_INPUT | HWMON_F_LABEL), + HWMON_CHANNEL_INFO(pwm, HWMON_PWM_ENABLE), + NULL +}; + +static const struct hwmon_chip_info um780xtx_chip_info = { + .ops = &um780xtx_hwmon_ops, + .info = um780xtx_info, +}; + +static int um780xtx_write_sys_point(u8 offset, u8 value) +{ + u8 readback; + int ret; + + ret = ec_write(offset, value); + if (ret) + return ret; + ret = ec_read(offset, &readback); + if (ret) + return ret; + + return readback == value ? 0 : -EIO; +} + +static int um780xtx_restore_sys_points(struct um780xtx_data *data, + u8 current_point2) +{ + u8 point1 = data->saved_sys_point1; + u8 point2 = data->saved_sys_point2; + int ret; + + /* Keep strict ordering valid after each individual EC write. */ + if (point1 >= current_point2) { + ret = um780xtx_write_sys_point(UM780XTX_EC_SYS_POINT2, point2); + if (ret) + return ret; + return um780xtx_write_sys_point(UM780XTX_EC_SYS_POINT1, + point1); + } + + ret = um780xtx_write_sys_point(UM780XTX_EC_SYS_POINT1, point1); + if (ret) + return ret; + return um780xtx_write_sys_point(UM780XTX_EC_SYS_POINT2, point2); +} + +static int um780xtx_read_initial_state(struct um780xtx_data *data) +{ + u8 profile; + u8 point1; + u8 point2; + int ret; + + ret = ec_read(UM780XTX_EC_CPU_PROFILE, &profile); + if (ret) + return ret; + if (profile != UM780XTX_EC_PROFILE_B1 && + profile != UM780XTX_EC_PROFILE_B2) + return -EINVAL; + + ret = ec_read(UM780XTX_EC_SYS_POINT1, &point1); + if (ret) + return ret; + ret = ec_read(UM780XTX_EC_SYS_POINT2, &point2); + if (ret) + return ret; + if (point1 >= point2 || point2 >= data->sys_point3) + return -EINVAL; + + data->saved_profile = profile; + data->saved_sys_point1 = point1; + data->saved_sys_point2 = point2; + return 0; +} + +static int um780xtx_restore_state(struct um780xtx_data *data) +{ + u8 readback; + u8 point2; + int ret; + + ret = ec_transaction(data->saved_profile, NULL, 0, NULL, 0); + if (ret) + return ret; + ret = ec_read(UM780XTX_EC_CPU_PROFILE, &readback); + if (ret) + return ret; + if (readback != data->saved_profile) + return -EIO; + + ret = ec_read(UM780XTX_EC_SYS_POINT2, &point2); + if (ret) + return ret; + + return um780xtx_restore_sys_points(data, point2); +} + +static int um780xtx_probe(struct platform_device *pdev) +{ + struct device *dev = &pdev->dev; + struct um780xtx_data *data; + struct device *hwmon; + int ret; + + data = devm_kzalloc(dev, sizeof(*data), GFP_KERNEL); + if (!data) + return -ENOMEM; + platform_set_drvdata(pdev, data); + + ret = ec_read(UM780XTX_EC_SYS_POINT3, &data->sys_point3); + if (ret) + return dev_err_probe(dev, ret, + "failed to read system fan limit\n"); + ret = um780xtx_read_initial_state(data); + if (ret) + return dev_err_probe(dev, ret, + "failed to read initial fan settings\n"); + + hwmon = devm_hwmon_device_register_with_info(dev, "um780xtx_ec", data, + &um780xtx_chip_info, + um780xtx_extra_groups); + if (IS_ERR(hwmon)) + return PTR_ERR(hwmon); + data->hwmon_dev = hwmon; + + return 0; +} + +static int um780xtx_resume(struct device *dev) +{ + struct um780xtx_data *data = dev_get_drvdata(dev); + + guard(hwmon_lock)(data->hwmon_dev); + return um780xtx_restore_state(data); +} + +static DEFINE_SIMPLE_DEV_PM_OPS(um780xtx_pm_ops, NULL, um780xtx_resume); + +static struct platform_driver um780xtx_driver = { + .driver = { + .name = "um780xtx-ec-hwmon", + .pm = pm_sleep_ptr(&um780xtx_pm_ops), + }, + .probe = um780xtx_probe, +}; + +static int __init um780xtx_init(void) +{ + int ret; + + if (!dmi_check_system(um780xtx_dmi_table) || + !um780xtx_firmware_match() || !ec_get_handle()) + return -ENODEV; + ret = platform_driver_register(&um780xtx_driver); + if (ret) + return ret; + + um780xtx_pdev = platform_device_register_simple("um780xtx-ec-hwmon", + PLATFORM_DEVID_NONE, + NULL, 0); + if (IS_ERR(um780xtx_pdev)) { + ret = PTR_ERR(um780xtx_pdev); + platform_driver_unregister(&um780xtx_driver); + return ret; + } + return 0; +} + +static void __exit um780xtx_exit(void) +{ + platform_device_unregister(um780xtx_pdev); + platform_driver_unregister(&um780xtx_driver); +} + +module_init(um780xtx_init); +module_exit(um780xtx_exit); + +MODULE_AUTHOR("Sebastián Peyrott "); +MODULE_DESCRIPTION("Minisforum UM780 XTX EC hwmon driver"); +MODULE_LICENSE("GPL"); From ced0b7d819ae0e6a9da2a1f39e1ec281aa9a6cb0 Mon Sep 17 00:00:00 2001 From: Sanman Pradhan Date: Tue, 1 Sep 2026 21:11:48 +0000 Subject: [PATCH 514/857] dt-bindings: trivial-devices: Add TI TPS53622 and TPS53659 TPS53622 and TPS53659 are PMBus-compliant D-CAP+ multiphase step-down controllers. Add their compatible strings. Signed-off-by: Sanman Pradhan Acked-by: Conor Dooley Link: https://patch.msgid.link/20260901211129.360792-2-sanman.pradhan@hpe.com Signed-off-by: Guenter Roeck --- Documentation/devicetree/bindings/trivial-devices.yaml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/Documentation/devicetree/bindings/trivial-devices.yaml b/Documentation/devicetree/bindings/trivial-devices.yaml index 6229a65163027b..94e63d8a58ad50 100644 --- a/Documentation/devicetree/bindings/trivial-devices.yaml +++ b/Documentation/devicetree/bindings/trivial-devices.yaml @@ -512,8 +512,12 @@ properties: - ti,tmp125 # TI DC-DC converter on PMBus - ti,tps40400 + # TI Dual channel DCAP+ multiphase controller TPS53622 + - ti,tps53622 # TI DCAP+ multiphase controller - ti,tps53647 + # TI Dual channel DCAP+ multiphase controller TPS53659 + - ti,tps53659 # TI DCAP+ multiphase controller - ti,tps53667 # TI Dual channel DCAP+ multiphase controller TPS53676 with AVSBus From ba08432bda66a7889d8f3d1581dabf10f59b25eb Mon Sep 17 00:00:00 2001 From: Sanman Pradhan Date: Tue, 1 Sep 2026 21:11:53 +0000 Subject: [PATCH 515/857] hwmon: (pmbus/tps53679) Add support for TPS53622 and TPS53659 TPS53622 and TPS53659 are dual-channel D-CAP+ step-down controllers that use the VID VOUT format and VOUT_MODE identification like the existing TPS53679/TPS53688, so they reuse tps53679_identify(). Shorten the Kconfig prompt to the family name and list the supported chips in the help text instead; this also adds TPS53685, which is already supported by the driver but was missing from the list. Update the driver documentation, including the per-attribute lists, and fix an existing "TPS53588" typo (should be TPS53688) in those lists. Signed-off-by: Sanman Pradhan Link: https://patch.msgid.link/20260901211129.360792-3-sanman.pradhan@hpe.com Signed-off-by: Guenter Roeck --- Documentation/hwmon/tps53679.rst | 24 ++++++++++++++++++++---- drivers/hwmon/pmbus/Kconfig | 5 +++-- drivers/hwmon/pmbus/tps53679.c | 9 ++++++++- 3 files changed, 31 insertions(+), 7 deletions(-) diff --git a/Documentation/hwmon/tps53679.rst b/Documentation/hwmon/tps53679.rst index dd5e4a37375d11..2280e043c4de92 100644 --- a/Documentation/hwmon/tps53679.rst +++ b/Documentation/hwmon/tps53679.rst @@ -3,6 +3,14 @@ Kernel driver tps53679 Supported chips: + * Texas Instruments TPS53622 + + Prefix: 'tps53622' + + Addresses scanned: - + + Datasheet: https://www.ti.com/lit/gpn/TPS53622 + * Texas Instruments TPS53647 Prefix: 'tps53647' @@ -11,6 +19,14 @@ Supported chips: Datasheet: https://www.ti.com/lit/gpn/tps53647 + * Texas Instruments TPS53659 + + Prefix: 'tps53659' + + Addresses scanned: - + + Datasheet: https://www.ti.com/lit/gpn/TPS53659 + * Texas Instruments TPS53667 Prefix: 'tps53667' @@ -108,7 +124,7 @@ in1_crit_alarm Input voltage critical high alarm. in[N]_label "vout[1-2]" - TPS53647, TPS53667: N=2 - - TPS53679, TPS53588: N=2,3 + - TPS53622, TPS53659, TPS53679, TPS53688: N=2,3 in[N]_input Measured output voltage. @@ -135,7 +151,7 @@ in[N]_crit_alarm Output voltage critical high alarm. temp[N]_input Measured temperature. - TPS53647, TPS53667: N=1 - - TPS53679, TPS53681, TPS53588: N=1,2 + - TPS53622, TPS53659, TPS53679, TPS53681, TPS53688: N=1,2 temp[N]_max Maximum temperature. @@ -152,7 +168,7 @@ power1_input Measured input power. power[N]_label "pout[1-2]". - TPS53647, TPS53667: N=2 - - TPS53676, TPS53679, TPS53681, TPS53588: N=2,3 + - TPS53622, TPS53659, TPS53676, TPS53679, TPS53681, TPS53688: N=2,3 power[N]_input Measured output power. @@ -175,7 +191,7 @@ curr[N]_label "iout[1-2]" or "iout1.[0-5]". telemetry supported on TPS53676 and TPS53681 only. - TPS53647, TPS53667: N=2 - - TPS53679, TPS53588: N=2,3 + - TPS53622, TPS53659, TPS53679, TPS53688: N=2,3 - TPS53676: N=2-8 - TPS53681: N=2-9 diff --git a/drivers/hwmon/pmbus/Kconfig b/drivers/hwmon/pmbus/Kconfig index eb71a0ac044b26..bcfdc4ce4c1006 100644 --- a/drivers/hwmon/pmbus/Kconfig +++ b/drivers/hwmon/pmbus/Kconfig @@ -790,10 +790,11 @@ config SENSORS_TPS40422 be called tps40422. config SENSORS_TPS53679 - tristate "TI TPS53647, TPS53667, TPS53676, TPS53679, TPS53681, TPS53688" + tristate "TI TPS536xx family" help If you say yes here you get hardware monitoring support for TI - TPS53647, TPS53667, TPS53676, TPS53679, TPS53681, and TPS53688. + TPS53622, TPS53647, TPS53659, TPS53667, TPS53676, TPS53679, TPS53681, + TPS53685, and TPS53688. This driver can also be built as a module. If so, the module will be called tps53679. diff --git a/drivers/hwmon/pmbus/tps53679.c b/drivers/hwmon/pmbus/tps53679.c index 31e54608b3c9c0..fa0fdf1e3e6c9b 100644 --- a/drivers/hwmon/pmbus/tps53679.c +++ b/drivers/hwmon/pmbus/tps53679.c @@ -16,7 +16,8 @@ #include "pmbus.h" enum chips { - tps53647, tps53667, tps53676, tps53679, tps53681, tps53685, tps53688 + tps53622, tps53647, tps53659, tps53667, tps53676, tps53679, tps53681, + tps53685, tps53688 }; #define TPS53647_PAGE_NUM 1 @@ -268,6 +269,8 @@ static int tps53679_probe(struct i2c_client *client) case tps53676: info->identify = tps53676_identify; break; + case tps53622: + case tps53659: case tps53679: case tps53688: info->pages = TPS53679_PAGE_NUM; @@ -292,7 +295,9 @@ static int tps53679_probe(struct i2c_client *client) static const struct i2c_device_id tps53679_id[] = { { .name = "bmr474", .driver_data = tps53676 }, + { .name = "tps53622", .driver_data = tps53622 }, { .name = "tps53647", .driver_data = tps53647 }, + { .name = "tps53659", .driver_data = tps53659 }, { .name = "tps53667", .driver_data = tps53667 }, { .name = "tps53676", .driver_data = tps53676 }, { .name = "tps53679", .driver_data = tps53679 }, @@ -305,7 +310,9 @@ static const struct i2c_device_id tps53679_id[] = { MODULE_DEVICE_TABLE(i2c, tps53679_id); static const struct of_device_id __maybe_unused tps53679_of_match[] = { + {.compatible = "ti,tps53622", .data = (void *)tps53622}, {.compatible = "ti,tps53647", .data = (void *)tps53647}, + {.compatible = "ti,tps53659", .data = (void *)tps53659}, {.compatible = "ti,tps53667", .data = (void *)tps53667}, {.compatible = "ti,tps53676", .data = (void *)tps53676}, {.compatible = "ti,tps53679", .data = (void *)tps53679}, From ccaa1524a18aaad2d3646f614f2cbd5941dca50a Mon Sep 17 00:00:00 2001 From: Jan Kara Date: Wed, 2 Sep 2026 16:36:57 +0200 Subject: [PATCH 516/857] gfs2: Fix journaled truncate range end offset The third argument to truncate_pagecache_range() is the inclusive end offset, but gfs2_journaled_truncate_range() passes the length of each chunk. A hole punch at an offset larger than its length therefore creates an inverted range. Commit 86066914edff ("gfs2: Don't support fallocate on jdata files") disables fallocate for jdata files. Otherwise, the following bug could occur: The MM truncation path then underflows the unsigned partial-folio length and passes an oversized range to folio_zero_range(), triggering a BUG. Pass the last byte of the chunk so page-cache invalidation covers the intended range. Fixes: 4e56a6411fbc ("gfs2: Implement fallocate(FALLOC_FL_PUNCH_HOLE)") Reported-by: VEGA Assisted-by: LLM Signed-off-by: Jan Kara Signed-off-by: Andreas Gruenbacher --- fs/gfs2/bmap.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/gfs2/bmap.c b/fs/gfs2/bmap.c index 3714364385c7cf..f0c4d054e5f04b 100644 --- a/fs/gfs2/bmap.c +++ b/fs/gfs2/bmap.c @@ -2429,7 +2429,7 @@ static int gfs2_journaled_truncate_range(struct inode *inode, loff_t offset, if (offs && chunk > PAGE_SIZE) chunk = offs + ((chunk - offs) & PAGE_MASK); - truncate_pagecache_range(inode, offset, chunk); + truncate_pagecache_range(inode, offset, offset + chunk - 1); offset += chunk; length -= chunk; From 76041b0fd27cc793cdd6ef6e865985daaf068cfa Mon Sep 17 00:00:00 2001 From: Hao Ge Date: Thu, 27 Aug 2026 11:05:03 +0800 Subject: [PATCH 517/857] module: fix lost error code from codetag_load_module() If codetag_load_module() fails, err is not set to reflect the failure and load_module() returns 0 after the module has been torn down. Also, if the module is a livepatch, mod->klp_info allocated by copy_module_elf() leaks on this error path. Free it via a new livepatch_cleanup label. Link: https://lore.kernel.org/20260827030503.49171-1-hao.ge@linux.dev Fixes: 044d2aee6c57 ("alloc_tag: handle module codetag load errors as module load failures") Signed-off-by: Hao Ge Reported-by: Sashiko Suggested-by: Petr Pavlu Reviewed-by: Bradley Morgan Cc: Aaron Tomlin Cc: Luis Chamberalin Cc: Sami Tolvanen Cc: Suren Baghdasaryan Cc: Signed-off-by: Andrew Morton --- kernel/module/main.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/kernel/module/main.c b/kernel/module/main.c index d0e1e0bd2ad06b..c1b34dc1e89ac6 100644 --- a/kernel/module/main.c +++ b/kernel/module/main.c @@ -3581,8 +3581,9 @@ static int load_module(struct load_info *info, const char __user *uargs, goto sysfs_cleanup; } - if (codetag_load_module(mod)) - goto sysfs_cleanup; + err = codetag_load_module(mod); + if (err) + goto livepatch_cleanup; /* Get rid of temporary copy. */ free_copy(info, flags); @@ -3592,6 +3593,9 @@ static int load_module(struct load_info *info, const char __user *uargs, return do_init_module(mod); + livepatch_cleanup: + if (is_livepatch_module(mod)) + free_module_elf(mod); sysfs_cleanup: mod_sysfs_teardown(mod); coming_cleanup: From 81cb6a7b0a63c3d94c5db4ac3491162c75d220ac Mon Sep 17 00:00:00 2001 From: Geert Uytterhoeven Date: Thu, 27 Aug 2026 09:20:56 +0200 Subject: [PATCH 518/857] MAINTAINERS: cover all of RAID While commit 3626738bc7147d52 ("raid6: move to lib/raid/") handled the move of RAID6, it didn't take into account there was already more RAID code under lib/raid/, as XOR got moved over in commit 9e229025e2474115 ("xor: move to lib/raid/") before. Link: https://lore.kernel.org/7a2e5de234cc0286e3fe9bc11b810433775f2280.1787815121.git.geert+renesas@glider.be Signed-off-by: Geert Uytterhoeven Reported-by: Andrew Morton Closes: https://lore.kernel.org/20260826205058.a6ff019d0584f75c7f50430b@linux-foundation.org Cc: Christoph Hellwig Cc: Song Liu Cc: Yu Kuai Cc: Li Nan Cc: Xiao Ni Signed-off-by: Andrew Morton --- MAINTAINERS | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/MAINTAINERS b/MAINTAINERS index 85cc77fe75b766..90ce4def17d9cd 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -25405,7 +25405,7 @@ F: drivers/md/md* F: drivers/md/raid* F: include/linux/raid/ F: include/uapi/linux/raid/ -F: lib/raid/raid6/ +F: lib/raid/ SOLIDRUN CLEARFOG SUPPORT M: Russell King From 2b9c3e4bef9c29ba655b1914466157f12aba4271 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Thu, 27 Aug 2026 11:34:35 +0100 Subject: [PATCH 519/857] MAINTAINERS: add Kiryl as a THP reviewer I have been working on transparent hugepages since 2012, starting with the huge zero page and file-backed THP. A lot of the code that causes pain now traces back to me. It is only fair if I share the review load for THP. Add myself to the reviewer list so get_maintainer.pl puts me on Cc: as well. It is also my commitment to be more active in reviewing this code. Link: https://lore.kernel.org/20260827103435.1371882-1-kas@kernel.org Signed-off-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Barry Song Acked-by: Zi Yan Reviewed-by: Lance Yang Acked-by: Usama Arif Acked-by: Baolin Wang Acked-by: SJ Park Cc: Dev Jain Cc: Liam R. Howlett Cc: Ryan Roberts Signed-off-by: Andrew Morton --- MAINTAINERS | 1 + 1 file changed, 1 insertion(+) diff --git a/MAINTAINERS b/MAINTAINERS index 90ce4def17d9cd..2133aec4a20046 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17421,6 +17421,7 @@ R: Dev Jain R: Barry Song R: Lance Yang R: Usama Arif +R: Kiryl Shutsemau L: linux-mm@kvack.org S: Maintained W: http://www.linux-mm.org From 26f099490599f89f89a56fbfec4a98fea00ad5ab Mon Sep 17 00:00:00 2001 From: Kazuki Hanai Date: Fri, 28 Aug 2026 00:25:16 +0900 Subject: [PATCH 520/857] tmpfs: fix unicode_map leaks in casefold option handling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit shmem_parse_opt_casefold() stores the unicode_map returned by utf8_load() in ctx->encoding. The casefold parameter can be supplied more than once for the same filesystem context, but replacing the stored map does not release the previous reference. The final reference is also leaked when an unmounted filesystem context is freed. Release the previous map before replacing it, clear ctx->encoding after transferring ownership to the superblock, and release any remaining reference from shmem_free_fc(). An unprivileged user can repeatedly set the casefold parameter on a tmpfs filesystem context from a user namespace. This causes unbounded kernel memory consumption and can result in a local denial of service. Link: https://lore.kernel.org/20260827152516.805622-1-hnkz.64@gmail.com Fixes: 58e55efd6c72 ("tmpfs: Add casefold lookup support") Signed-off-by: Kazuki Hanai Cc: Baolin Wang Cc: Hugh Dickins Cc: André Almeida Cc: Christian Brauner Cc: Signed-off-by: Andrew Morton --- mm/shmem.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/mm/shmem.c b/mm/shmem.c index 897fa2b61346f6..de144a9a9558b7 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -4527,6 +4527,7 @@ static int shmem_parse_opt_casefold(struct fs_context *fc, struct fs_parameter * pr_info("tmpfs: Using encoding : utf8-%u.%u.%u\n", unicode_major(version), unicode_minor(version), unicode_rev(version)); + utf8_unload(ctx->encoding); ctx->encoding = encoding; return 0; @@ -4995,6 +4996,7 @@ static int shmem_fill_super(struct super_block *sb, struct fs_context *fc) if (ctx->encoding) { sb->s_encoding = ctx->encoding; + ctx->encoding = NULL; set_default_d_op(sb, &shmem_ci_dentry_ops); if (ctx->strict_encoding) sb->s_encoding_flags = SB_ENC_STRICT_MODE_FL; @@ -5092,6 +5094,9 @@ static void shmem_free_fc(struct fs_context *fc) struct shmem_options *ctx = fc->fs_private; if (ctx) { +#if IS_ENABLED(CONFIG_UNICODE) + utf8_unload(ctx->encoding); +#endif mpol_put(ctx->mpol); kfree(ctx); } From ad71cfad5966a483077a6656c67a1e19dddcb074 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Tue, 25 Aug 2026 08:55:26 +0100 Subject: [PATCH 521/857] mm/mremap: reset unfaulted VMA page offset for MREMAP_DONTUNMAP Uniquely an mremap() invocation using the MREMAP_DONTUNMAP flag can reset a faulted VMA into an unfaulted one. It does so after the page tables have been moved to the copied VMA with MREMAP_DONTUNMAP leaving the old VMA in place which is naturally unfaulted as the page tables it had are no longer present. However, in doing so, it violates the invariant that the anonymous page offset of an unfaulted VMA is vma->vm_start >> PAGE_SHIFT. This is because a VMA may have been faulted in, mremap()'d (causing a delta between its page offset and vma->vm_start >> PAGE_SHIFT), and then mremap()'d again with MREMAP_DONTUNMAP resulting in the unfaulting. This condition is a violation of a fundamental assumption in mm, but now also triggers an assert in assert_sane_pgoff() which explicitly checks for this condition. Correct it by resetting the VMA's page offset at the point of completing the MREMAP_DONTUNMAP operation. Link: https://lore.kernel.org/20260825-fix-mremap-dontunmap-pgoff-v1-1-39a40b2c98b3@kernel.org Fixes: 1583aa278f5f ("mm: mremap: unlink anon_vmas when mremap with MREMAP_DONTUNMAP success") Signed-off-by: Lorenzo Stoakes (ARM) Reported-by: syzbot+f12658786a4153df5113@syzkaller.appspotmail.com Closes: https://lore.kernel.org/all/6a87853b.ae6ddae5.3da009.0023.GAE@google.com/ Acked-by: Vlastimil Babka (SUSE) Reviewed-by: Kunwu Chan Reviewed-by: Pedro Falcato Cc: Jann Horn Cc: Liam R. Howlett Cc: Li Xinhai Cc: Signed-off-by: Andrew Morton --- mm/mremap.c | 22 +++++++++++++++++----- 1 file changed, 17 insertions(+), 5 deletions(-) diff --git a/mm/mremap.c b/mm/mremap.c index e8df5cdb0ac9f1..2b4b523a86b87c 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -1331,18 +1331,30 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, { unsigned long start = vrm->addr; unsigned long end = vrm->addr + vrm->old_len; - unsigned long old_start = vrm->vma->vm_start; - unsigned long old_end = vrm->vma->vm_end; + struct vm_area_struct *vma = vrm->vma; + unsigned long old_start = vma->vm_start; + unsigned long old_end = vma->vm_end; /* We always clear VMA_LOCKED[ONFAULT]_BIT on the old VMA. */ - vma_clear_flags_mask(vrm->vma, VMA_LOCKED_MASK); + vma_clear_flags_mask(vma, VMA_LOCKED_MASK); /* * anon_vma links of the old vma is no longer needed after its page * table has been moved. */ - if (new_vma != vrm->vma && start == old_start && end == old_end) - unlink_anon_vmas(vrm->vma); + if (new_vma != vma && start == old_start && end == old_end) { + const pgoff_t pgoff_unfaulted = vma->vm_start >> PAGE_SHIFT; + + unlink_anon_vmas(vma); + /* + * The VMA is now unfaulted and it is an invariant that + * unfaulted anonymous VMAs have page offset equal to + * vma->vm_start >> PAGE_SHIFT. + */ + vma_set_anon_pgoff(vma, pgoff_unfaulted); + if (vma_is_anonymous(vma) && !vma->vm_file) + vma_set_pgoff(vma, pgoff_unfaulted); + } /* Because we won't unmap we don't need to touch locked_vm. */ } From 9f0ad4da49a2df18e6808d83d920e8ff2b224f48 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Wed, 26 Aug 2026 17:30:35 +0100 Subject: [PATCH 522/857] mm/secretmem: properly account locked pages secretmem accounts folios by treating memory as if it were mlock()'d and thus limited by the RLIMIT_MEMLOCK limit. However the folios are unevictable and remain so until the inode is evicted, eliminating usual mlock() semantics - mapping folios then unmapping them does not clear their unevictable state, since it depends on AS_UNEVICTABLE, not PG_mlocked. A user can therefore easily work around the RLIMIT_MEMLOCK limit - simply map then unmap and VmLck no longer counts the secretmem range. Worse, folios are not accounted in the process's RSS, meaning the OOM killer won't know to kill the process. Repeatedly mapping/unmapping (or forking) can then result in the consumption of all available system memory with unevictable folios and cause system instability. A secretmem fd can be passed between processes and over fork so a per-process limit simply does not make sense, so follow the precedent set by io_uring, perf, skbuff, iommufd and xdp by tracking the number of locked pages in user_struct->locked_vm. Since the scope tracked is actually inode lifetime, the RLIMIT_MEMLOCK applies per-user not per-process, so it doesn't make sense to bypass for users with CAP_IPC_LOCK, therefore remove this bypass. There is simply no reason to carry on marking the mapping as mlock()'d since it's misleading and the lifecycle is now correctly handled, so remove this too. Note that secretmem does not support any form of truncation (including hole punching) and the folios are unreclaimable, so the folios need only be accounted on fault and unaccounted on inode destruction. __secretmem_account_pages() is more or less a duplicate of the code that io_uring etc. use, but since this is a bug fix that needs backporting, defer any de-duplication efforts to a follow-up. test_mlock_limit() asserts mlock_future_ok() on mmap(), however this has been removed, so remove the test altogether for the fix. A new test will be sent separately for upstream. Link: https://lore.kernel.org/20260826-secretmem-accounting-v3-1-94cb04399510@kernel.org Fixes: 1507f51255c9 ("mm: introduce memfd_secret system call to create "secret" memory areas") Signed-off-by: Lorenzo Stoakes (ARM) Reported-by: Daehyeon Ko <4ncienth@gmail.com> Closes: https://lore.kernel.org/linux-mm/20260813225328.2010303-1-4ncienth@gmail.com/ Reviewed-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Daehyeon Ko <4ncienth@gmail.com> Cc: Alexei Starovoitov Cc: David Hildenbrand Cc: David S. Miller Cc: Hagen Paul Pfeifer Cc: Jakub Kacinski Cc: James Bottomley Cc: Jesper Dangaard Brouer Cc: John Fastabend Cc: Liam R. Howlett Cc: Michal Hocko Cc: Stanislav Fomichev Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Signed-off-by: Andrew Morton --- include/linux/sched/user.h | 3 +- mm/secretmem.c | 116 ++++++++++++++++++++-- tools/testing/selftests/mm/memfd_secret.c | 30 +----- 3 files changed, 110 insertions(+), 39 deletions(-) diff --git a/include/linux/sched/user.h b/include/linux/sched/user.h index 4cc52698e214e2..8d7e5521f7cdd9 100644 --- a/include/linux/sched/user.h +++ b/include/linux/sched/user.h @@ -25,7 +25,8 @@ struct user_struct { #if defined(CONFIG_PERF_EVENTS) || defined(CONFIG_BPF_SYSCALL) || \ defined(CONFIG_NET) || defined(CONFIG_IO_URING) || \ - defined(CONFIG_VFIO_PCI_ZDEV_KVM) || IS_ENABLED(CONFIG_IOMMUFD) + defined(CONFIG_VFIO_PCI_ZDEV_KVM) || IS_ENABLED(CONFIG_IOMMUFD) || \ + defined(CONFIG_SECRETMEM) atomic_long_t locked_vm; #endif #ifdef CONFIG_WATCH_QUEUE diff --git a/mm/secretmem.c b/mm/secretmem.c index d29865075b6ea0..384f5cfc457f9e 100644 --- a/mm/secretmem.c +++ b/mm/secretmem.c @@ -18,6 +18,8 @@ #include #include #include +#include +#include #include @@ -47,10 +49,69 @@ bool secretmem_active(void) return !!atomic_read(&secretmem_users); } +struct secretmem_inode_state { + struct user_struct *user; + atomic_long_t nr_pages_accounted; +}; + +static bool __secretmem_account_pages(struct user_struct *user, + unsigned long nr_pages) +{ + unsigned long page_limit, cur_pages, new_pages; + + if (!nr_pages) + return true; + + page_limit = rlimit(RLIMIT_MEMLOCK) >> PAGE_SHIFT; + + cur_pages = atomic_long_read(&user->locked_vm); + do { + new_pages = cur_pages + nr_pages; + if (new_pages > page_limit) + return false; + } while (!atomic_long_try_cmpxchg(&user->locked_vm, + &cur_pages, new_pages)); + return true; +} + +static bool secretmem_account_folio(struct secretmem_inode_state *state, + const struct folio *folio) +{ + const unsigned long nr_pages = folio_nr_pages(folio); + + if (!__secretmem_account_pages(state->user, nr_pages)) + return false; + + atomic_long_add(nr_pages, &state->nr_pages_accounted); + return true; +} + +static void __secretmem_unaccount_pages(struct secretmem_inode_state *state, + unsigned long nr_pages) +{ + atomic_long_sub(nr_pages, &state->user->locked_vm); + atomic_long_sub(nr_pages, &state->nr_pages_accounted); +} + +static void secretmem_unaccount_folio(struct secretmem_inode_state *state, + struct folio *folio) +{ + __secretmem_unaccount_pages(state, folio_nr_pages(folio)); +} + +static void secretmem_unaccount_all_folios(struct secretmem_inode_state *state) +{ + const unsigned long nr_pages_accounted = + atomic_long_read(&state->nr_pages_accounted); + + __secretmem_unaccount_pages(state, nr_pages_accounted); +} + static vm_fault_t secretmem_fault(struct vm_fault *vmf) { struct address_space *mapping = vmf->vma->vm_file->f_mapping; struct inode *inode = file_inode(vmf->vma->vm_file); + struct secretmem_inode_state *state = inode->i_private; pgoff_t offset = vmf->pgoff; gfp_t gfp = vmf->gfp_mask; unsigned long addr; @@ -72,8 +133,15 @@ static vm_fault_t secretmem_fault(struct vm_fault *vmf) goto out; } + if (!secretmem_account_folio(state, folio)) { + folio_put(folio); + ret = VM_FAULT_SIGBUS; + goto out; + } + err = set_direct_map_invalid_noflush(folio_page(folio, 0)); if (err) { + secretmem_unaccount_folio(state, folio); folio_put(folio); ret = vmf_error(err); goto out; @@ -82,6 +150,7 @@ static vm_fault_t secretmem_fault(struct vm_fault *vmf) __folio_mark_uptodate(folio); err = filemap_add_folio(mapping, folio, offset, gfp); if (unlikely(err)) { + secretmem_unaccount_folio(state, folio); /* * If a split of large page was required, it * already happened when we marked the page invalid @@ -112,22 +181,30 @@ static const struct vm_operations_struct secretmem_vm_ops = { .fault = secretmem_fault, }; +static void secretmem_destroy_inode_priv(struct inode *inode) +{ + struct secretmem_inode_state *state = inode->i_private; + + secretmem_unaccount_all_folios(state); + free_uid(state->user); + kfree(state); + inode->i_private = NULL; +} + static int secretmem_release(struct inode *inode, struct file *file) { atomic_dec(&secretmem_users); + secretmem_destroy_inode_priv(inode); + return 0; } static int secretmem_mmap_prepare(struct vm_area_desc *desc) { - const unsigned long len = vma_desc_size(desc); - if (!vma_desc_test_any(desc, VMA_SHARED_BIT, VMA_MAYSHARE_BIT)) return -EINVAL; - vma_desc_set_flags(desc, VMA_LOCKED_BIT, VMA_DONTDUMP_BIT); - if (!mlock_future_ok(desc->mm, /*is_vma_locked=*/ true, len)) - return -EAGAIN; + vma_desc_set_flags(desc, VMA_DONTDUMP_BIT); desc->vm_ops = &secretmem_vm_ops; return 0; @@ -187,20 +264,40 @@ static const struct inode_operations secretmem_iops = { static struct vfsmount *secretmem_mnt; +static int secretmem_init_inode_priv(struct inode *inode) +{ + struct secretmem_inode_state *state; + + state = kzalloc_obj(*state); + if (!state) + return -ENOMEM; + + state->user = get_uid(current_user()); + inode->i_private = state; + return 0; +} + static struct file *secretmem_file_create(unsigned long flags) { struct file *file; struct inode *inode; const char *anon_name = "[secretmem]"; + int err; inode = anon_inode_make_secure_inode(secretmem_mnt->mnt_sb, anon_name, NULL); if (IS_ERR(inode)) return ERR_CAST(inode); + err = secretmem_init_inode_priv(inode); + if (err) + goto err_free_inode; + file = alloc_file_pseudo(inode, secretmem_mnt, "secretmem", O_RDWR | O_LARGEFILE, &secretmem_fops); - if (IS_ERR(file)) - goto err_free_inode; + if (IS_ERR(file)) { + err = PTR_ERR(file); + goto err_free_priv; + } mapping_set_gfp_mask(inode->i_mapping, GFP_USER); mapping_set_unevictable(inode->i_mapping); @@ -215,10 +312,11 @@ static struct file *secretmem_file_create(unsigned long flags) atomic_inc(&secretmem_users); return file; - +err_free_priv: + secretmem_destroy_inode_priv(inode); err_free_inode: iput(inode); - return file; + return ERR_PTR(err); } SYSCALL_DEFINE1(memfd_secret, unsigned int, flags) diff --git a/tools/testing/selftests/mm/memfd_secret.c b/tools/testing/selftests/mm/memfd_secret.c index aac4f795c327bd..c55d84c5e613a4 100644 --- a/tools/testing/selftests/mm/memfd_secret.c +++ b/tools/testing/selftests/mm/memfd_secret.c @@ -57,33 +57,6 @@ static void test_file_apis(int fd) pass("file IO is blocked as expected\n"); } -static void test_mlock_limit(int fd) -{ - size_t len; - char *mem; - - len = mlock_limit_cur; - if (len % page_size != 0) - len = (len/page_size) * page_size; - - mem = mmap(NULL, len, prot, mode, fd, 0); - if (mem == MAP_FAILED) { - fail("unable to mmap secret memory\n"); - return; - } - munmap(mem, len); - - len = mlock_limit_max * 2; - mem = mmap(NULL, len, prot, mode, fd, 0); - if (mem != MAP_FAILED) { - fail("unexpected mlock limit violation\n"); - munmap(mem, len); - return; - } - - pass("mlock limit is respected\n"); -} - static void test_vmsplice(int fd, const char *desc) { ssize_t transferred; @@ -297,7 +270,7 @@ static void prepare(void) strerror(errno)); } -#define NUM_TESTS 6 +#define NUM_TESTS 5 int main(int argc, char *argv[]) { @@ -319,7 +292,6 @@ int main(int argc, char *argv[]) if (ftruncate(fd, page_size)) ksft_exit_fail_msg("ftruncate failed: %s\n", strerror(errno)); - test_mlock_limit(fd); test_file_apis(fd); /* * We have to run the first vmsplice test before any secretmem page was From 73c686fb5f147116a4db3d5f83e896c1c601ed81 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 27 Aug 2026 20:55:57 +0100 Subject: [PATCH 523/857] mm/huge_memory: bypass THP tuneables for huge pfnmap mappings The sysfs THP tuneables at /sys/kernel/mm/transparent_huge_pages/ rather confusingly only control the behaviour of THP in some instances. They are not applicable to MADV_COLLAPSE operations, nor to DAX mappings. Long-term, THP is predicated upon compaction being able to obtain large folios to populate THP ranges. However, vm_normal_folio() returns NULL for PFN map mappings, thus their reference count is maintained by the driver, not core mm. As a consequence, the folios are not subject to reclaim nor compaction, so are not truly part of the THP mechanism at all. However, since commit 5dd40721f147 ("mm: allow THP orders for PFNMAPs") introduced the ability to establish huge PFN maps, they have been subject to THP tuneables. This is incorrect - if a huge PFN map is available (defined by vma->vm_ops->huge_fault being non-NULL for a VMA_PFNMAP_BIT VMA), then it should be mapped huge upon fault-in. Correct this by explicitly checking for this while ensuring that smaps continues to accurately report THPeligible statistics. While here, abstract the entire file-backed THP check in vma_can_map_huge_file(), with sensible separation of logic into helper functions. Note that drm_gem_shmem_mmap() and panthor_gem_mmap() establish huge PFN maps of shmem folios, however they are marked unevictable in drm_gem_get_pages(), and in any case would fail the reference check in __remove_mapping() even if they weren't. Failing to map huge PFN maps has resulted in significant real-world performance degradation, see links for details. Link: https://lore.kernel.org/20260827-hugepfn-allowable-orders-v1-1-94819c8807c8@kernel.org Fixes: 5dd40721f147 ("mm: allow THP orders for PFNMAPs") Signed-off-by: Lorenzo Stoakes (ARM) Reported-by: Cedric Le Goater Closes: https://lore.kernel.org/linux-mm/20260805055544.1568534-1-clg@redhat.com/ Reported-by: Saravanan D Closes: https://lore.kernel.org/linux-mm/20260821070520.25759-1-saravanand@crusoe.ai/ Reviewed-by: Zi Yan Tested-by: Saravanan D Tested-by: Lance Yang Reviewed-by: SJ Park Reviewed-by: Baolin Wang Cc: Barry Song Cc: David Hildenbrand Cc: Dev Jain Cc: Jason Gunthorpe Cc: Liam R. Howlett Cc: Peter Xu Cc: Ryan Roberts Cc: Signed-off-by: Andrew Morton --- mm/huge_memory.c | 86 +++++++++++++++++++++++++++++++++++------------- 1 file changed, 64 insertions(+), 22 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index afbb5974bd225a..4bf7b670586df0 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -92,7 +92,7 @@ unsigned long huge_anon_orders_madvise __read_mostly; unsigned long huge_anon_orders_inherit __read_mostly; static bool anon_orders_configured __initdata; -static inline bool file_thp_enabled(struct vm_area_struct *vma) +static inline bool file_thp_enabled(const struct vm_area_struct *vma) { struct inode *inode; @@ -118,6 +118,67 @@ static bool vma_is_special_huge(const struct vm_area_struct *vma) return vma_test_any(vma, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); } +static bool vma_bypass_thp_tuneables_file(const struct vm_area_struct *vma, + enum tva_type type) +{ + const bool has_huge_fault = vma->vm_ops->huge_fault; + + /* MADV_COLLAPSE ignores tuneables. */ + if (type == TVA_FORCED_COLLAPSE) + return true; + /* Huge PFN mappings are uncompactable so the policy doesn't apply. */ + if (vma_test(vma, VMA_PFNMAP_BIT) && has_huge_fault) + return true; + return false; +} + +static bool vma_thp_tuneables_allow_file(vm_flags_t vm_flags) +{ + /* THP=always? */ + if (hugepage_global_always()) + return true; + /* THP=madvise and marked MADV_HUGEPAGE? */ + if (hugepage_global_enabled() && (vm_flags & VM_HUGEPAGE)) + return true; + return false; +} + +static bool vma_check_thp_tuneables_file(const struct vm_area_struct *vma, + vm_flags_t vm_flags, enum tva_type type) +{ + return vma_bypass_thp_tuneables_file(vma, type) || + vma_thp_tuneables_allow_file(vm_flags); +} + +static bool vma_can_map_huge_file(const struct vm_area_struct *vma, + vm_flags_t vm_flags, enum tva_type type) +{ + const bool has_huge_fault = vma->vm_ops->huge_fault; + + /* + * Enforce THP collapse requirements as necessary. Anonymous vmas + * were already handled in thp_vma_allowable_orders(). + */ + if (!vma_check_thp_tuneables_file(vma, vm_flags, type)) + return false; + + switch (type) { + case TVA_PAGEFAULT: + /* + * Trust that ->huge_fault() handlers know what they are doing + * in fault path. + */ + return has_huge_fault; + case TVA_SMAPS: + if (has_huge_fault) + return true; + fallthrough; + default: + /* Only regular file is valid in collapse path. */ + return file_thp_enabled(vma); + } +} + unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, vm_flags_t vm_flags, enum tva_type type, @@ -190,27 +251,8 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, vma, vma_start_pgoff(vma), 0, forced_collapse); - if (!vma_is_anonymous(vma)) { - /* - * Enforce THP collapse requirements as necessary. Anonymous vmas - * were already handled in thp_vma_allowable_orders(). - */ - if (!forced_collapse && - (!hugepage_global_enabled() || (!(vm_flags & VM_HUGEPAGE) && - !hugepage_global_always()))) - return 0; - - /* - * Trust that ->huge_fault() handlers know what they are doing - * in fault path. - */ - if (((in_pf || smaps)) && vma->vm_ops->huge_fault) - return orders; - /* Only regular file is valid in collapse path */ - if (((!in_pf || smaps)) && file_thp_enabled(vma)) - return orders; - return 0; - } + if (!vma_is_anonymous(vma)) + return vma_can_map_huge_file(vma, vm_flags, type) ? orders : 0; if (vma_is_temporary_stack(vma)) return 0; From ef85840d81314097ece1b3f46e5d32ceab496194 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sat, 29 Aug 2026 10:02:02 -0400 Subject: [PATCH 524/857] renames for Lorenzo's mm-huge_memory-bypass-thp-tuneables-for-huge-pfnmap-mappings patch rename some functions Link: https://lore.kernel.org/DL1HIHWYJ7TB.1CY76SJS0V03L@nvidia.com Signed-off-by: Zi Yan Cc: Baolin Wang Cc: Barry Song Cc: Cedric Le Goater Cc: David Hildenbrand Cc: Dev Jain Cc: Jason Gunthorpe Cc: Lance Yang Cc: Liam R. Howlett Cc: "Lorenzo Stoakes (ARM)" Cc: Peter Xu Cc: Ryan Roberts Cc: Saravanan D Cc: SJ Park Signed-off-by: Andrew Morton --- mm/huge_memory.c | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 4bf7b670586df0..1e5d68acf62a52 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -118,7 +118,7 @@ static bool vma_is_special_huge(const struct vm_area_struct *vma) return vma_test_any(vma, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); } -static bool vma_bypass_thp_tuneables_file(const struct vm_area_struct *vma, +static bool vma_file_bypass_thp_tuneables(const struct vm_area_struct *vma, enum tva_type type) { const bool has_huge_fault = vma->vm_ops->huge_fault; @@ -132,7 +132,7 @@ static bool vma_bypass_thp_tuneables_file(const struct vm_area_struct *vma, return false; } -static bool vma_thp_tuneables_allow_file(vm_flags_t vm_flags) +static bool vma_file_allow_thp_tuneables(vm_flags_t vm_flags) { /* THP=always? */ if (hugepage_global_always()) @@ -143,11 +143,11 @@ static bool vma_thp_tuneables_allow_file(vm_flags_t vm_flags) return false; } -static bool vma_check_thp_tuneables_file(const struct vm_area_struct *vma, +static bool vma_file_check_thp_tuneables(const struct vm_area_struct *vma, vm_flags_t vm_flags, enum tva_type type) { - return vma_bypass_thp_tuneables_file(vma, type) || - vma_thp_tuneables_allow_file(vm_flags); + return vma_file_bypass_thp_tuneables(vma, type) || + vma_file_allow_thp_tuneables(vm_flags); } static bool vma_can_map_huge_file(const struct vm_area_struct *vma, @@ -159,7 +159,7 @@ static bool vma_can_map_huge_file(const struct vm_area_struct *vma, * Enforce THP collapse requirements as necessary. Anonymous vmas * were already handled in thp_vma_allowable_orders(). */ - if (!vma_check_thp_tuneables_file(vma, vm_flags, type)) + if (!vma_file_check_thp_tuneables(vma, vm_flags, type)) return false; switch (type) { From 887d9fc42b4ad642b0092d3b8ed802650b44675e Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Sun, 23 Aug 2026 12:40:51 +0800 Subject: [PATCH 525/857] mm/hugetlb: do not dissolve gigantic pages without runtime support dissolve_free_hugetlb_folio() doesn't check hstate_is_gigantic_no_runtime(h) though remove_hugetlb_folio()/ update_and_free_hugetlb_folio() silently bail for such folios, so it frees a still-listed folio and, on vmemmap restore failure, the add_hugetlb_folio() rollback corrupts the free list. Link: https://lore.kernel.org/20260823044118.1097121-2-xialonglong2025@163.com Fixes: 6eb4e88a6d27 ("hugetlb: create remove_hugetlb_page() to separate functionality") Signed-off-by: Longlong Xia Assisted-by: Codex:gpt-5.6-sol Acked-by: Muchun Song Cc: David Hildenbrand Cc: Miaohe Lin Cc: Michal Hocko Cc: Oscar Salvador Cc: Signed-off-by: Andrew Morton --- mm/hugetlb.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 4f6f58bf3db6c1..d28972cd33f582 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -1967,6 +1967,15 @@ int dissolve_free_hugetlb_folio(struct folio *folio) struct hstate *h = folio_hstate(folio); bool adjust_surplus = false; + /* + * remove_hugetlb_folio()/update_and_free_hugetlb_folio() bail + * for gigantic hstates without runtime support, so dissolving one + * here would leave it on the free list and, on vmemmap restore + * failure, the add_hugetlb_folio() rollback corrupts that list. + */ + if (hstate_is_gigantic_no_runtime(h)) + goto out; + if (!available_huge_pages(h)) goto out; From 58a8596d69879cabbcf988c2a248758ef5b0fbdf Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 28 Aug 2026 12:20:37 +0100 Subject: [PATCH 526/857] mm/mremap: account mm->locked_vm correctly for MREMAP_DONTUNMAP When a VMA is mremap()'d with MREMAP_DONTUNMAP set, that results in the VMA being copied, but the source VMA not being unmapped. If the VMA is mlock()'d this is a legal operation, though the source VMA has its VMA_LOCKED_BIT cleared. However this is done in dontunmap_complete(), after mm->locked_vm was incremented via vrm_stat_account(), resulting in double-counting. Worse, this is not even corrected when source VMA is unmapped, due to the VMA_LOCKED_BIT flag having been cleared. This all works fine in the usual mremap() case (without MREMAP_DONTUNMAP), as the source VMA is unmapped with VMA_LOCKED_BIT intact, at which time mm->locked_vm is decremented accordingly. Resolve the issue by invoking vrm_stat_account() only after dontunmap_complete() has run. Note that MREMAP_DONTUNMAP requires old_len == new_len, so no need to account for a delta in size in this case. The bug was introduced by commit b714ccb02a76 ("mm/mremap: complete refactor of move_vma()") which incorrectly reordered the accounting and the clearing of the VMA_LOCKED_BIT flag. Link: https://lore.kernel.org/20260828-mremap-fix-locked-vm-v1-1-c80be7505d1e@kernel.org Fixes: b714ccb02a76 ("mm/mremap: complete refactor of move_vma()") Signed-off-by: Lorenzo Stoakes (ARM) Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260825-fix-mremap-dontunmap-pgoff-v1-1-39a40b2c98b3@kernel.org Reported-by: Kunwu Chan Closes: https://lore.kernel.org/all/20260828094823.594279-1-kunwu.chan@linux.dev/ Acked-by: Vlastimil Babka (SUSE) Tested-by: Kunwu Chan Reviewed-by: Kunwu Chan Cc: Jann Horn Cc: Liam R. Howlett Cc: Pedro Falcato Cc: Signed-off-by: Andrew Morton --- mm/mremap.c | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/mm/mremap.c b/mm/mremap.c index 2b4b523a86b87c..7c368440fafe24 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -1355,12 +1355,11 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, if (vma_is_anonymous(vma) && !vma->vm_file) vma_set_pgoff(vma, pgoff_unfaulted); } - - /* Because we won't unmap we don't need to touch locked_vm. */ } static unsigned long move_vma(struct vma_remap_struct *vrm) { + const bool is_dontunmap = vrm->flags & MREMAP_DONTUNMAP; struct mm_struct *mm = current->mm; struct vm_area_struct *new_vma; unsigned long hiwater_vm; @@ -1401,10 +1400,10 @@ static unsigned long move_vma(struct vma_remap_struct *vrm) */ hiwater_vm = mm->hiwater_vm; - vrm_stat_account(vrm, vrm->new_len); - if (unlikely(!err && (vrm->flags & MREMAP_DONTUNMAP))) + if (unlikely(is_dontunmap && !err)) dontunmap_complete(vrm, new_vma); - else + vrm_stat_account(vrm, vrm->new_len); + if (!is_dontunmap || err) unmap_source_vma(vrm); mm->hiwater_vm = hiwater_vm; From 01dd3cc4659ca66de780565cc768c229372a2e20 Mon Sep 17 00:00:00 2001 From: Coiby Xu Date: Fri, 28 Aug 2026 16:41:06 +0800 Subject: [PATCH 527/857] mailmap: map Coiby Xu's address Point to my gmail address as I've left Red Hat. Link: https://lore.kernel.org/20260828084106.1494733-1-coiby.xu@gmail.com Signed-off-by: Coiby Xu Signed-off-by: Andrew Morton --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index 9dc7096b79f7dc..da366e7d2a1f65 100644 --- a/.mailmap +++ b/.mailmap @@ -221,6 +221,7 @@ Chuck Lever Chuck Lever Chuck Lever Claudiu Beznea +Coiby Xu Colin Ian King Corey Minyard Damian Hobson-Garcia From 9e5901b42be0b49212447e0e285f2c8fcec41058 Mon Sep 17 00:00:00 2001 From: Nhat Pham Date: Fri, 28 Aug 2026 12:14:33 -0700 Subject: [PATCH 528/857] mm, swap: fix SWAP_USAGE_OFFLIST_BIT collision with real usage count SWAP_USAGE_OFFLIST_BIT is embedded in the si->inuse_pages usage counter, and is meant to sit above any value that counter can reach. However, it is defined from BITS_PER_TYPE(atomic_t), so it is bit 30. On a system with 4 KiB pages the flag collides with the usage count once that count reaches 4 TiB. swap_usage_in_pages() masks bit 30 out, so whenever the real count has that bit set, every caller of it reads 4 TiB low: * /proc/swaps understates Used by 4 TiB. * A raw count of exactly 2^30 masks to zero, so try_to_unuse() takes its "if (!swap_usage_in_pages(si)) goto success;" early exit and swapoff tears the device down while pages are still swapped out. Nothing in the rest of swapoff aborts the teardown, so those pages are lost. Independently of swapoff, the collision also corrupts the counter and the plist. On a device in normal use, a free that leaves bit 30 set in the count makes swap_usage_sub() see the flag where there is only count, and call add_to_avail_list(). It clears the bit with fetch_and(~SWAP_USAGE_OFFLIST_BIT), leaving the stored count 4 TiB below the real one, and calls plist_add() on a device that is already listed, tripping the WARN_ON(!plist_node_empty(node)) in plist_add() and linking the node a second time. Change the definition of SWAP_USAGE_OFFLIST_BIT to be based on atomic_long_t instead. Note that the usage counter field itself is of this same type, so it is still a valid bit. Link: https://lore.kernel.org/20260828191433.3304458-1-nphamcs@gmail.com Fixes: b228386cf237 ("mm, swap: clean up plist removal and adding") Signed-off-by: Nhat Pham Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260825153238.2695446-1-nphamcs%40gmail.com Suggested-by: Andrew Morton Reviewed-by: Andrew Morton Acked-by: Kairui Song Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: Gregory Price Cc: Johannes Weiner Cc: Joshua Hahn Cc: Kemeng Shi Cc: Shakeel Butt Cc: Youngjun Park Cc: Signed-off-by: Andrew Morton --- mm/swapfile.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 53bf01d5f7f112..601979b97f95b2 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -156,7 +156,7 @@ static struct swap_info_struct *swap_entry_to_info(swp_entry_t entry) * This bit will be set if the device is not on the plist and not * usable, will be cleared if the device is on the plist. */ -#define SWAP_USAGE_OFFLIST_BIT (1UL << (BITS_PER_TYPE(atomic_t) - 2)) +#define SWAP_USAGE_OFFLIST_BIT (1UL << (BITS_PER_TYPE(atomic_long_t) - 2)) #define SWAP_USAGE_COUNTER_MASK (~SWAP_USAGE_OFFLIST_BIT) static long swap_usage_in_pages(struct swap_info_struct *si) { From 95940065b7dd5b15f54da2aaa9ad99af9b68ee01 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 28 Aug 2026 19:32:51 -0700 Subject: [PATCH 529/857] memcg: avoid charging the root memcg from obj_cgroup_charge_pages() obj_cgroup_charge_pages() resolves the objcg to its memcg and calls try_charge_memcg(), which does not short circuit the root memcg. That memcg can be the root memcg: obj_cgroup_is_root() reflects the memcg the objcg was created for and is never updated, while memcg_reparent_objcgs() does redirect objcg->memcg to the parent on rmdir. An objcg of a dying child of root therefore passes every obj_cgroup_is_root() filter but resolves to the root memcg. Folios keep the objcg they were charged with, so this is easy to reach through zswap: allocate anon memory in a cgroup, move the task out, remove the cgroup, then write to the root cgroup's memory.reclaim. The reclaimed folios are charged through the reparented objcg and end up in refill_stock() with the root memcg: WARNING: mm/memcontrol.c:2198 at refill_stock+0x644/0x940 refill_stock+0x644/0x940 try_charge_memcg+0x12d6/0x1570 __obj_cgroup_charge+0x35/0xf0 obj_cgroup_charge+0x1de/0x210 obj_cgroup_charge_zswap+0x83/0x270 zswap_store+0x1620/0x2000 swap_writeout+0x94c/0x14c0 shrink_folio_list+0x3388/0x52b0 [...] try_to_free_mem_cgroup_pages+0x30d/0x830 user_proactive_reclaim+0x504/0x840 memory_reclaim+0x1f/0x30 Beyond the warning, the charge is asymmetric: obj_cgroup_uncharge_pages() skips refill_stock() for the root memcg, so the root's page counter grows and is never uncharged. It is not user visible, since memory.current is not exposed on the root, but it is a leak. Use try_charge(), which returns early for the root memcg, restoring the symmetry with obj_cgroup_uncharge_pages(). The above sequence was scripted into a standalone reproducer (zswap on, swap on a virtio disk, 512MB of anon memory faulted in inside a child of the root cgroup, the task then migrated to the root cgroup, the child removed, followed by "echo 600M swappiness=max > memory.reclaim" on the root) and run in a CONFIG_DEBUG_VM=y VM. It reproduces the splat on the first zswap store of a reparented folio, with the same call chain as the report. With this patch applied the splat is gone while the zswap store count over the run is unchanged, so the same path is still exercised. cgroup selftests test_zswap, test_kmem and test_memcontrol show no new failures. Link: https://lore.kernel.org/20260829023251.474083-1-shakeel.butt@linux.dev Fixes: 20d6c1725228 ("memcg: avoid refill_stock for root memcg") Signed-off-by: Shakeel Butt Reported-by: Farhad Alemi Closes: https://lore.kernel.org/all/CA+0ovCgWzUMK+nNbbtH7eV65Ca=fDN4Ozu7iASgryjvv8Tk8zQ@mail.gmail.com/ Reviewed-by: Muchun Song Reviewed-by: Johannes Weiner Cc: Michal Hocko Cc: Roman Gushchin Cc: Signed-off-by: Andrew Morton --- mm/memcontrol.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1271d390b617e4..856a7d07586ccc 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -3158,7 +3158,7 @@ static int obj_cgroup_charge_pages(struct obj_cgroup *objcg, gfp_t gfp, memcg = get_mem_cgroup_from_objcg(objcg); - ret = try_charge_memcg(memcg, gfp, nr_pages); + ret = try_charge(memcg, gfp, nr_pages); if (ret) goto out; From 2668025608fbf701a1c6d3e7e0950d93d92cbcea Mon Sep 17 00:00:00 2001 From: Christopher Obbard Date: Sat, 29 Aug 2026 12:28:23 +0100 Subject: [PATCH 530/857] mailmap: update entry for Christopher Obbard I have changed employer; update my mailmap entry to point at my new email address. Link: https://lore.kernel.org/20260829-update-mail-oss-qualcomm-v2-1-1670c515f225@oss.qualcomm.com Signed-off-by: Christopher Obbard Signed-off-by: Andrew Morton --- .mailmap | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.mailmap b/.mailmap index da366e7d2a1f65..22f016bc4f7f8c 100644 --- a/.mailmap +++ b/.mailmap @@ -210,7 +210,8 @@ Christophe Leroy Christophe Leroy Christophe Leroy Christophe Ricard -Christopher Obbard +Christopher Obbard +Christopher Obbard Christoph Hellwig Christoph Manszewski Christoph Paasch From d8601f138566f70d55114be1ee0638c76d1bceb1 Mon Sep 17 00:00:00 2001 From: Wenjie Qi Date: Sun, 30 Aug 2026 01:36:12 +0800 Subject: [PATCH 531/857] mm: filemap: retain mapped dropbehind folios Fault-around can map ready dropbehind folios without going through the normal page-cache lookup that clears dropbehind. A mapping represents a competing cached user, so retain the folio instead of forcibly unmapping it when writeback completes. For a mapped folio, folio_unmap_invalidate() can call unmap_mapping_folio(), which takes i_mmap_rwsem and may sleep. Retaining mapped folios avoids this path when folio_end_dropbehind() runs in non-preemptible task context. Tal was able to trigger a sleeping-in-atomic warning due to this [1]. Unmapped dropbehind folios continue through the existing invalidation path. Link: https://lore.kernel.org/4aba05e1a2c3b61cb337d373eb9b7a8db4ddd822.1788024049.git.qiwenjie@xiaomi.com Link: https://lore.kernel.org/076bb01b-6fcf-4691-be8c-0e8507c9fe64@columbia.edu [1] Fixes: fb7d3bc41493 ("mm/filemap: drop streaming/uncached pages when writeback completes") Signed-off-by: Wenjie Qi Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Tal Zussman Tested-by: Tal Zussman Cc: Barry Song Cc: Jan Kara Cc: Jens Axboe Cc: Trond Myklebust Cc: Signed-off-by: Andrew Morton --- mm/filemap.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/filemap.c b/mm/filemap.c index 6afec636881fb4..00fd89cf6f5509 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -1616,7 +1616,7 @@ static void filemap_end_dropbehind(struct folio *folio) return; if (!folio_test_clear_dropbehind(folio)) return; - if (mapping) + if (mapping && !folio_mapped(folio)) folio_unmap_invalidate(mapping, folio, 0); } From 75c7d7853e78434daeb6954d1a00a7c78b96a06f Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:25 +0800 Subject: [PATCH 532/857] fs: fix missed removal of super_fs_objects_eligible() Commit 0ef8faff490be ("fs: push nr_cached_objects memcg gating into individual filesystems") was meant to drop the blanket memcg gate in fs/super.c and let each ->nr_cached_objects() implementation decide for itself whether it is meaningful in per-memcg reclaim. However, when that patch was applied the removal of super_fs_objects_eligible() and its two call sites in super_cache_scan() / super_cache_count() was lost, so the helper is still gating every ->nr_cached_objects() hook and 0ef8faff490be is effectively a no-op. Consequences of the leftover gate: - XFS's inode-reclaim hook, which is intentionally driven from per-memcg contexts to free memcg-charged slab, is still short-circuited in fs/super.c exactly the regression from commit 0baad6f9b997 ("fs/super: skip non-memcg-aware nr_cached_objects in memcg slab shrink") that 0ef8faff490be was written to undo. Memcg-charged XFS inode slab therefore keeps piling up under per-memcg pressure until global reclaim kicks in. - Any future ->nr_cached_objects()/->free_cached_objects() that grows memcg awareness is likewise blocked before it can run, so filesystems cannot opt in to per-memcg reclaim on their own defeating the whole point of pushing the gating decision down into the callbacks. Drop the leftover helper and its call sites so the intent of 0ef8faff490be actually takes effect. Link: https://lore.kernel.org/cover.1786955972.git.zhengqi.arch@bytedance.com Link: https://lore.kernel.org/3b038d373c70ebac7cdabfb0035bb91d1d6e6cfe.1786955972.git.zhengqi.arch@bytedance.com Link: https://lore.kernel.org/all/20260715103516.2410175-1-usama.arif@linux.dev/ [0] Fixes: 0ef8faff490b ("fs: push nr_cached_objects memcg gating into individual filesystems") Signed-off-by: Qi Zheng Acked-by: Usama Arif Cc: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Signed-off-by: Andrew Morton --- fs/super.c | 18 ++---------------- 1 file changed, 2 insertions(+), 16 deletions(-) diff --git a/fs/super.c b/fs/super.c index 05e44317303874..3ecce24328f674 100644 --- a/fs/super.c +++ b/fs/super.c @@ -171,19 +171,6 @@ static void super_wake(struct super_block *sb, unsigned int flag) wake_up_var(&sb->s_flags); } -/* - * The s_op->nr_cached_objects hooks (used for example by btrfs and xfs) - * operate on filesystem-global state and ignore sc->memcg. Driving them - * from per-memcg shrink_slab_memcg() invocations only burns CPU walking - * per-cpu counters and queueing duplicate work: the actual reclaim happens on - * the global path (kswapd or root direct reclaim) regardless. Restrict them - * to that path. - */ -static inline bool super_fs_objects_eligible(struct shrink_control *sc) -{ - return !sc->memcg || mem_cgroup_is_root(sc->memcg); -} - /* * One thing we have to be careful of with a per-sb shrinker is that we don't * drop the last active reference to the superblock from within the shrinker. @@ -213,7 +200,7 @@ static unsigned long super_cache_scan(struct shrinker *shrink, if (!super_trylock_shared(sb)) return SHRINK_STOP; - if (sb->s_op->nr_cached_objects && super_fs_objects_eligible(sc)) + if (sb->s_op->nr_cached_objects) fs_objects = sb->s_op->nr_cached_objects(sb, sc); inodes = list_lru_shrink_count(&sb->s_inode_lru, sc); @@ -274,8 +261,7 @@ static unsigned long super_cache_count(struct shrinker *shrink, return 0; smp_rmb(); - if (sb->s_op && sb->s_op->nr_cached_objects && - super_fs_objects_eligible(sc)) + if (sb->s_op && sb->s_op->nr_cached_objects) total_objects = sb->s_op->nr_cached_objects(sb, sc); total_objects += list_lru_shrink_count(&sb->s_dentry_lru, sc); From 2eda0e4aed6d132acfa3a7105eace41856d47b9a Mon Sep 17 00:00:00 2001 From: Seunguk Shin Date: Mon, 3 Aug 2026 13:34:55 +0100 Subject: [PATCH 533/857] fs/dax: check zero or empty entry before converting xarray entry Calling dax_to_folio() with empty entry causes kernel panic below when booting a VM with DAX enabled storage. This patch checks empty entry before calling dax_to_folio() on dax_associate_entry(), dax_disassociate_entry(), and dax_busy_page(). Commit 98c183a4fccf ("fs/dax: don't disassociate zero page entries") added guards in the associate and disassociate paths, but the guards still come after dax_to_folio(), and dax_busy_page() still has the same problem. [ 0.737679] EXT4-fs (pmem0p1): mounted filesystem 79676804-7c8b-491a-b2a6-9bae3c72af70 ro with ordered data mode. Quota mode: disabled. [ 0.737891] VFS: Mounted root (ext4 filesystem) readonly on device 259:1. [ 0.739119] devtmpfs: mounted [ 0.739476] Freeing unused kernel memory: 1920K [ 0.740156] Run /sbin/init as init process [ 0.740229] with arguments: [ 0.740286] /sbin/init [ 0.740321] with environment: [ 0.740369] HOME=/ [ 0.740400] TERM=linux [ 0.743162] Unable to handle kernel paging request at virtual address fffffdffbf000008 [ 0.743285] Mem abort info: [ 0.743316] ESR = 0x0000000096000006 [ 0.743371] EC = 0x25: DABT (current EL), IL = 32 bits [ 0.743444] SET = 0, FnV = 0 [ 0.743489] EA = 0, S1PTW = 0 [ 0.743545] FSC = 0x06: level 2 translation fault [ 0.743610] Data abort info: [ 0.743656] ISV = 0, ISS = 0x00000006, ISS2 = 0x00000000 [ 0.743720] CM = 0, WnR = 0, TnD = 0, TagAccess = 0 [ 0.743785] GCS = 0, Overlay = 0, DirtyBit = 0, Xs = 0 [ 0.743848] swapper pgtable: 4k pages, 48-bit VAs, pgdp=00000000b9d17000 [ 0.743931] [fffffdffbf000008] pgd=10000000bfa3d403, p4d=10000000bfa3d403, pud=1000000040bfe403, pmd=0000000000000000 [ 0.744070] Internal error: Oops: 0000000096000006 [#1] SMP [ 0.748888] CPU: 0 UID: 0 PID: 1 Comm: init Not tainted 6.18.4 #1 NONE [ 0.749421] pstate: 004000c5 (nzcv daIF +PAN -UAO -TCO -DIT -SSBS BTYPE=--) [ 0.749969] pc : dax_disassociate_entry.constprop.0+0x20/0x50 [ 0.750444] lr : dax_insert_entry+0xcc/0x408 [ 0.750802] sp : ffff80008000b9e0 [ 0.751083] x29: ffff80008000b9e0 x28: 0000000000000000 x27: 0000000000000000 [ 0.751682] x26: 0000000001963d01 x25: ffff0000004f7d90 x24: 0000000000000000 [ 0.752264] x23: 0000000000000000 x22: ffff80008000bcc8 x21: 0000000000000011 [ 0.752836] x20: ffff80008000ba90 x19: 0000000001963d01 x18: 0000000000000000 [ 0.753407] x17: 0000000000000000 x16: 0000000000000000 x15: 0000000000000000 [ 0.753970] x14: ffffbf3154b9ae70 x13: 0000000000000000 x12: ffffbf3154b9ae70 [ 0.754548] x11: ffffffffffffffff x10: 0000000000000000 x9 : 0000000000000000 [ 0.755122] x8 : 000000000000000d x7 : 000000000000001f x6 : 0000000000000000 [ 0.755707] x5 : 0000000000000000 x4 : 0000000000000000 x3 : fffffdffc0000000 [ 0.756287] x2 : 0000000000000008 x1 : 0000000040000000 x0 : fffffdffbf000000 [ 0.756871] Call trace: [ 0.757107] dax_disassociate_entry.constprop.0+0x20/0x50 (P) [ 0.757592] dax_iomap_pte_fault+0x4fc/0x808 [ 0.757951] dax_iomap_fault+0x28/0x30 [ 0.758258] ext4_dax_huge_fault+0x80/0x2dc [ 0.758594] ext4_dax_fault+0x10/0x3c [ 0.758892] __do_fault+0x38/0x12c [ 0.759175] __handle_mm_fault+0x530/0xcf0 [ 0.759518] handle_mm_fault+0xe4/0x230 [ 0.759833] do_page_fault+0x17c/0x4dc [ 0.760144] do_translation_fault+0x30/0x38 [ 0.760483] do_mem_abort+0x40/0x8c [ 0.760771] el0_ia+0x4c/0x170 [ 0.761032] el0t_64_sync_handler+0xd8/0xdc [ 0.761371] el0t_64_sync+0x168/0x16c [ 0.761677] Code: f9453021 f2dfbfe3 cb813080 8b001860 (f9400401) [ 0.762168] ---[ end trace 0000000000000000 ]--- [ 0.762550] note: init[1] exited with irqs disabled [ 0.762631] Kernel panic - not syncing: Attempted to kill init! exitcode=0x0000000b Link: https://lore.kernel.org/m2y0enxtzk.fsf@arm.com Fixes: 38607c62b34b ("fs/dax: properly refcount fs dax pages") Signed-off-by: Seunguk Shin Reviewed-by: Jan Kara Reviewed-by: Alistair Popple Reported-by: Kiara Grouwstra Cc: Al Viro Cc: Christian Brauner Cc: Matthew Wilcox (Oracle) Signed-off-by: Andrew Morton --- fs/dax.c | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/fs/dax.c b/fs/dax.c index 6ba50142eeb2fd..1fbba0d21c13d8 100644 --- a/fs/dax.c +++ b/fs/dax.c @@ -480,11 +480,12 @@ static void dax_associate_entry(void *entry, struct address_space *mapping, unsigned long address, bool shared) { unsigned long size = dax_entry_size(entry), index; - struct folio *folio = dax_to_folio(entry); + struct folio *folio; if (dax_is_zero_entry(entry) || dax_is_empty_entry(entry)) return; + folio = dax_to_folio(entry); index = linear_page_index(vma, address & ~(size - 1)); if (shared && (folio->mapping || dax_folio_is_shared(folio))) { if (folio->mapping) @@ -505,21 +506,23 @@ static void dax_associate_entry(void *entry, struct address_space *mapping, static void dax_disassociate_entry(void *entry, struct address_space *mapping, bool trunc) { - struct folio *folio = dax_to_folio(entry); + struct folio *folio; if (dax_is_zero_entry(entry) || dax_is_empty_entry(entry)) return; + folio = dax_to_folio(entry); dax_folio_put(folio); } static struct page *dax_busy_page(void *entry) { - struct folio *folio = dax_to_folio(entry); + struct folio *folio; if (dax_is_zero_entry(entry) || dax_is_empty_entry(entry)) return NULL; + folio = dax_to_folio(entry); if (folio_ref_count(folio) - folio_mapcount(folio)) return &folio->page; else From 0b803916067bcd612d9c4410cc629f446626d7b6 Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Tue, 1 Sep 2026 13:09:29 -0700 Subject: [PATCH 534/857] remove old lib/alloc_tag.c This was moved into mm/, but the original lib/ file somehow remained. Remove it. Reported-by: Suren Baghdasaryan Cc: Lorenzo Stoakes Signed-off-by: Andrew Morton --- lib/alloc_tag.c | 1029 ----------------------------------------------- 1 file changed, 1029 deletions(-) delete mode 100644 lib/alloc_tag.c diff --git a/lib/alloc_tag.c b/lib/alloc_tag.c deleted file mode 100644 index e5b218176c5afe..00000000000000 --- a/lib/alloc_tag.c +++ /dev/null @@ -1,1029 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-only -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#define ALLOCINFO_FILE_NAME "allocinfo" -#define MODULE_ALLOC_TAG_VMAP_SIZE (100000UL * sizeof(struct alloc_tag)) -#define SECTION_START(NAME) (CODETAG_SECTION_START_PREFIX NAME) -#define SECTION_STOP(NAME) (CODETAG_SECTION_STOP_PREFIX NAME) - -#ifdef CONFIG_MEM_ALLOC_PROFILING_ENABLED_BY_DEFAULT -static bool mem_profiling_support = true; -#else -static bool mem_profiling_support; -#endif - -/* - * Memory allocation profiling is permanently disabled and cannot be enabled. - * Must be called after setup_early_mem_profiling(). - */ -bool mem_alloc_profiling_permanently_disabled(void) -{ - return !mem_profiling_support; -} - -static struct codetag_type *alloc_tag_cttype; - -#ifdef CONFIG_ARCH_MODULE_NEEDS_WEAK_PER_CPU -DEFINE_PER_CPU(struct alloc_tag_counters, _shared_alloc_tag); -EXPORT_SYMBOL(_shared_alloc_tag); -#endif - -DEFINE_STATIC_KEY_MAYBE(CONFIG_MEM_ALLOC_PROFILING_ENABLED_BY_DEFAULT, - mem_alloc_profiling_key); -EXPORT_SYMBOL(mem_alloc_profiling_key); - -DEFINE_STATIC_KEY_FALSE(mem_profiling_compressed); - -struct alloc_tag_kernel_section kernel_tags = { NULL, 0 }; -unsigned long alloc_tag_ref_mask; -int alloc_tag_ref_offs; - -struct allocinfo_private { - struct codetag_iterator iter; - struct codetag_iterator reported_iter; - bool print_header; -}; - -static void *allocinfo_start(struct seq_file *m, loff_t *pos) -{ - struct allocinfo_private *priv; - loff_t node = *pos; - - priv = (struct allocinfo_private *)m->private; - codetag_lock_module_list(alloc_tag_cttype); - if (node == 0) { - priv->print_header = true; - priv->iter = codetag_get_ct_iter(alloc_tag_cttype); - } else { - priv->iter = priv->reported_iter; - } - codetag_next_ct(&priv->iter); - return priv->iter.ct ? priv : NULL; -} - -static void *allocinfo_next(struct seq_file *m, void *arg, loff_t *pos) -{ - struct allocinfo_private *priv = (struct allocinfo_private *)arg; - struct codetag *ct; - - priv->reported_iter = priv->iter; - ct = codetag_next_ct(&priv->iter); - (*pos)++; - if (!ct) - return NULL; - - return priv; -} - -static void allocinfo_stop(struct seq_file *m, void *arg) -{ - codetag_unlock_module_list(alloc_tag_cttype); -} - -static void print_allocinfo_header(struct seq_buf *buf) -{ - /* Output format version, so we can change it. */ - seq_buf_printf(buf, "allocinfo - version: 2.0\n"); - seq_buf_printf(buf, "# \n"); -} - -static void alloc_tag_to_text(struct seq_buf *out, struct codetag *ct) -{ - struct alloc_tag *tag = ct_to_alloc_tag(ct); - struct alloc_tag_counters counter = alloc_tag_read(tag); - s64 bytes = counter.bytes; - - seq_buf_printf(out, "%12lli %8llu ", bytes, counter.calls); - codetag_to_text(out, ct); - if (unlikely(alloc_tag_is_inaccurate(tag))) - seq_buf_printf(out, " accurate:no"); - seq_buf_putc(out, ' '); - seq_buf_putc(out, '\n'); -} - -static int allocinfo_show(struct seq_file *m, void *arg) -{ - struct allocinfo_private *priv = (struct allocinfo_private *)arg; - char *bufp; - size_t n = seq_get_buf(m, &bufp); - struct seq_buf buf; - - seq_buf_init(&buf, bufp, n); - if (priv->print_header) { - print_allocinfo_header(&buf); - priv->print_header = false; - } - alloc_tag_to_text(&buf, priv->iter.ct); - seq_commit(m, seq_buf_used(&buf)); - return 0; -} - -static const struct seq_operations allocinfo_seq_op = { - .start = allocinfo_start, - .next = allocinfo_next, - .stop = allocinfo_stop, - .show = allocinfo_show, -}; - -size_t alloc_tag_top_users(struct codetag_bytes *tags, size_t count, bool can_sleep) -{ - struct codetag_iterator iter; - struct codetag *ct; - struct codetag_bytes n; - unsigned int i, nr = 0; - - if (IS_ERR_OR_NULL(alloc_tag_cttype)) - return 0; - - if (can_sleep) - codetag_lock_module_list(alloc_tag_cttype); - else if (!codetag_trylock_module_list(alloc_tag_cttype)) - return 0; - - iter = codetag_get_ct_iter(alloc_tag_cttype); - while ((ct = codetag_next_ct(&iter))) { - struct alloc_tag_counters counter = alloc_tag_read(ct_to_alloc_tag(ct)); - - n.ct = ct; - n.bytes = counter.bytes; - - for (i = 0; i < nr; i++) - if (n.bytes > tags[i].bytes) - break; - - if (i < count) { - nr -= nr == count; - memmove(&tags[i + 1], - &tags[i], - sizeof(tags[0]) * (nr - i)); - nr++; - tags[i] = n; - } - } - - codetag_unlock_module_list(alloc_tag_cttype); - - return nr; -} - -void pgalloc_tag_split(struct folio *folio, int old_order, int new_order) -{ - int i; - struct alloc_tag *tag; - unsigned int nr_pages = 1 << new_order; - - if (!mem_alloc_profiling_enabled()) - return; - - tag = __pgalloc_tag_get(&folio->page); - if (!tag) - return; - - for (i = nr_pages; i < (1 << old_order); i += nr_pages) { - union pgtag_ref_handle handle; - union codetag_ref ref; - - if (get_page_tag_ref(folio_page(folio, i), &ref, &handle)) { - /* Set new reference to point to the original tag */ - alloc_tag_ref_set(&ref, tag); - update_page_tag_ref(handle, &ref); - put_page_tag_ref(handle); - } - } -} - -void pgalloc_tag_swap(struct folio *new, struct folio *old) -{ - union pgtag_ref_handle handle_old, handle_new; - union codetag_ref ref_old, ref_new; - struct alloc_tag *tag_old, *tag_new; - - if (!mem_alloc_profiling_enabled()) - return; - - tag_old = __pgalloc_tag_get(&old->page); - if (!tag_old) - return; - tag_new = __pgalloc_tag_get(&new->page); - if (!tag_new) - return; - - if (!get_page_tag_ref(&old->page, &ref_old, &handle_old)) - return; - if (!get_page_tag_ref(&new->page, &ref_new, &handle_new)) { - put_page_tag_ref(handle_old); - return; - } - - /* - * Clear tag references to avoid debug warning when using - * __alloc_tag_ref_set() with non-empty reference. - */ - set_codetag_empty(&ref_old); - set_codetag_empty(&ref_new); - - /* swap tags */ - __alloc_tag_ref_set(&ref_old, tag_new); - update_page_tag_ref(handle_old, &ref_old); - __alloc_tag_ref_set(&ref_new, tag_old); - update_page_tag_ref(handle_new, &ref_new); - - put_page_tag_ref(handle_old); - put_page_tag_ref(handle_new); -} - -static void shutdown_mem_profiling(bool remove_file) -{ - if (mem_alloc_profiling_enabled()) - static_branch_disable(&mem_alloc_profiling_key); - - if (!mem_profiling_support) - return; - - if (remove_file) - remove_proc_entry(ALLOCINFO_FILE_NAME, NULL); - mem_profiling_support = false; -} - -void __init alloc_tag_sec_init(void) -{ - struct alloc_tag *last_codetag; - - if (!mem_profiling_support) - return; - - if (!static_key_enabled(&mem_profiling_compressed)) - return; - - kernel_tags.first_tag = (struct alloc_tag *)kallsyms_lookup_name( - SECTION_START(ALLOC_TAG_SECTION_NAME)); - last_codetag = (struct alloc_tag *)kallsyms_lookup_name( - SECTION_STOP(ALLOC_TAG_SECTION_NAME)); - kernel_tags.count = last_codetag - kernel_tags.first_tag; - - /* Check if kernel tags fit into page flags */ - if (kernel_tags.count > (1UL << NR_UNUSED_PAGEFLAG_BITS)) { - shutdown_mem_profiling(false); /* allocinfo file does not exist yet */ - pr_err("%lu allocation tags cannot be references using %d available page flag bits. Memory allocation profiling is disabled!\n", - kernel_tags.count, NR_UNUSED_PAGEFLAG_BITS); - return; - } - - alloc_tag_ref_offs = (LRU_REFS_PGOFF - NR_UNUSED_PAGEFLAG_BITS); - alloc_tag_ref_mask = ((1UL << NR_UNUSED_PAGEFLAG_BITS) - 1); - pr_debug("Memory allocation profiling compression is using %d page flag bits!\n", - NR_UNUSED_PAGEFLAG_BITS); -} - -#ifdef CONFIG_MODULES - -static struct maple_tree mod_area_mt = MTREE_INIT(mod_area_mt, MT_FLAGS_ALLOC_RANGE); -static struct vm_struct *vm_module_tags; -/* A dummy object used to indicate an unloaded module */ -static struct module unloaded_mod; -/* A dummy object used to indicate a module prepended area */ -static struct module prepend_mod; - -struct alloc_tag_module_section module_tags; - -static inline unsigned long alloc_tag_align(unsigned long val) -{ - if (!static_key_enabled(&mem_profiling_compressed)) { - /* No alignment requirements when we are not indexing the tags */ - return val; - } - - if (val % sizeof(struct alloc_tag) == 0) - return val; - return ((val / sizeof(struct alloc_tag)) + 1) * sizeof(struct alloc_tag); -} - -static bool ensure_alignment(unsigned long align, unsigned int *prepend) -{ - if (!static_key_enabled(&mem_profiling_compressed)) { - /* No alignment requirements when we are not indexing the tags */ - return true; - } - - /* - * If alloc_tag size is not a multiple of required alignment, tag - * indexing does not work. - */ - if (!IS_ALIGNED(sizeof(struct alloc_tag), align)) - return false; - - /* Ensure prepend consumes multiple of alloc_tag-sized blocks */ - if (*prepend) - *prepend = alloc_tag_align(*prepend); - - return true; -} - -static inline bool tags_addressable(void) -{ - unsigned long tag_idx_count; - - if (!static_key_enabled(&mem_profiling_compressed)) - return true; /* with page_ext tags are always addressable */ - - tag_idx_count = CODETAG_ID_FIRST + kernel_tags.count + - module_tags.size / sizeof(struct alloc_tag); - - return tag_idx_count < (1UL << NR_UNUSED_PAGEFLAG_BITS); -} - -static bool needs_section_mem(struct module *mod, unsigned long size) -{ - if (!mem_profiling_support) - return false; - - return size >= sizeof(struct alloc_tag); -} - -static bool clean_unused_counters(struct alloc_tag *start_tag, - struct alloc_tag *end_tag) -{ - struct alloc_tag *tag; - bool ret = true; - - for (tag = start_tag; tag <= end_tag; tag++) { - struct alloc_tag_counters counter; - - if (!tag->counters) - continue; - - counter = alloc_tag_read(tag); - if (!counter.bytes) { - free_percpu(tag->counters); - tag->counters = NULL; - } else { - ret = false; - } - } - - return ret; -} - -/* Called with mod_area_mt locked */ -static void clean_unused_module_areas_locked(void) -{ - MA_STATE(mas, &mod_area_mt, 0, module_tags.size); - struct module *val; - - mas_for_each(&mas, val, module_tags.size) { - struct alloc_tag *start_tag; - struct alloc_tag *end_tag; - - if (val != &unloaded_mod) - continue; - - /* Release area if all tags are unused */ - start_tag = (struct alloc_tag *)(module_tags.start_addr + mas.index); - end_tag = (struct alloc_tag *)(module_tags.start_addr + mas.last); - if (clean_unused_counters(start_tag, end_tag)) - mas_erase(&mas); - } -} - -/* Called with mod_area_mt locked */ -static bool find_aligned_area(struct ma_state *mas, unsigned long section_size, - unsigned long size, unsigned int prepend, unsigned long align) -{ - bool cleanup_done = false; - -repeat: - /* Try finding exact size and hope the start is aligned */ - if (!mas_empty_area(mas, 0, section_size - 1, prepend + size)) { - if (IS_ALIGNED(mas->index + prepend, align)) - return true; - - /* Try finding larger area to align later */ - mas_reset(mas); - if (!mas_empty_area(mas, 0, section_size - 1, - size + prepend + align - 1)) - return true; - } - - /* No free area, try cleanup stale data and repeat the search once */ - if (!cleanup_done) { - clean_unused_module_areas_locked(); - cleanup_done = true; - mas_reset(mas); - goto repeat; - } - - return false; -} - -static int vm_module_tags_populate(void) -{ - unsigned long phys_end = ALIGN_DOWN(module_tags.start_addr, PAGE_SIZE) + - (vm_module_tags->nr_pages << PAGE_SHIFT); - unsigned long new_end = module_tags.start_addr + module_tags.size; - - if (phys_end < new_end) { - struct page **next_page = vm_module_tags->pages + vm_module_tags->nr_pages; - unsigned long old_shadow_end = ALIGN(phys_end, MODULE_ALIGN); - unsigned long new_shadow_end = ALIGN(new_end, MODULE_ALIGN); - unsigned long more_pages; - unsigned long nr = 0; - - more_pages = ALIGN(new_end - phys_end, PAGE_SIZE) >> PAGE_SHIFT; - while (nr < more_pages) { - unsigned long allocated; - - allocated = alloc_pages_bulk_node(GFP_KERNEL | __GFP_NOWARN, - NUMA_NO_NODE, more_pages - nr, next_page + nr); - - if (!allocated) - break; - nr += allocated; - } - - if (nr < more_pages || - vmap_pages_range(phys_end, phys_end + (nr << PAGE_SHIFT), PAGE_KERNEL, - next_page, PAGE_SHIFT) < 0) { - release_pages_arg arg = { .pages = next_page }; - - /* Clean up and error out */ - release_pages(arg, nr); - return -ENOMEM; - } - - vm_module_tags->nr_pages += nr; - - /* - * Kasan allocates 1 byte of shadow for every 8 bytes of data. - * When kasan_alloc_module_shadow allocates shadow memory, - * its unit of allocation is a page. - * Therefore, here we need to align to MODULE_ALIGN. - */ - if (old_shadow_end < new_shadow_end) - kasan_alloc_module_shadow((void *)old_shadow_end, - new_shadow_end - old_shadow_end, - GFP_KERNEL); - } - - /* - * Mark the pages as accessible, now that they are mapped. - * With hardware tag-based KASAN, marking is skipped for - * non-VM_ALLOC mappings, see __kasan_unpoison_vmalloc(). - */ - kasan_unpoison_vmalloc((void *)module_tags.start_addr, - new_end - module_tags.start_addr, - KASAN_VMALLOC_PROT_NORMAL); - - return 0; -} - -static void *reserve_module_tags(struct module *mod, unsigned long size, - unsigned int prepend, unsigned long align) -{ - unsigned long section_size = module_tags.end_addr - module_tags.start_addr; - MA_STATE(mas, &mod_area_mt, 0, section_size - 1); - unsigned long offset; - void *ret = NULL; - - /* If no tags return error */ - if (size < sizeof(struct alloc_tag)) - return ERR_PTR(-EINVAL); - - /* - * align is always power of 2, so we can use IS_ALIGNED and ALIGN. - * align 0 or 1 means no alignment, to simplify set to 1. - */ - if (!align) - align = 1; - - if (!ensure_alignment(align, &prepend)) { - shutdown_mem_profiling(true); - pr_err("%s: alignment %lu is incompatible with allocation tag indexing. Memory allocation profiling is disabled!\n", - mod->name, align); - return ERR_PTR(-EINVAL); - } - - mas_lock(&mas); - if (!find_aligned_area(&mas, section_size, size, prepend, align)) { - ret = ERR_PTR(-ENOMEM); - goto unlock; - } - - /* Mark found area as reserved */ - offset = mas.index; - offset += prepend; - offset = ALIGN(offset, align); - if (offset != mas.index) { - unsigned long pad_start = mas.index; - - mas.last = offset - 1; - mas_store(&mas, &prepend_mod); - if (mas_is_err(&mas)) { - ret = ERR_PTR(xa_err(mas.node)); - goto unlock; - } - mas.index = offset; - mas.last = offset + size - 1; - mas_store(&mas, mod); - if (mas_is_err(&mas)) { - mas.index = pad_start; - mas_erase(&mas); - ret = ERR_PTR(xa_err(mas.node)); - } - } else { - mas.last = offset + size - 1; - mas_store(&mas, mod); - if (mas_is_err(&mas)) - ret = ERR_PTR(xa_err(mas.node)); - } -unlock: - mas_unlock(&mas); - - if (IS_ERR(ret)) - return ret; - - if (module_tags.size < offset + size) { - int grow_res; - - module_tags.size = offset + size; - if (mem_alloc_profiling_enabled() && !tags_addressable()) { - shutdown_mem_profiling(true); - pr_warn("With module %s there are too many tags to fit in %d page flag bits. Memory allocation profiling is disabled!\n", - mod->name, NR_UNUSED_PAGEFLAG_BITS); - } - - grow_res = vm_module_tags_populate(); - if (grow_res) { - shutdown_mem_profiling(true); - pr_err("Failed to allocate memory for allocation tags in the module %s. Memory allocation profiling is disabled!\n", - mod->name); - return ERR_PTR(grow_res); - } - } - - return (struct alloc_tag *)(module_tags.start_addr + offset); -} - -static void release_module_tags(struct module *mod, bool used) -{ - MA_STATE(mas, &mod_area_mt, module_tags.size, module_tags.size); - struct alloc_tag *start_tag; - struct alloc_tag *end_tag; - struct module *val; - - mas_lock(&mas); - mas_for_each_rev(&mas, val, 0) - if (val == mod) - break; - - if (!val) /* module not found */ - goto out; - - if (!used) - goto release_area; - - start_tag = (struct alloc_tag *)(module_tags.start_addr + mas.index); - end_tag = (struct alloc_tag *)(module_tags.start_addr + mas.last); - if (!clean_unused_counters(start_tag, end_tag)) { - struct alloc_tag *tag; - - for (tag = start_tag; tag <= end_tag; tag++) { - struct alloc_tag_counters counter; - - if (!tag->counters) - continue; - - counter = alloc_tag_read(tag); - pr_info("%s:%u module %s func:%s has %llu allocated at module unload\n", - tag->ct.filename, tag->ct.lineno, tag->ct.modname, - tag->ct.function, counter.bytes); - } - } else { - used = false; - } -release_area: - mas_store(&mas, used ? &unloaded_mod : NULL); - val = mas_prev_range(&mas, 0); - if (val == &prepend_mod) - mas_store(&mas, NULL); -out: - mas_unlock(&mas); -} - -static int load_module(struct module *mod, struct codetag *start, struct codetag *stop) -{ - /* Allocate module alloc_tag percpu counters */ - struct alloc_tag *start_tag; - struct alloc_tag *stop_tag; - struct alloc_tag *tag; - - /* percpu counters for core allocations are already statically allocated */ - if (!mod) - return 0; - - start_tag = ct_to_alloc_tag(start); - stop_tag = ct_to_alloc_tag(stop); - for (tag = start_tag; tag < stop_tag; tag++) { - WARN_ON(tag->counters); - tag->counters = alloc_percpu(struct alloc_tag_counters); - if (!tag->counters) { - while (--tag >= start_tag) { - free_percpu(tag->counters); - tag->counters = NULL; - } - pr_err("Failed to allocate memory for allocation tag percpu counters in the module %s\n", - mod->name); - return -ENOMEM; - } - - /* - * Avoid a kmemleak false positive. The pointer to the counters is stored - * in the alloc_tag section of the module and cannot be directly accessed. - */ - kmemleak_ignore_percpu(tag->counters); - } - return 0; -} - -static void replace_module(struct module *mod, struct module *new_mod) -{ - MA_STATE(mas, &mod_area_mt, 0, module_tags.size); - struct module *val; - - mas_lock(&mas); - mas_for_each(&mas, val, module_tags.size) { - if (val != mod) - continue; - - mas_store_gfp(&mas, new_mod, GFP_KERNEL); - break; - } - mas_unlock(&mas); -} - -static int __init alloc_mod_tags_mem(void) -{ - /* Map space to copy allocation tags */ - vm_module_tags = execmem_vmap(MODULE_ALLOC_TAG_VMAP_SIZE); - if (!vm_module_tags) { - pr_err("Failed to map %lu bytes for module allocation tags\n", - MODULE_ALLOC_TAG_VMAP_SIZE); - module_tags.start_addr = 0; - return -ENOMEM; - } - - vm_module_tags->pages = kmalloc_objs(struct page *, - get_vm_area_size(vm_module_tags) >> PAGE_SHIFT, - GFP_KERNEL | __GFP_ZERO); - if (!vm_module_tags->pages) { - free_vm_area(vm_module_tags); - return -ENOMEM; - } - - module_tags.start_addr = (unsigned long)vm_module_tags->addr; - module_tags.end_addr = module_tags.start_addr + MODULE_ALLOC_TAG_VMAP_SIZE; - /* Ensure the base is alloc_tag aligned when required for indexing */ - module_tags.start_addr = alloc_tag_align(module_tags.start_addr); - - return 0; -} - -static void __init free_mod_tags_mem(void) -{ - release_pages_arg arg = { .pages = vm_module_tags->pages }; - - module_tags.start_addr = 0; - release_pages(arg, vm_module_tags->nr_pages); - kfree(vm_module_tags->pages); - free_vm_area(vm_module_tags); -} - -#else /* CONFIG_MODULES */ - -static inline int alloc_mod_tags_mem(void) { return 0; } -static inline void free_mod_tags_mem(void) {} - -#endif /* CONFIG_MODULES */ - -/* See: Documentation/mm/allocation-profiling.rst */ -static int __init setup_early_mem_profiling(char *str) -{ - bool compressed = false; - bool enable; - - if (!str || !str[0]) - return -EINVAL; - - if (!strncmp(str, "never", 5)) { - enable = false; - mem_profiling_support = false; - pr_info("Memory allocation profiling is disabled!\n"); - } else { - char *token = strsep(&str, ","); - - if (kstrtobool(token, &enable)) - return -EINVAL; - - if (str) { - - if (strcmp(str, "compressed")) - return -EINVAL; - - compressed = true; - } - mem_profiling_support = true; - pr_info("Memory allocation profiling is enabled %s compression and is turned %s!\n", - compressed ? "with" : "without", str_on_off(enable)); - } - - if (enable != mem_alloc_profiling_enabled()) { - if (enable) - static_branch_enable(&mem_alloc_profiling_key); - else - static_branch_disable(&mem_alloc_profiling_key); - } - if (compressed != static_key_enabled(&mem_profiling_compressed)) { - if (compressed) - static_branch_enable(&mem_profiling_compressed); - else - static_branch_disable(&mem_profiling_compressed); - } - - return 0; -} -early_param("sysctl.vm.mem_profiling", setup_early_mem_profiling); - -static __init bool need_page_alloc_tagging(void) -{ - if (static_key_enabled(&mem_profiling_compressed)) - return false; - - return mem_profiling_support; -} - -#ifdef CONFIG_MEM_ALLOC_PROFILING_DEBUG -/* - * Track page allocations before page_ext is initialized. - * Some pages are allocated before page_ext becomes available, leaving - * their codetag uninitialized. Track these early PFNs so we can clear - * their codetag refs later to avoid warnings when they are freed. - * - * Each page is cast to a pfn_pool: the first few bytes hold metadata - * (next pointer and slot count), the remainder stores PFNs. - */ -struct pfn_pool { - struct pfn_pool *next; - atomic_t count; - unsigned long pfns[]; -}; - -#define PFN_POOL_SIZE ((PAGE_SIZE - offsetof(struct pfn_pool, pfns)) / \ - sizeof(unsigned long)) - -/* - * Skip early PFN recording for a page allocation. Reuses the - * %__GFP_NO_OBJ_EXT bit. Used by __alloc_tag_add_early_pfn() to avoid - * recursion when allocating pages for the early PFN tracking list - * itself. - * - * Codetags of the pages allocated with __GFP_NO_CODETAG should be - * cleared (via clear_page_tag_ref()) before freeing the pages to prevent - * alloc_tag_sub_check() from triggering a warning. - */ -#define __GFP_NO_CODETAG __GFP_NO_OBJ_EXT - -static struct pfn_pool *current_pfn_pool __initdata; - -static void __init __alloc_tag_add_early_pfn(unsigned long pfn) -{ - struct pfn_pool *pool; - int idx; - - do { - pool = READ_ONCE(current_pfn_pool); - if (!pool || atomic_read(&pool->count) >= PFN_POOL_SIZE) { - struct page *new_page = alloc_page(__GFP_HIGH | __GFP_NO_CODETAG); - struct pfn_pool *new; - - if (!new_page) { - pr_warn_once("early PFN tracking page allocation failed\n"); - return; - } - new = page_address(new_page); - new->next = pool; - atomic_set(&new->count, 0); - if (cmpxchg(¤t_pfn_pool, pool, new) != pool) { - clear_page_tag_ref(new_page); - __free_page(new_page); - continue; - } - pool = new; - } - idx = atomic_read(&pool->count); - if (idx >= PFN_POOL_SIZE) - continue; - if (atomic_cmpxchg(&pool->count, idx, idx + 1) == idx) - break; - } while (1); - - pool->pfns[idx] = pfn; -} - -typedef void alloc_tag_add_func(unsigned long pfn); -static alloc_tag_add_func __rcu *alloc_tag_add_early_pfn_ptr __refdata = - RCU_INITIALIZER(__alloc_tag_add_early_pfn); - -void alloc_tag_add_early_pfn(unsigned long pfn, gfp_t gfp_flags) -{ - alloc_tag_add_func *alloc_tag_add; - - if (static_key_enabled(&mem_profiling_compressed)) - return; - - /* Skip allocations for the tracking list itself to avoid recursion. */ - if (gfp_flags & __GFP_NO_CODETAG) - return; - - rcu_read_lock(); - alloc_tag_add = rcu_dereference(alloc_tag_add_early_pfn_ptr); - if (alloc_tag_add) - alloc_tag_add(pfn); - rcu_read_unlock(); -} - -static void __init clear_early_alloc_pfn_tag_refs(void) -{ - struct pfn_pool *pool, *next; - struct page *page; - int i; - - if (static_key_enabled(&mem_profiling_compressed)) - return; - - rcu_assign_pointer(alloc_tag_add_early_pfn_ptr, NULL); - /* Make sure we are not racing with __alloc_tag_add_early_pfn() */ - synchronize_rcu(); - - for (pool = current_pfn_pool; pool; pool = next) { - int nr_pfns = atomic_read(&pool->count); - - for (i = 0; i < nr_pfns; i++) { - unsigned long pfn = pool->pfns[i]; - - if (pfn_valid(pfn)) { - union pgtag_ref_handle handle; - union codetag_ref ref; - - if (get_page_tag_ref(pfn_to_page(pfn), &ref, &handle)) { - /* - * An early-allocated page could be freed and reallocated - * after its page_ext is initialized but before we clear it. - * In that case, it already has a valid tag set. - * We should not overwrite that valid tag - * with CODETAG_EMPTY. - * - * Note: there is still a small race window between checking - * ref.ct and calling set_codetag_empty(). We accept this - * race as it's unlikely and the extra complexity of atomic - * cmpxchg is not worth it for this debug-only code path. - */ - if (ref.ct) { - put_page_tag_ref(handle); - continue; - } - - set_codetag_empty(&ref); - update_page_tag_ref(handle, &ref); - put_page_tag_ref(handle); - } - } - } - - next = pool->next; - page = virt_to_page(pool); - clear_page_tag_ref(page); - __free_page(page); - } -} -#else /* !CONFIG_MEM_ALLOC_PROFILING_DEBUG */ -static inline void __init clear_early_alloc_pfn_tag_refs(void) {} -#endif /* CONFIG_MEM_ALLOC_PROFILING_DEBUG */ - -static __init void init_page_alloc_tagging(void) -{ - clear_early_alloc_pfn_tag_refs(); -} - -struct page_ext_operations page_alloc_tagging_ops = { - .size = sizeof(union codetag_ref), - .need = need_page_alloc_tagging, - .init = init_page_alloc_tagging, -}; -EXPORT_SYMBOL(page_alloc_tagging_ops); - -#ifdef CONFIG_SYSCTL -/* - * Not using proc_do_static_key() directly to prevent enabling profiling - * after it was shut down. - */ -static int proc_mem_profiling_handler(const struct ctl_table *table, int write, - void *buffer, size_t *lenp, loff_t *ppos) -{ - if (write) { - /* - * Call from do_sysctl_args() which is a no-op since the same - * value was already set by setup_early_mem_profiling. - * Return success to avoid warnings from do_sysctl_args(). - */ - if (!current->mm) - return 0; - -#ifdef CONFIG_MEM_ALLOC_PROFILING_DEBUG - /* User can't toggle profiling while debugging */ - return -EACCES; -#endif - if (!mem_profiling_support) - return -EINVAL; - } - - return proc_do_static_key(table, write, buffer, lenp, ppos); -} - - -static const struct ctl_table memory_allocation_profiling_sysctls[] = { - { - .procname = "mem_profiling", - .data = &mem_alloc_profiling_key, - .mode = 0644, - .proc_handler = proc_mem_profiling_handler, - }, -}; - -static void __init sysctl_init(void) -{ - register_sysctl_init("vm", memory_allocation_profiling_sysctls); -} -#else /* CONFIG_SYSCTL */ -static inline void sysctl_init(void) {} -#endif /* CONFIG_SYSCTL */ - -static int __init alloc_tag_init(void) -{ - const struct codetag_type_desc desc = { - .section = ALLOC_TAG_SECTION_NAME, - .tag_size = sizeof(struct alloc_tag), -#ifdef CONFIG_MODULES - .needs_section_mem = needs_section_mem, - .alloc_section_mem = reserve_module_tags, - .free_section_mem = release_module_tags, - .module_load = load_module, - .module_replaced = replace_module, -#endif - }; - int res; - - sysctl_init(); - - if (!mem_profiling_support) { - pr_info("Memory allocation profiling is not supported!\n"); - return 0; - } - - if (!proc_create_seq_private(ALLOCINFO_FILE_NAME, 0400, NULL, &allocinfo_seq_op, - sizeof(struct allocinfo_private), NULL)) { - pr_err("Failed to create %s file\n", ALLOCINFO_FILE_NAME); - shutdown_mem_profiling(false); - return -ENOMEM; - } - - res = alloc_mod_tags_mem(); - if (res) { - pr_err("Failed to reserve address space for module tags, errno = %d\n", res); - shutdown_mem_profiling(true); - return res; - } - - alloc_tag_cttype = codetag_register_type(&desc); - if (IS_ERR(alloc_tag_cttype)) { - pr_err("Allocation tags registration failed, errno = %pe\n", alloc_tag_cttype); - free_mod_tags_mem(); - shutdown_mem_profiling(true); - return PTR_ERR(alloc_tag_cttype); - } - - return 0; -} -module_init(alloc_tag_init); From 76d412c2caf4e80a00f0f28709e43b1cd9feca83 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Tue, 1 Sep 2026 11:01:09 -0700 Subject: [PATCH 535/857] mm/mlock: use the IRQ-safe accessor for NR_MLOCK in __munlock_folio() NR_MLOCK is updated from interrupt context. __free_pages_prepare() clears a stray PG_mlocked and adjusts NR_MLOCK, and a folio can reach it with the flag still set from a bio completion handler: __free_pages_ok+0x6af/0x7a0 __bio_release_pages+0xde/0x260 __iomap_dio_bio_end_io+0x16e/0x1a0 blk_update_request+0x14b/0x3d0 blk_mq_end_request+0x18/0x30 blk_done_softirq+0x49/0x60 The folio gets there like this. A MAP_SHARED file mapping is mlocked, so its page cache folios carry PG_mlocked, and an O_DIRECT write sourced from that mapping GUP-pins those same folios. munlock() then runs mlock_vma_pages_range(), which clears VM_LOCKED before walking the page tables to munlock each folio. A concurrent hole punch reaches the folio through the rmap (i_mmap_rwsem, not mmap_lock) and can land inside that window: __folio_remove_rmap() -> munlock_vma_folio() sees VM_LOCKED already clear, so it neither queues the folio on the mlock batch nor takes a reference, and the pte it clears makes the pending mlock_pte_range() walk skip the folio at its !pte_present() check. filemap_remove_folio() then drops the page cache reference, leaving the bio's pin as the last one, released from the completion handler above. So __zone_stat_mod_folio() here needs interrupts disabled, not merely preemption, and __munlock_folio() has a path where they are not: when the folio has already been taken off the LRU by somebody else the function jumps straight to the counter update without taking the lruvec lock. The read-modify-write of the per-CPU NR_MLOCK diff can then be interrupted by the softirq above, and one of the two decrements is lost, leaving Mlocked in /proc/meminfo permanently overstated. Use zone_stat_mod_folio(). mod_zone_state()'s this_cpu_try_cmpxchg() is atomic against a same-CPU interrupt and retries, and on the path where the lruvec lock is held its cost is negligible next to the lock itself. The UNEVICTABLE_PG* events are deliberately left on the __ accessors: they occupy different vm_event_states slots from the UNEVICTABLE_PGCLEARED that __free_pages_prepare() bumps, and nothing updates those two from interrupt context. Link: https://lore.kernel.org/20260901180109.3797944-1-shakeel.butt@linux.dev Fixes: 2fbb0c10d1e8 ("mm/munlock: mlock_page() munlock_page() batch by pagevec") Signed-off-by: Shakeel Butt Reported-by: syzbot+cd2073ee6d958a8d0fcd@syzkaller.appspotmail.com Closes: https://lore.kernel.org/linux-mm/6a931c5a.08e933ee.dbf97.0093.GAE@google.com/ Acked-by: Hugh Dickins Cc: Jann Horn Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Matthew Wilcox (Oracle) Cc: Pedro Falcato Cc: Vlastimil Babka Cc: Signed-off-by: Andrew Morton --- mm/mlock.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/mlock.c b/mm/mlock.c index efa6716e4dfbdf..39215a3eab1fbf 100644 --- a/mm/mlock.c +++ b/mm/mlock.c @@ -141,7 +141,7 @@ static struct lruvec *__munlock_folio(struct folio *folio, struct lruvec *lruvec munlock: if (folio_test_clear_mlocked(folio)) { - __zone_stat_mod_folio(folio, NR_MLOCK, -nr_pages); + zone_stat_mod_folio(folio, NR_MLOCK, -nr_pages); if (isolated || !folio_test_unevictable(folio)) __count_vm_events(UNEVICTABLE_PGMUNLOCKED, nr_pages); else From a0ae2d452e820432c318b8fe9b8aacbae55f5d61 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Wed, 2 Sep 2026 19:08:08 +0100 Subject: [PATCH 536/857] mm/vma: correctly unaccount on mmap_prepare() failure __mmap_setup() accounts memory for relevant mappings via: security_vm_enough_memory_mm() -> __vm_enough_memory() -> vm_acct_memory() If __mmap_setup() fails, this indicates that this accounting did not take place, and thus it's appropriate for __mmap_region() to jump to abort_munmap. However if call_mmap_prepare() fails, it also jumps there and any accounted memory is not correctly unaccounted. Fix this by handling each error separately. Link: https://lore.kernel.org/20260902-fix-unaccount-mmap_prepare-v1-1-ea070189fdfb@kernel.org Fixes: c84bf6dd2b83 ("mm: introduce new .mmap_prepare() file callback") Signed-off-by: Lorenzo Stoakes (ARM) Cc: Jann Horn Cc: Liam R. Howlett Cc: Pedro Falcato Cc: Vlastimil Babka Cc: Signed-off-by: Andrew Morton --- mm/vma.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/mm/vma.c b/mm/vma.c index 35e7a64855fadc..f29abb30956bb9 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2859,10 +2859,12 @@ static unsigned long __mmap_region(struct file *file, unsigned long addr, map.check_ksm_early = can_set_ksm_flags_early(&map); error = __mmap_setup(&map, &desc, uf); - if (!error && have_mmap_prepare) - error = call_mmap_prepare(&map, &desc); if (error) goto abort_munmap; + if (have_mmap_prepare) + error = call_mmap_prepare(&map, &desc); + if (error) + goto unacct_error; if (map.check_ksm_early) update_ksm_flags(&map); From 8ef8543b6951d75c6cc092966cf40f7ef4d9ce36 Mon Sep 17 00:00:00 2001 From: Jiayuan Chen Date: Wed, 2 Sep 2026 15:37:59 +0800 Subject: [PATCH 537/857] mm/shrinker: fix bogus set_shrinker_bit() with cgroup.memory=nokmem With cgroup.memory=nokmem, shrinker_memcg_alloc() bails out early and never allocates an id, so shrinker->id keeps the 0 it got from the kzalloc() in shrinker_alloc(). __list_lru_init() then copies that 0 into lru->shrinker_id, where it looks like a valid bit index. Nothing calls expand_shrinker_info() on nokmem either, so shrinker_nr_max stays 0 and every memcg ends up with an empty map (map_nr_max == 0). deferred_split_folio() hands a real memcg to __list_lru_add() regardless of whether the lru is memcg aware, so the first THP queued in a cgroup does set_shrinker_bit(memcg, nid, 0) and trips the bounds check: WARNING: mm/shrinker.c:212 at set_shrinker_bit+0x7d/0x90, CPU#126 Call Trace: deferred_split_folio+0x18c/0x220 map_anon_folio_pmd_nopf+0xdd/0x130 map_anon_folio_pmd_pf+0x14/0xb0 do_huge_pmd_anonymous_page+0x1a1/0x620 __handle_mm_fault+0xea9/0x10d0 handle_mm_fault+0xe5/0x320 do_user_addr_fault+0x1cc/0x870 exc_page_fault+0x81/0x1b0 asm_exc_page_fault+0x27/0x30 Harmless, the WARN_ON_ONCE() is what keeps the out of bounds unit[] read from happening, but the id should not look valid in the first place. Clear it before returning. Two other spots could paper over this: drop the id in __list_lru_init() when nokmem turns memcg_aware off, or make deferred_split_folio() pass NULL like list_lru_add_obj() does. Both leave shrinker->id lying around for the next caller, so fix it where the id is handed out. Link: https://lore.kernel.org/20260902073800.305481-1-jiayuan.chen@linux.dev Fixes: fafaeceb89a5 ("mm: switch deferred split shrinker to list_lru") Signed-off-by: Jiayuan Chen Acked-by: Shakeel Butt Cc: Usama Arif Cc: Dave Chinner Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: Signed-off-by: Andrew Morton --- mm/shrinker.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/shrinker.c b/mm/shrinker.c index a70aab124a0e7f..7ec2a9704f6f2f 100644 --- a/mm/shrinker.c +++ b/mm/shrinker.c @@ -227,6 +227,8 @@ static int shrinker_memcg_alloc(struct shrinker *shrinker) { int id; + shrinker->id = -1; + if (mem_cgroup_disabled()) return -ENOSYS; if (mem_cgroup_kmem_disabled() && !(shrinker->flags & SHRINKER_NONSLAB)) From e2c9fc9641d29fec99fd7353019499b0ef1ae95d Mon Sep 17 00:00:00 2001 From: Ackerley Tng Date: Tue, 1 Sep 2026 20:38:40 -0700 Subject: [PATCH 538/857] mm/folio: EXPORT_SYMBOL_FOR_KVM(lru_cache_drain_for_folio) To simplify independent development in the KVM and MM subsystems, now export to KVM the lru_cache_drain_for_folio() which MM added in 7.3-rc1. Link: https://lore.kernel.org/lkml/bd6c9c74-e374-a9d3-ba1f-8b6f430894fc@google.com/T/#u Link: https://lore.kernel.org/02876cea-5727-2ca4-bead-73659ea6fec4@google.com Signed-off-by: Ackerley Tng Signed-off-by: Hugh Dickins Acked-by: Vlastimil Babka (SUSE) Suggested-by: David Hildenbrand Reviewed-by: Fuad Tabba Reviewed-by: Binbin Wu Cc: Matthew Wilcox (Oracle) Cc: Sean Christopherson Signed-off-by: Andrew Morton --- mm/folio.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/folio.c b/mm/folio.c index c02dcea9c03c24..50a6dbe55998e7 100644 --- a/mm/folio.c +++ b/mm/folio.c @@ -33,6 +33,7 @@ #include #include #include +#include #include "internal.h" #include "page_alloc.h" @@ -926,6 +927,7 @@ void lru_cache_drain_for_folio(const struct folio *folio, *drained = LRU_CACHE_DRAINED_ALL; } } +EXPORT_SYMBOL_FOR_KVM(lru_cache_drain_for_folio); atomic_t lru_disable_count = ATOMIC_INIT(0); From bafa373b8bf4c3fae126cd380c2eb55014b1d36b Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Wed, 19 Aug 2026 17:51:43 +0800 Subject: [PATCH 539/857] mm: use a folio in the softleaf_is_device_private path Use the folio APIs in the device_private migration path of do_swap_page(), replacing four calls to compound_head() with two page_folio() calls. The second one re-fetches the folio from vmf->page after migrate_to_ram(), which might have split the folio. Link: https://lore.kernel.org/20260819095144.45660-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Suggested-by: David Hildenbrand (Arm) Link: https://lore.kernel.org/all/e20678ed-3fa1-4677-a1d7-e2af481e8302@kernel.org/ Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Anshuman Khandual Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/memory.c | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 8b0c2c735d3de7..9cbce5c90bffde 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4926,18 +4926,21 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) goto unlock; /* - * Get a page reference while we know the page can't be - * freed. + * Get a folio reference while we know the folio can't + * be freed. */ - if (trylock_page(vmf->page)) { + folio = page_folio(vmf->page); + if (folio_trylock(folio)) { struct dev_pagemap *pgmap; - get_page(vmf->page); + folio_get(folio); pte_unmap_unlock(vmf->pte, vmf->ptl); pgmap = page_pgmap(vmf->page); ret = pgmap->ops->migrate_to_ram(vmf); - unlock_page(vmf->page); - put_page(vmf->page); + /* migrate_to_ram() might have split the folio. */ + folio = page_folio(vmf->page); + folio_unlock(folio); + folio_put(folio); } else { pte_unmap(vmf->pte); softleaf_entry_wait_on_locked(entry, vmf->ptl); From b190a5ed2798c48bc96ba8d17672a05b7ff914d5 Mon Sep 17 00:00:00 2001 From: Qi Xi Date: Wed, 19 Aug 2026 16:20:52 +0800 Subject: [PATCH 540/857] mm: drop stale MAX_ORDER references The treewide rename in commit 5e0a760b4441 ("mm, treewide: rename MAX_ORDER to MAX_PAGE_ORDER") left a few spots still using the old name: - two comments in include/net/mana/mana.h and mm/page_alloc.c; - the gdb helper scripts/gdb/linux/mm.py, where self.MAX_ORDER is a local mirror of the kernel's MAX_ORDER define. Rename the leftover instances to MAX_PAGE_ORDER so the tree is consistent. No functional changes. Link: https://lore.kernel.org/20260819082052.3338603-1-xiqi2@huawei.com Signed-off-by: Qi Xi Reviewed-by: Zi Yan Cc: Jan Kiszka Cc: Johannes Weiner Cc: Kefeng Wang Cc: Kieran Bingham Cc: Konstantin Taranov Cc: Long Li Cc: Michal Hocko Cc: Nanyong Sun Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/net/mana/mana.h | 4 ++-- mm/page_alloc.c | 2 +- scripts/gdb/linux/mm.py | 8 ++++---- 3 files changed, 7 insertions(+), 7 deletions(-) diff --git a/include/net/mana/mana.h b/include/net/mana/mana.h index 83b7eff4646ead..e8fda092be37fb 100644 --- a/include/net/mana/mana.h +++ b/include/net/mana/mana.h @@ -47,8 +47,8 @@ enum mana_priv_flag_bits { #define COMP_ENTRY_SIZE 64 /* This Max value for RX buffers is derived from __alloc_page()'s max page - * allocation calculation. It allows maximum 2^(MAX_ORDER -1) pages. RX buffer - * size beyond this value gets rejected by __alloc_page() call. + * allocation calculation. It allows maximum 2^MAX_PAGE_ORDER pages. RX + * buffer size beyond this value gets rejected by __alloc_page() call. */ #define MAX_RX_BUFFERS_PER_QUEUE 8192 #define DEF_RX_BUFFERS_PER_QUEUE 1024 diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 12fac9084c483d..ab385bc252ccc0 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -7978,7 +7978,7 @@ static bool cond_accept_memory(struct zone *zone, unsigned int order, /* * Watermarks have not been initialized yet. * - * Accepting one MAX_ORDER page to ensure progress. + * Accepting one MAX_PAGE_ORDER page to ensure progress. */ if (!wmark) return try_to_accept_memory_one(zone); diff --git a/scripts/gdb/linux/mm.py b/scripts/gdb/linux/mm.py index dffadccbb01d24..28d33624c38bd6 100644 --- a/scripts/gdb/linux/mm.py +++ b/scripts/gdb/linux/mm.py @@ -56,7 +56,7 @@ def __init__(self): self.MAX_PHYSMEM_BITS = 46 self.SECTION_SIZE_BITS = 27 - self.MAX_ORDER = 10 + self.MAX_PAGE_ORDER = 10 self.SECTIONS_SHIFT = self.MAX_PHYSMEM_BITS - self.SECTION_SIZE_BITS self.NR_MEM_SECTIONS = 1 << self.SECTIONS_SHIFT @@ -233,11 +233,11 @@ def __init__(self): self.SECTIONS_SHIFT = self.MAX_PHYSMEM_BITS - self.SECTION_SIZE_BITS if str(constants.LX_CONFIG_ARCH_FORCE_MAX_ORDER).isdigit(): - self.MAX_ORDER = constants.LX_CONFIG_ARCH_FORCE_MAX_ORDER + self.MAX_PAGE_ORDER = constants.LX_CONFIG_ARCH_FORCE_MAX_ORDER else: - self.MAX_ORDER = 10 + self.MAX_PAGE_ORDER = 10 - self.MAX_ORDER_NR_PAGES = 1 << (self.MAX_ORDER) + self.MAX_ORDER_NR_PAGES = 1 << (self.MAX_PAGE_ORDER) self.PFN_SECTION_SHIFT = self.SECTION_SIZE_BITS - self.PAGE_SHIFT self.NR_MEM_SECTIONS = 1 << self.SECTIONS_SHIFT self.PAGES_PER_SECTION = 1 << self.PFN_SECTION_SHIFT From 49a9594a61ae7a9bd7c0e9b51c10a094a41ae486 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Wed, 26 Aug 2026 20:44:09 +0800 Subject: [PATCH 541/857] mm/vmscan: drop the combined limit gate in __node_reclaim() __node_reclaim() is called from two paths: node_reclaim() and user_proactive_reclaim(). node_reclaim() already bails out early unless node_pagecache_reclaimable() is over pgdat->min_unmapped_pages or the reclaimable slab is over pgdat->min_slab_pages. The identical check inside __node_reclaim() that guards the shrink_node() loop is therefore redundant for this path. user_proactive_reclaim() is proactive reclaim driven by userspace and should not be gated by the per-node min_unmapped_pages / min_slab_pages limits at all [1]. With the gate in place, a proactive request is silently turned into a no-op whenever the node happens to sit below both thresholds. Drop the gate in __node_reclaim() and always run the shrink_node() loop. The node_reclaim() path is unchanged, since its caller has already applied the same test; the proactive path is no longer wrongly gated. Link: https://lore.kernel.org/20260826124409.35569-1-ridong.chen@linux.dev Link: https://sashiko.dev/#/patchset/20260723045718.2052070-1-ridong.chen@linux.dev [1] Fixes: b980077899ea ("mm: introduce per-node proactive reclaim interface") Assisted-by: Claude:claude-opus-4-8 Acked-by: Johannes Weiner Signed-off-by: Ridong Chen Acked-by: Michal Hocko Acked-by: Shakeel Butt Acked-by: Davidlohr Bueso Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Roman Gushchin Cc: Wei Xu Cc: Yuanchu Xie Signed-off-by: Andrew Morton --- mm/vmscan.c | 13 +++---------- 1 file changed, 3 insertions(+), 10 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index f11491ee9ed5c1..6dff207ad8c612 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -7851,16 +7851,9 @@ static unsigned long __node_reclaim(struct pglist_data *pgdat, noreclaim_flag = memalloc_noreclaim_save(); set_task_reclaim_state(p, &sc->reclaim_state); - if (node_pagecache_reclaimable(pgdat) > pgdat->min_unmapped_pages || - node_page_state_pages(pgdat, NR_SLAB_RECLAIMABLE_B) > pgdat->min_slab_pages) { - /* - * Free memory by calling shrink node with increasing - * priorities until we have enough memory freed. - */ - do { - shrink_node(pgdat, sc); - } while (sc->nr_reclaimed < nr_pages && --sc->priority >= 0); - } + do { + shrink_node(pgdat, sc); + } while (sc->nr_reclaimed < nr_pages && --sc->priority >= 0); set_task_reclaim_state(p, NULL); memalloc_noreclaim_restore(noreclaim_flag); From aef3e76c024036b403ea49b1aa8de4e57d897747 Mon Sep 17 00:00:00 2001 From: JonasZhou-oc Date: Tue, 25 Aug 2026 18:46:59 +0800 Subject: [PATCH 542/857] mm/vmalloc: avoid false sharing with drain_vmap_work free_vmap_area_noflush() queues drain_vmap_work after the number of lazily freed pages exceeds lazy_max_pages(). Until the worker purges those pages, concurrent frees keep calling schedule_work(). Even if the work is already pending, queue_work_on() performs a locked test_and_set_bit() on the pending bit in the work item. On the tested x86-64 build, drain_vmap_work and vmap_nodes occupy the same 64-byte cache line. The work item starts at offset 0 and the vmap_nodes pointer at offset 32. The latter is read by vmap allocation and free paths, so updates to the work item invalidate a cache line read by all CPUs. Put drain_vmap_work in the cacheline-aligned data section. Tests were run on Linux 7.2. On a two-socket Intel Xeon Silver 4208 system using 16 workers, the runtimes of vmalloc.fix_align, vmalloc.fix_size, and vmalloc.no_block_alloc decreased by 12.61%, 5.78%, and 6.87%, respectively. HITM samples for the affected cache line and total HITM samples decreased by 96.55% and 13.36%, respectively. Link: https://lore.kernel.org/20260825104659.100134-1-jonaszhou-oc@zhaoxin.com Signed-off-by: JonasZhou Reviewed-by: Uladzislau Rezki (Sony) Cc: Signed-off-by: Andrew Morton --- mm/vmalloc.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index bea9f76ed7e742..cfeac79856e8bc 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -1089,7 +1089,12 @@ RB_DECLARE_CALLBACKS_MAX(static, free_vmap_area_rb_augment_cb, static void reclaim_and_purge_vmap_areas(void); static BLOCKING_NOTIFIER_HEAD(vmap_notify_list); static void drain_vmap_area_work(struct work_struct *work); -static DECLARE_WORK(drain_vmap_work, drain_vmap_area_work); +/* + * Keep the work item, whose pending bit is updated by freeing CPUs, + * away from vmap metadata read by allocation and free paths. + */ +static __cacheline_aligned_in_smp +DECLARE_WORK(drain_vmap_work, drain_vmap_area_work); static __cacheline_aligned_in_smp atomic_long_t vmap_lazy_nr; From f2cacb8cc15e67d99e11376ef9fa20e4ca3349fb Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Tue, 25 Aug 2026 10:10:13 +0800 Subject: [PATCH 543/857] mm/hugetlb: fix resv_huge_pages double decrement in memfd error path alloc_hugetlb_folio_reserve() decrements h->resv_huge_pages when dequeuing a folio, but unlike the use_global_reservation handling in hugetlb_alloc_folio(), it does not set HPageRestoreReserve on the folio. Its sole caller memfd_alloc_folio() pre-allocates a reservation via hugetlb_reserve_pages() before allocating. When hugetlb_add_to_page_cache() fails, folio_put() drops the folio without HPageRestoreReserve set, so free_huge_folio() does not restore the reservation. The subsequent hugetlb_unreserve_pages() on the err_unresv path decrements the counter a second time, leaving resv_huge_pages off by one for every failed allocation. Set HPageRestoreReserve when consuming the reservation in alloc_hugetlb_folio_reserve(). On the error path, free_huge_folio() then restores the reservation before hugetlb_unreserve_pages() releases it. The success path is unaffected, as hugetlb_add_to_page_cache() clears the flag once the folio is added to the page cache. Link: https://lore.kernel.org/20260825021013.25672-1-hongfu.li@linux.dev Fixes: 26a8ea80929c ("mm/hugetlb: fix memfd_pin_folios resv_huge_pages leak") Signed-off-by: Hongfu Li Reviewed-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador Cc: Steven Sistare Cc: Vivek Kasireddy Signed-off-by: Andrew Morton --- mm/hugetlb.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index d28972cd33f582..9d47f3be4b6843 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2187,8 +2187,10 @@ struct folio *alloc_hugetlb_folio_reserve(struct hstate *h, int preferred_nid, folio = dequeue_hugetlb_folio_nodemask(h, gfp_mask, preferred_nid, nmask); - if (folio) + if (folio) { + folio_set_hugetlb_restore_reserve(folio); h->resv_huge_pages--; + } spin_unlock_irq(&hugetlb_lock); return folio; From 7b23949480301482630be6f850da1cb9a96afd89 Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Tue, 25 Aug 2026 21:47:15 +0530 Subject: [PATCH 544/857] selftests/mm: remove the local PKEY_UNRESTRICTED fallback pkey-helpers.h defines PKEY_UNRESTRICTED itself when the macro is not already known, a stopgap from when the generic definition was still under review. It has been merged since, commit 6d61527d931b ("mm/pkey: Add PKEY_UNRESTRICTED macro"), so the guard is never taken and the FIXME can be honoured. The definition comes from tools/include/uapi/asm-generic/mman-common.h via TOOLS_INCLUDES, which commit e076eaca5906 ("selftests: break the dependency upon local header files") added so that the mm selftests build without "make headers". It is reached through the that the system includes. Building the pkey tests with KHDR_INCLUDES pointing at an empty directory confirms that; emptying TOOLS_INCLUDES as well is what makes the macro go missing. No functional change intended. Assisted-by: Cursor:claude-opus-5 Link: https://lore.kernel.org/20260825161715.2807297-1-hemanth.selam@gmail.com Signed-off-by: Hemanth Selam Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: SJ Park Cc: Kevin Brodsky Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Yury Khrustalev Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/pkey-helpers.h | 7 ------- 1 file changed, 7 deletions(-) diff --git a/tools/testing/selftests/mm/pkey-helpers.h b/tools/testing/selftests/mm/pkey-helpers.h index 46a8a1878dc1fd..9b949951743dd6 100644 --- a/tools/testing/selftests/mm/pkey-helpers.h +++ b/tools/testing/selftests/mm/pkey-helpers.h @@ -114,13 +114,6 @@ void record_pkey_malloc(void *ptr, long size, int prot); #define PKEY_MASK (PKEY_DISABLE_ACCESS | PKEY_DISABLE_WRITE) #endif -/* - * FIXME: Remove once the generic PKEY_UNRESTRICTED definition is merged. - */ -#ifndef PKEY_UNRESTRICTED -#define PKEY_UNRESTRICTED 0x0 -#endif - #ifndef set_pkey_bits static inline u64 set_pkey_bits(u64 reg, int pkey, u64 flags) { From 224fd53dc765731c216edb039c24700dbba27c23 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Fri, 21 Aug 2026 10:16:06 +0800 Subject: [PATCH 545/857] mm/mglru: preserve inactive placement when enabling MGLRU When the LRU is switched to MGLRU (echo y > /sys/kernel/mm/lru_gen/ enabled), fill_evictable() re-inserts every folio via lru_gen_add_folio(..., false). With reclaiming hardcoded to false, an inactive anonymous folio (no PG_active, not in the swapcache) takes the "gen = MIN_NR_GENS" branch in lru_gen_folio_seq() and is seeded at seq = max_seq - 1, which lru_gen_is_active() treats as active. Its inactive placement is lost and NR_INACTIVE_ANON is folded into NR_ACTIVE_ANON. Pass reclaiming=!active so a folio from an inactive list is seeded into an older generation. Folios from the active list carry PG_active and hit the first branch either way, so they are unchanged. Reclaiming also selects the insertion end in lru_gen_add_folio(): list_add_tail() for inactive folios, list_add() for active ones. Both the legacy LRU and a MGLRU generation keep the hottest folios at the head and the coldest at the tail, and reclaim takes from the tail. To preserve that order the folio must be taken from the end matching the insertion end, so take inactive folios from the head and active folios from the tail; otherwise hot/cold would be reversed within the generation. Tested on x86_64, next-20260812, 2G VM + 1G swap, ~1.5G anon pushed onto the inactive list before enabling MGLRU: Active(anon) Inactive(anon) before switch (legacy) 2952 1548792 kB after `echo y`, unpatched 1552052 0 kB after `echo y`, patched 15144 1536636 kB Inactive file folios stay inactive either way (NR_INACTIVE_FILE is preserved). Link: https://lore.kernel.org/20260821021606.877330-1-ridong.chen@linux.dev Fixes: 354ed5974429 ("mm: multi-gen LRU: kill switch") Signed-off-by: Ridong Chen Suggested-by: Barry Song Assisted-by: Claude:claude-opus-4-8 Acked-by: Barry Song Cc: Axel Rasmussen Cc: David Hildenbrand Cc: Jan Alexander Steffens (heftig) Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oleksandr Natalenko Cc: Shakeel Butt Cc: Steven Barrett Cc: Suleiman Souhlal Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Signed-off-by: Andrew Morton --- mm/vmscan.c | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 6dff207ad8c612..fdd13299a04a93 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -5307,7 +5307,17 @@ static bool fill_evictable(struct lruvec *lruvec) while (!list_empty(head)) { bool success; - struct folio *folio = lru_to_folio(head); + struct folio *folio; + + /* + * lru_gen_add_folio() uses list_add_tail() rather + * than list_add() when reclaiming is true. Match + * its ordering to avoid cold/hot inversion. + */ + if (active) + folio = lru_to_folio(head); + else + folio = list_first_entry(head, struct folio, lru); VM_WARN_ON_ONCE_FOLIO(folio_test_unevictable(folio), folio); VM_WARN_ON_ONCE_FOLIO(folio_test_active(folio) != active, folio); @@ -5315,7 +5325,11 @@ static bool fill_evictable(struct lruvec *lruvec) VM_WARN_ON_ONCE_FOLIO(folio_lru_gen(folio) != -1, folio); lruvec_del_folio(lruvec, folio); - success = lru_gen_add_folio(lruvec, folio, false); + /* + * Borrow reclaiming=true to place inactive folios in + * the older gens. + */ + success = lru_gen_add_folio(lruvec, folio, !active); VM_WARN_ON_ONCE(!success); if (!--remaining) From bb8cc000efb2d7b14f3223e0e59d720edfe608b3 Mon Sep 17 00:00:00 2001 From: Chengfeng Ye Date: Mon, 24 Aug 2026 19:24:33 +0800 Subject: [PATCH 546/857] mm/ksm: mark migration stores with WRITE_ONCE() ksm_get_folio() deliberately samples stable_node->kpfn and folio->mapping without taking the folio lock because the KSM folio may be migrated concurrently. folio_migrate_ksm() updates the same state using plain assignments. The reader can load the old kpfn, then the migrator can store the new kpfn, execute smp_wmb(), and clear the old folio's mapping before the reader checks that mapping. Thus the initial kpfn load can overlap its update and the subsequent mapping load can overlap the clear, with no common lock. This leaves marked READ_ONCE() accesses racing with plain stores. The kernel reported: BUG: KCSAN: data-race in folio_migrate_ksm / ksm_get_folio read (marked) to 0xffff8ce401421330 of 8 bytes by task 48 on cpu 3: ksm_get_folio+0x7f/0x2a0 ksm_scan_thread+0x1635/0x3330 kthread+0x1af/0x1f0 write to 0xffff8ce401421330 of 8 bytes by task 102 on cpu 1: folio_migrate_ksm+0x6a/0xd0 folio_migrate_flags+0x193/0x420 __migrate_folio.isra.0+0x162/0x1a0 migrate_folio+0x4c/0x70 move_to_new_folio+0xd6/0x170 Use WRITE_ONCE() for both stores to pair them with the existing lockless reads. This preserves the existing smp_wmb()/smp_rmb() migration protocol and control flow while preventing compiler transformations of the shared accesses. Link: https://lore.kernel.org/20260824112433.191301-1-nicoyip.dev@gmail.com Fixes: c8d6553b9580 ("ksm: make KSM page migration possible") Signed-off-by: Chengfeng Ye Acked-by: Xu Xin Acked-by: David Hildenbrand (Arm) Cc: Chengming Zhou Cc: Hugh Dickins Cc: xu xin Cc: Signed-off-by: Andrew Morton --- mm/ksm.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/mm/ksm.c b/mm/ksm.c index 49d48d1e099801..fe42c17490b89c 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -1115,7 +1115,8 @@ static inline void folio_set_stable_node(struct folio *folio, struct ksm_stable_node *stable_node) { VM_WARN_ON_FOLIO(folio_test_anon(folio) && PageAnonExclusive(&folio->page), folio); - folio->mapping = (void *)((unsigned long)stable_node | FOLIO_MAPPING_KSM); + WRITE_ONCE(folio->mapping, + (void *)((unsigned long)stable_node | FOLIO_MAPPING_KSM)); } #ifdef CONFIG_SYSFS @@ -3316,7 +3317,7 @@ void folio_migrate_ksm(struct folio *newfolio, struct folio *folio) stable_node = folio_stable_node(folio); if (stable_node) { VM_BUG_ON_FOLIO(stable_node->kpfn != folio_pfn(folio), folio); - stable_node->kpfn = folio_pfn(newfolio); + WRITE_ONCE(stable_node->kpfn, folio_pfn(newfolio)); /* * newfolio->mapping was set in advance; now we need smp_wmb() * to make sure that the new stable_node->kpfn is visible From 394aff7091a0278b0495549b5d2ff3b318e639d2 Mon Sep 17 00:00:00 2001 From: Hao Ge Date: Mon, 17 Aug 2026 14:27:25 +0800 Subject: [PATCH 547/857] alloc_tag: skip percpu counter allocation when profiling is disabled Patch series "alloc_tag: fix a leak and a deadlock around shutdown_mem_profiling()", v2. Two fixes for issues reported by sashiko: 1. percpu counter leak on modules loaded after profiling is disabled. 2. AB-BA deadlock between module load and /proc/allocinfo readers. This patch (of 2): After shutdown_mem_profiling() clears mem_profiling_support, needs_section_mem() returns false, so later modules have their codetag section placed as regular data and never enter the alloc_tag maple tree. codetag_load_module() still called load_module(), which allocated a percpu counter for every tag; release_module_tags() could not find these modules on unload, so the counters leaked. Return -EOPNOTSUPP from load_module() when profiling is off: codetag_module_init() drops the module's cmod, no counters are allocated and the module loads without its tags. codetag_unload_module() now always calls free_section_mem(), since a module whose module_load() returned -EOPNOTSUPP is not in the idr but may still hold a reserved section. Link: https://lore.kernel.org/20260817062726.106511-1-hao.ge@linux.dev Link: https://lore.kernel.org/20260817062726.106511-2-hao.ge@linux.dev Fixes: 4835f747d3ed ("alloc_tag: support for page allocation tag compression") Signed-off-by: Hao Ge Reported-by: Sashiko Suggested-by: Suren Baghdasaryan Acked-by: Suren Baghdasaryan Cc: Kent Overstreet Cc: Signed-off-by: Andrew Morton --- lib/codetag.c | 10 ++++++++-- mm/alloc_tag.c | 4 ++++ 2 files changed, 12 insertions(+), 2 deletions(-) diff --git a/lib/codetag.c b/lib/codetag.c index a9cda4c962a303..a0b600720afc14 100644 --- a/lib/codetag.c +++ b/lib/codetag.c @@ -240,7 +240,9 @@ static int codetag_module_init(struct codetag_type *cttype, struct module *mod) if (err < 0) { kfree(cmod); - return err; + /* -EOPNOTSUPP means we can load the module without its tag. */ + if (err != -EOPNOTSUPP) + return err; } return 0; @@ -388,7 +390,11 @@ void codetag_unload_module(struct module *mod) ++cttype->content_id; } up_write(&cttype->mod_lock); - if (found && cttype->desc.free_section_mem) + /* + * A module whose module_load() returned -EOPNOTSUPP is not + * in the idr but may still hold reserved section memory. + */ + if (cttype->desc.free_section_mem) cttype->desc.free_section_mem(mod, true); } mutex_unlock(&codetag_lock); diff --git a/mm/alloc_tag.c b/mm/alloc_tag.c index b3341031047796..470deb2af2d741 100644 --- a/mm/alloc_tag.c +++ b/mm/alloc_tag.c @@ -975,6 +975,10 @@ static int load_module(struct module *mod, struct codetag *start, struct codetag struct alloc_tag *stop_tag; struct alloc_tag *tag; + /* Profiling disabled: load the module without its tags. */ + if (!mem_profiling_support) + return -EOPNOTSUPP; + /* percpu counters for core allocations are already statically allocated */ if (!mod) return 0; From 4cd27d57fd9468491f2ea48cee93e233c55edda8 Mon Sep 17 00:00:00 2001 From: Hao Ge Date: Mon, 17 Aug 2026 14:27:26 +0800 Subject: [PATCH 548/857] alloc_tag: remove /proc/allocinfo outside of mod_lock shutdown_mem_profiling() calls remove_proc_entry() from reserve_module_tags(), which runs under mod_lock held for write. remove_proc_entry() waits for readers, and a reader takes mod_lock for read in allocinfo_start(): CPU0 (insmod) CPU1 (read /proc/allocinfo) ---------------- ---------------------------- reserve_module_tags() down_write(&mod_lock) [held] use_pde() [in_use++] allocinfo_start() down_read(&mod_lock) <- blocks shutdown_mem_profiling() remove_proc_entry() wait for in_use == 0 <- blocks Move remove_proc_entry() to a workqueue. Link: https://lore.kernel.org/20260817062726.106511-3-hao.ge@linux.dev Fixes: 4835f747d3ed ("alloc_tag: support for page allocation tag compression") Signed-off-by: Hao Ge Reported-by: Sashiko Acked-by: Suren Baghdasaryan Cc: Kent Overstreet Cc: Signed-off-by: Andrew Morton --- mm/alloc_tag.c | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/mm/alloc_tag.c b/mm/alloc_tag.c index 470deb2af2d741..f30ef8dd24c70b 100644 --- a/mm/alloc_tag.c +++ b/mm/alloc_tag.c @@ -15,6 +15,7 @@ #include #include #include +#include #include #include @@ -591,6 +592,13 @@ void pgalloc_tag_swap(struct folio *new, struct folio *old) put_page_tag_ref(handle_new); } +static void remove_allocinfo_file(struct work_struct *work) +{ + remove_proc_entry(ALLOCINFO_FILE_NAME, NULL); +} + +static DECLARE_WORK(remove_allocinfo_work, remove_allocinfo_file); + static void shutdown_mem_profiling(bool remove_file) { if (mem_alloc_profiling_enabled()) @@ -600,7 +608,7 @@ static void shutdown_mem_profiling(bool remove_file) return; if (remove_file) - remove_proc_entry(ALLOCINFO_FILE_NAME, NULL); + schedule_work(&remove_allocinfo_work); mem_profiling_support = false; } From acc745457f20e4137e55f15479084b8e23a0e5ec Mon Sep 17 00:00:00 2001 From: Kaitao Cheng Date: Mon, 24 Aug 2026 23:16:55 +0800 Subject: [PATCH 549/857] mm/hugetlb: use hugetlb_vmemmap_optimizable() in boolean contexts The two boot-time sites in hugetlb_hstate_alloc_pages_onenode() and hugetlb_pages_alloc_boot_node() only need to know whether HVO is applicable, not the exact optimizable size. Switch them from hugetlb_vmemmap_optimizable_size() to hugetlb_vmemmap_optimizable() to make intent explicit. No functional change intended. Link: https://lore.kernel.org/20260824151655.30840-1-kaitao.cheng@linux.dev Signed-off-by: Kaitao Cheng Reviewed-by: Joshua Hahn Reviewed-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador Signed-off-by: Andrew Morton --- mm/hugetlb.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 9d47f3be4b6843..871b81670cbe9c 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3427,7 +3427,7 @@ static void __init hugetlb_hstate_alloc_pages_onenode(struct hstate *h, int nid) folio = only_alloc_fresh_hugetlb_folio(h, gfp_mask, nid, &node_states[N_MEMORY], NULL); if (!folio && !list_empty(&folio_list) && - hugetlb_vmemmap_optimizable_size(h)) { + hugetlb_vmemmap_optimizable(h)) { prep_and_add_allocated_folios(h, &folio_list); INIT_LIST_HEAD(&folio_list); folio = only_alloc_fresh_hugetlb_folio(h, gfp_mask, nid, @@ -3496,7 +3496,7 @@ static void __init hugetlb_pages_alloc_boot_node(unsigned long start, unsigned l for (i = 0; i < num; ++i) { struct folio *folio; - if (hugetlb_vmemmap_optimizable_size(h) && + if (hugetlb_vmemmap_optimizable(h) && (si_mem_available() == 0) && !list_empty(&folio_list)) { prep_and_add_allocated_folios(h, &folio_list); INIT_LIST_HEAD(&folio_list); From ed778eaa27888b9b3888edf6828b214b0b16121e Mon Sep 17 00:00:00 2001 From: Jinjiang Tu Date: Mon, 24 Aug 2026 14:10:09 +0800 Subject: [PATCH 550/857] docs: ksm: fix typos in sysfs knob names Patch series "docs/ksm: fix advisor documentation and comment", v3. This series fixes two problems left in the KSM advisor documentation and code comment: - Patch 1 fixes two typos in sysfs knob names ("adivsor_max_cpu" and "adivsor_max_pages_to_scan") in ksm.rst that don't match the actual knob names. - Patch 2 fixes the description of advisor_min_pages_to_scan: it is described as the lower limit of pages_to_scan, but is actually only used as it's initial value, and the runtime pages_to_scan can drop below advisor_min_pages_to_scan. This patch (of 2): The sysfs knob names in mm/ksm.c are "advisor_max_cpu" and "advisor_max_pages_to_scan", but the ksm.rst documentation spelled both as "adivsor_*", fix the two typos. Link: https://lore.kernel.org/20260824061010.3343959-1-tujinjiang@huawei.com Link: https://lore.kernel.org/20260824061010.3343959-2-tujinjiang@huawei.com Signed-off-by: Jinjiang Tu Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Randy Dunlap Cc: Chengming Zhou Cc: Jonathan Corbet Cc: Kefeng Wang Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Nanyong Sun Cc: Stefan Roesch Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin Signed-off-by: Andrew Morton --- Documentation/admin-guide/mm/ksm.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/admin-guide/mm/ksm.rst b/Documentation/admin-guide/mm/ksm.rst index ad8e7a41f3b5d2..c9f533b10f6f9d 100644 --- a/Documentation/admin-guide/mm/ksm.rst +++ b/Documentation/admin-guide/mm/ksm.rst @@ -174,7 +174,7 @@ advisor_mode The section about ``advisor`` explains in detail how the scan time advisor works. -adivsor_max_cpu +advisor_max_cpu specifies the upper limit of the cpu percent usage of the ksmd background thread. The default is 70. @@ -186,7 +186,7 @@ advisor_min_pages_to_scan specifies the lower limit of the ``pages_to_scan`` parameter of the scan time advisor. The default is 500. -adivsor_max_pages_to_scan +advisor_max_pages_to_scan specifies the upper limit of the ``pages_to_scan`` parameter of the scan time advisor. The default is 30000. From c513d80c7f1f9be2b5972f5577e564ddf58a5885 Mon Sep 17 00:00:00 2001 From: Jinjiang Tu Date: Mon, 24 Aug 2026 14:10:10 +0800 Subject: [PATCH 551/857] mm/ksm: fix advisor_min_pages_to_scan description Both Documentation/admin-guide/mm/ksm.rst and the comment next to the variable definition in mm/ksm.c describe advisor_min_pages_to_scan as a lower limit of the pages_to_scan parameter, but that is not how the scan-time advisor actually uses it. commit 4e5fa4f5eff6 ("mm/ksm: add ksm advisor") only uses it to initialize ksm_thread_pages_to_scan when the scan-time advisor is enabled. ksm_thread_pages_to_scan is adjusted by scan_time_advisor() after a full scan finishes. ksm_thread_pages_to_scan could be increased or decreased depend on the real scan time is longer or shorter than the target scan time. The min value of ksm_thread_pages_to_scan is only limited by KSM_ADVISOR_MIN_CPU, so ksm_thread_pages_to_scan could be smaller than ksm_advisor_min_pages_to_scan. The semantics of advisor_min_pages_to_scan was updated in the v2 patchset [1], but the documentation wasn't updated. Update the documentation and comment to match the semantics of advisor_min_pages_to_scan. Link: https://lore.kernel.org/linux-mm/20231028000945.2428830-2-shr@devkernel.io/ [1] Link: https://lore.kernel.org/20260824061010.3343959-3-tujinjiang@huawei.com Signed-off-by: Jinjiang Tu Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Chengming Zhou Cc: Jonathan Corbet Cc: Kefeng Wang Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Nanyong Sun Cc: Stefan Roesch Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin Cc: Randy Dunlap Signed-off-by: Andrew Morton --- Documentation/admin-guide/mm/ksm.rst | 4 ++-- mm/ksm.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/Documentation/admin-guide/mm/ksm.rst b/Documentation/admin-guide/mm/ksm.rst index c9f533b10f6f9d..c329ca747b8c47 100644 --- a/Documentation/admin-guide/mm/ksm.rst +++ b/Documentation/admin-guide/mm/ksm.rst @@ -183,8 +183,8 @@ advisor_target_scan_time pages. The default value is 200 seconds. advisor_min_pages_to_scan - specifies the lower limit of the ``pages_to_scan`` parameter of the - scan time advisor. The default is 500. + specifies the initial value of the ``pages_to_scan`` parameter of + the scan time advisor. The default is 500. advisor_max_pages_to_scan specifies the upper limit of the ``pages_to_scan`` parameter of the diff --git a/mm/ksm.c b/mm/ksm.c index fe42c17490b89c..624f37975e1295 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -348,7 +348,7 @@ static enum ksm_advisor_type ksm_advisor; * Only called through the sysfs control interface: */ -/* At least scan this many pages per batch. */ +/* Initial number of pages to scan per batch. */ static unsigned long ksm_advisor_min_pages_to_scan = 500; static void set_advisor_defaults(void) From ee9f92bcf4a0a82b0606bfb51b0f3b963142debd Mon Sep 17 00:00:00 2001 From: Anshuman Date: Wed, 26 Aug 2026 11:43:00 +0530 Subject: [PATCH 552/857] selftests/mm: fix line buffer leak in mremap_test is_range_mapped() is_range_mapped() uses getline() to read /proc/self/maps line by line, but never frees the buffer it allocates. Every exit path (parse failure, match found, or reaching EOF) returns without calling free(line), leaking the buffer on each call. The function is called multiple times in this test, so the leak accumulates across calls. Free line before returning. Link: https://lore.kernel.org/20260826061300.14038-1-anshumantewari123@gmail.com Signed-off-by: Anshuman Acked-by: David Hildenbrand (Arm) Reviewed-by: SJ Park Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/mremap_test.c | 1 + 1 file changed, 1 insertion(+) diff --git a/tools/testing/selftests/mm/mremap_test.c b/tools/testing/selftests/mm/mremap_test.c index 131d9d6db86790..779ef2d5f1b9f7 100644 --- a/tools/testing/selftests/mm/mremap_test.c +++ b/tools/testing/selftests/mm/mremap_test.c @@ -156,6 +156,7 @@ static bool is_range_mapped(FILE *maps_fp, unsigned long start, } } + free(line); return success; } From 984a607d1012c87d7ff8e37eef1fd80b7b3c7c09 Mon Sep 17 00:00:00 2001 From: Jiayuan Chen Date: Thu, 27 Aug 2026 10:54:56 +0800 Subject: [PATCH 553/857] mm/memcontrol: fix data-race on reading jiffies_64 KCSAN reported a data-race between tick_do_update_jiffies64() updating jiffies_64 and mem_cgroup_flush_stats_ratelimited() reading it directly. Unlike jiffies, jiffies_64 is not volatile, so raw reads are plain accesses and can even be torn on 32-bit. Use get_jiffies_64() instead, and fix the same pattern in mem_cgroup_flush_foreign(). Link: https://lore.kernel.org/20260827025457.116191-1-jiayuan.chen@linux.dev Fixes: 508bed884767 ("mm: memcg: change flush_next_time to flush_last_time") Fixes: 97b27821b485 ("writeback, memcg: Implement foreign dirty flushing") Signed-off-by: Jiayuan Chen Reported-by: syzbot+ced4d9a8cadb5ef3adae@syzkaller.appspotmail.com Acked-by: Muchun Song Acked-by: Johannes Weiner Acked-by: Michal Hocko Cc: Chris Li Cc: Jan Kara Cc: Jens Axboe Cc: Roman Gushchin Cc: Shakeel Butt Cc: Tejun Heo Cc: Signed-off-by: Andrew Morton --- mm/memcontrol.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 856a7d07586ccc..d399710e9799ab 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -757,7 +757,7 @@ static void __mem_cgroup_flush_stats(struct mem_cgroup *memcg, bool force) return; if (mem_cgroup_is_root(memcg)) - WRITE_ONCE(flush_last_time, jiffies_64); + WRITE_ONCE(flush_last_time, get_jiffies_64()); css_rstat_flush(&memcg->css); } @@ -785,7 +785,7 @@ void mem_cgroup_flush_stats(struct mem_cgroup *memcg) void mem_cgroup_flush_stats_ratelimited(struct mem_cgroup *memcg) { /* Only flush if the periodic flusher is one full cycle late */ - if (time_after64(jiffies_64, READ_ONCE(flush_last_time) + 2*FLUSH_TIME)) + if (time_after64(get_jiffies_64(), READ_ONCE(flush_last_time) + 2 * FLUSH_TIME)) mem_cgroup_flush_stats(memcg); } @@ -3946,7 +3946,7 @@ void mem_cgroup_flush_foreign(struct bdi_writeback *wb) { struct mem_cgroup *memcg = mem_cgroup_from_css(wb->memcg_css); unsigned long intv = msecs_to_jiffies(dirty_expire_interval * 10); - u64 now = jiffies_64; + u64 now = get_jiffies_64(); int i; for (i = 0; i < MEMCG_CGWB_FRN_CNT; i++) { From 10df392985f074e4acae90a67e60783b921dca9e Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Thu, 27 Aug 2026 15:05:46 +0800 Subject: [PATCH 554/857] mm/vmstat: annotate data race for per-cpu pageset fields zoneinfo_show_print() reads pcp->count, pcp->high, pcp->batch, pcp->high_min, pcp->high_max and the per-cpu stat_threshold while holding only zone->lock, which does not synchronize these fields. The writers are the page allocation and free fast paths under pcp->lock, decay_pcp_high() which updates pcp->high without any lock, pageset_update() which writes batch/high_min/high_max with WRITE_ONCE(), and refresh_zone_stat_thresholds() which writes stat_threshold locklessly. The race is benign: the values are only printed to /proc/zoneinfo, they are naturally aligned integers, and pageset_update() already documents that users of batch/high_min/high_max must cope with the fields changing asynchronously. Annotate the reads with data_race(), following commit af1c31acc853 ("mm/vmstat: annotate data race for zone->free_area[order].nr_free"). Found by KCSAN testing on an older kernel; the same race still exists on mainline. No functional change intended. BUG: KCSAN: data-race in zoneinfo_show_print+0x355/0x520 root/klinux/mm/vmstat.c:1774 race at unknown origin, with read to 0xffff8e1835410808 of 4 bytes by task 22653 on cpu 12: zoneinfo_show_print+0x355/0x520 root/klinux/mm/vmstat.c:1774 walk_zones_in_node root/klinux/mm/vmstat.c:1496 [inline] zoneinfo_show+0x41/0x70 root/klinux/mm/vmstat.c:1806 seq_read_iter+0x30c/0x970 root/klinux/fs/seq_file.c:230 proc_reg_read_iter+0x10c/0x170 root/klinux/fs/proc/inode.c:305 copy_splice_read+0x2a1/0x4e0 root/klinux/fs/splice.c:365 do_splice_read root/klinux/fs/splice.c:985 [inline] do_splice_read+0x139/0x1a0 root/klinux/fs/splice.c:959 splice_direct_to_actor+0x16b/0x540 root/klinux/fs/splice.c:1089 do_splice_direct_actor root/klinux/fs/splice.c:1207 [inline] do_splice_direct+0x10a/0x180 root/klinux/fs/splice.c:1233 do_sendfile+0x6ea/0x7e0 root/klinux/fs/read_write.c:1363 __do_sys_sendfile64 root/klinux/fs/read_write.c:1424 [inline] __se_sys_sendfile64 root/klinux/fs/read_write.c:1410 [inline] __x64_sys_sendfile64+0x117/0x130 root/klinux/fs/read_write.c:1410 x64_sys_call+0x1cc7/0x1ee0 root/klinux/./arch/x86/include/generated/asm/syscalls_64.h:41 do_syscall_x64 root/klinux/arch/x86/entry/common.c:46 [inline] do_syscall_64+0x75/0x2c0 root/klinux/arch/x86/entry/common.c:76 entry_SYSCALL_64_after_hwframe+0x76/0xe0 value changed: 0x000001b2 -> 0x000001b1 Reported by Kernel Concurrency Sanitizer on: CPU: 12 PID: 22653 Comm: syz-executor.12 Not tainted 6.6.140+ #672 Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS rel-1.16.3-0-ga6ed6b701f0a-prebuilt.qemu.org 04/01/2014 Link: https://lore.kernel.org/20260827070546.1336383-1-hui.zhu@linux.dev Signed-off-by: Hui Zhu Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Signed-off-by: Andrew Morton --- mm/vmstat.c | 17 +++++++++++------ 1 file changed, 11 insertions(+), 6 deletions(-) diff --git a/mm/vmstat.c b/mm/vmstat.c index cb57714539fb5e..a3e809c57f295a 100644 --- a/mm/vmstat.c +++ b/mm/vmstat.c @@ -1837,6 +1837,11 @@ static void zoneinfo_show_print(struct seq_file *m, pg_data_t *pgdat, struct per_cpu_zonestat __maybe_unused *pzstats; pcp = per_cpu_ptr(zone->per_cpu_pageset, i); + /* + * Access to the per-cpu pageset fields is lockless as they + * are used only for printing purposes. Use data_race to + * avoid KCSAN warning. + */ seq_printf(m, "\n cpu: %i" "\n count: %i" @@ -1845,15 +1850,15 @@ static void zoneinfo_show_print(struct seq_file *m, pg_data_t *pgdat, "\n high_min: %i" "\n high_max: %i", i, - pcp->count, - pcp->high, - pcp->batch, - pcp->high_min, - pcp->high_max); + data_race(pcp->count), + data_race(pcp->high), + data_race(pcp->batch), + data_race(pcp->high_min), + data_race(pcp->high_max)); #ifdef CONFIG_SMP pzstats = per_cpu_ptr(zone->per_cpu_zonestats, i); seq_printf(m, "\n vm stats threshold: %d", - pzstats->stat_threshold); + data_race(pzstats->stat_threshold)); #endif } seq_printf(m, From 1307a0ff80562c8083b110afb55bf837fbd7700a Mon Sep 17 00:00:00 2001 From: Hao Li Date: Thu, 27 Aug 2026 15:18:06 +0800 Subject: [PATCH 555/857] mm: remove unused anon_vma_trylock_write() The last user of anon_vma_trylock_write() was removed by cc22b9978509, leaving the helper unused. Remove it. Link: https://lore.kernel.org/20260827071845.17636-1-hao.li@linux.dev Signed-off-by: Hao Li Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Reviewed-by: SJ Park Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/internal.h | 5 ----- 1 file changed, 5 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 38b1165212c941..da833cafcd599d 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -295,11 +295,6 @@ static inline void anon_vma_lock_write(struct anon_vma *anon_vma) down_write(&anon_vma->root->rwsem); } -static inline int anon_vma_trylock_write(struct anon_vma *anon_vma) -{ - return down_write_trylock(&anon_vma->root->rwsem); -} - static inline void anon_vma_unlock_write(struct anon_vma *anon_vma) { up_write(&anon_vma->root->rwsem); From 505cbae766939df7dcdbf0890c1c673d472f0783 Mon Sep 17 00:00:00 2001 From: Yue Haibing Date: Thu, 27 Aug 2026 16:27:22 +0800 Subject: [PATCH 556/857] mm/swap: remove unused declaration swapcache_clear() Commit c246d236b18b ("mm/shmem: never bypass the swap cache for SWP_SYNCHRONOUS_IO") removed the implementations but leave this. Link: https://lore.kernel.org/20260827082722.1809702-1-yuehaibing@huawei.com Signed-off-by: Yue Haibing Reviewed-by: Baoquan He Acked-by: Kairui Song Acked-by: Nick Huang Reviewed-by: Barry Song Reviewed-by: SJ Park Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Signed-off-by: Andrew Morton --- mm/swap.h | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/swap.h b/mm/swap.h index 90a551a88df63b..fddba7a87500a4 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -324,7 +324,6 @@ void __swap_cache_replace_folio(struct swap_cluster_info *ci, struct folio *old, struct folio *new); void show_swap_cache_info(void); -void swapcache_clear(struct swap_info_struct *si, swp_entry_t entry, int nr); struct folio *read_swap_cache_async(struct swap_io_ctx *ctx, swp_entry_t entry, gfp_t gfp_mask, struct vm_area_struct *vma, unsigned long addr); struct folio *swap_cluster_readahead(swp_entry_t entry, gfp_t flag, From 9007ee950af2c454123afdcaf7cad616047083e4 Mon Sep 17 00:00:00 2001 From: Tao Cui Date: Wed, 26 Aug 2026 10:17:53 +0800 Subject: [PATCH 557/857] docs: cgroup: document empty-write behavior of memory limit knobs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A maintenance script on a cluster wrote an unset variable into memory.max of a workload cgroup; the variable expanded to an empty string, the write succeeded, and the workload in the cgroup was OOM-killed. Nothing pointed back at the write, so it took quite some time to trace the OOM kills to that script. The memory controller documentation does not say what an empty write does; the cpuset controller documents its empty-value semantics. The actual behavior is that the empty string is accepted as 0. Reproduced on a k8s cluster (v1.29, cgroup v2, two-container pod, 384M limit): # LIMIT= # echo "$LIMIT" > $CG/memory.max # echo $? 0 m6demo 0/2 OOMKilled 0 Memory cgroup out of memory: Killed process 339529 (sleep) ... anon-rss:32kB State it where the interface files are introduced, alongside the existing notes on units and page rounding. Link: https://lore.kernel.org/all/aoVUlFdZYLFn_gvJ@tiehlicka/ Link: https://lore.kernel.org/20260826021753.197871-1-cui.tao@linux.dev Signed-off-by: Tao Cui Acked-by: Michal Hocko Acked-by: Shakeel Butt Cc: Johannes Weiner Cc: Michal Koutný Cc: Muchun Song Cc: Roman Gushchin Cc: Tejun Heo Signed-off-by: Andrew Morton --- Documentation/admin-guide/cgroup-v2.rst | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/Documentation/admin-guide/cgroup-v2.rst b/Documentation/admin-guide/cgroup-v2.rst index 86a2a0099178ea..8d2603751c51a1 100644 --- a/Documentation/admin-guide/cgroup-v2.rst +++ b/Documentation/admin-guide/cgroup-v2.rst @@ -1321,6 +1321,10 @@ All memory amounts are in bytes. If a value which is not aligned to PAGE_SIZE is written, the value may be rounded up to the closest PAGE_SIZE multiple when read back. +For the limit files described below, an empty or all-whitespace +write is accepted and sets the limit to 0. To disable a limit, +write "max"; to set it to zero explicitly, write "0". + memory.current A read-only single value file which exists on non-root cgroups. From 62db8b658cd356dbb480e33564498e5a7c31c3a3 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Tue, 25 Aug 2026 13:20:59 +0200 Subject: [PATCH 558/857] selftests/mm: khugepaged: remove str_dup() usage We don't check str_dup() return value and never free it. While both things are irrelevant in practice, let's just clean it up by working on argv[0] directly and avoiding the str_dup(). Nobody after us needs these parts of the argv[0] string anyway. This patch is inspired by previous work from Anshuman Tewari [1]. Link: https://lore.kernel.org/r/20260821114416.12255-1-anshumantewari123@gmail.com [1] Link: https://lore.kernel.org/20260825-remove_str_dup-v1-1-0ba2121a820c@kernel.org Signed-off-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Anshuman Tewari Reviewed-by: Zi Yan Reviewed-by: Lance Yang Reviewed-by: Dev Jain Acked-by: Usama Arif Reviewed-by: Baolin Wang Reviewed-by: Barry Song Reviewed-by: Andrew Morton Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/khugepaged.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 1d2d6bd72fd2af..83d27d069c4139 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1227,7 +1227,7 @@ static void parse_test_type(int argc, char **argv) return; } - buf = strdup(argv[0]); + buf = argv[0]; token = strsep(&buf, ":"); if (!strcmp(token, "all")) { From b41311de029c168456147c6c730ab73a405287b3 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Tue, 25 Aug 2026 20:01:53 +0800 Subject: [PATCH 559/857] mm/memcontrol: remove unused memcg parameter in calculate_high_delay() The memcg argument of calculate_high_delay() is never referenced in its function body. The delay calculation only depends on nr_pages and max_overage, and both callers have already obtained max_overage from the same memcg. Drop this unused parameter and update the two call sites inside __mem_cgroup_handle_over_high(). Link: https://lore.kernel.org/20260825120153.1405-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Acked-by: Michal Hocko Reviewed-by: Gregory Price (Meta) Reviewed-by: SJ Park Acked-by: Shakeel Butt Signed-off-by: Andrew Morton --- mm/memcontrol.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index d399710e9799ab..1709ac96bbdec5 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2515,8 +2515,7 @@ static u64 swap_find_max_overage(struct mem_cgroup *memcg) * Get the number of jiffies that we should penalise a mischievous cgroup which * is exceeding its memory.high by checking both it and its ancestors. */ -static unsigned long calculate_high_delay(struct mem_cgroup *memcg, - unsigned int nr_pages, +static unsigned long calculate_high_delay(unsigned int nr_pages, u64 max_overage) { unsigned long penalty_jiffies; @@ -2594,10 +2593,10 @@ void __mem_cgroup_handle_over_high(gfp_t gfp_mask) * memory.high is breached and reclaim is unable to keep up. Throttle * allocators proactively to slow down excessive growth. */ - penalty_jiffies = calculate_high_delay(memcg, nr_pages, + penalty_jiffies = calculate_high_delay(nr_pages, mem_find_max_overage(memcg)); - penalty_jiffies += calculate_high_delay(memcg, nr_pages, + penalty_jiffies += calculate_high_delay(nr_pages, swap_find_max_overage(memcg)); /* From 5094cd253f2468db2aa29e308d7af3f79b3f1c81 Mon Sep 17 00:00:00 2001 From: Qi Xi Date: Tue, 25 Aug 2026 20:05:48 +0800 Subject: [PATCH 560/857] mm/page_isolation: fix UBSAN shift-out-of-bounds warning Patch series "mm/page_isolation: fix UBSAN shift-out-of-bounds in isolate_single_pageblock", v3. Patch 1 fixes a UBSAN shift-out-of-bounds warning in the PageBuddy branch of isolate_single_pageblock() triggered by concurrent buddy allocation. Patch 2 addresses the same class of issue in the PageCompound branch, where racy compound_order() and compound_head() reads could lead to out-of-range shifts or incorrect page skipping. This patch (of 2): A contig-range allocation racing with buddy allocation on the adjacent pageblock can trigger: UBSAN: shift-out-of-bounds in mm/page_isolation.c:393:15 shift exponent -749042176 is negative Call trace: isolate_single_pageblock start_isolate_page_range alloc_contig_frozen_range_noprof alloc_contig_range_noprof isolate_single_pageblock() first calls set_migratetype_isolate() with zone->lock held, which marks the pageblock MIGRATE_ISOLATE and moves any free page straddling the boundary out of the way. Once the lock is dropped, it scans the MAX_ORDER_NR_PAGES-aligned window [start_pfn, boundary_pfn) locklessly, only to skip the free pages already handled above and to detect in-use pages straddling the boundary. Since this scan only reads page state to decide how far to skip and returns -EBUSY on a straddling in-use page, it does not take the lock. The window also covers the adjacent pageblock, whose free pages stay on the normal movable/CMA freelist and can be allocated concurrently. So after the scan observes PageBuddy(page), another CPU can allocate the page, leaving a stale value in page->private that makes "1 << order" shift out of range. Use buddy_order_unsafe() with READ_ONCE to read the order, and validate it is within MAX_PAGE_ORDER before shifting to prevent UBSAN warnings. Since pageblock_isolate_and_move_free_pages() already handles free pages straddling boundary_pfn under zone->lock, bail out with -EBUSY instead of VM_WARN_ON_ONCE() when a PageBuddy page appears to cross the boundary during the lockless scan. Link: https://lore.kernel.org/20260825120549.966271-2-xiqi2@huawei.com Fixes: b2c9e2fbba32 ("mm: make alloc_contig_range work at pageblock granularity") Signed-off-by: Qi Xi Reviewed-by: Zi Yan Cc: Johannes Weiner Cc: Kefeng Wang Cc: Michal Hocko Cc: Nanyong Sun Cc: Qi Xi Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Zi Yan Cc: Brendan Jackman Cc: Signed-off-by: Andrew Morton --- mm/page_isolation.c | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/mm/page_isolation.c b/mm/page_isolation.c index e5dfc7bf494460..e4ee998c00f3bc 100644 --- a/mm/page_isolation.c +++ b/mm/page_isolation.c @@ -388,13 +388,19 @@ static int isolate_single_pageblock(unsigned long boundary_pfn, } if (PageBuddy(page)) { - int order = buddy_order(page); + unsigned int order = buddy_order_unsafe(page); - /* pageblock_isolate_and_move_free_pages() handled this */ - VM_WARN_ON_ONCE(pfn + (1 << order) > boundary_pfn); - - pfn += 1UL << order; - continue; + /* buddy_order_unsafe() is racy. Validate the order before shifting. */ + if (order <= MAX_PAGE_ORDER && + /* + * pageblock_isolate_and_move_free_pages() splits + * cross-boundary PageBuddy, verify it. + */ + pfn + (1UL << order) <= boundary_pfn) { + pfn += 1UL << order; + continue; + } + goto failed; } /* From ca49fe8408c623329f0078517cb23da79dafda1d Mon Sep 17 00:00:00 2001 From: Qi Xi Date: Tue, 25 Aug 2026 20:05:49 +0800 Subject: [PATCH 561/857] mm/page_isolation: guard compound_order() against racing The PageCompound branch reads compound_head() without holding a reference. A racing split or free can cause compound_head() to return a stale pointer, and compound_nr() reads the order from that stale head, leading to out-of-range shifts and making the skip distance meaningless. Read the order explicitly with compound_order() and validate it is within MAX_FOLIO_ORDER before shifting. Also verify the derived head_pfn against the legitimate pfn: the head must not be past pfn, must be aligned to nr_pages, and pfn must fall within the compound page. Bail out with -EBUSY if any check fails. Link: https://lore.kernel.org/20260825120549.966271-3-xiqi2@huawei.com Fixes: b2c9e2fbba32 ("mm: make alloc_contig_range work at pageblock granularity") Signed-off-by: Qi Xi Suggested-by: Zi Yan Reviewed-by: Zi Yan Cc: Brendan Jackman Cc: Johannes Weiner Cc: Kefeng Wang Cc: Michal Hocko Cc: Nanyong Sun Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Signed-off-by: Andrew Morton --- mm/page_isolation.c | 22 ++++++++++++++++++++-- 1 file changed, 20 insertions(+), 2 deletions(-) diff --git a/mm/page_isolation.c b/mm/page_isolation.c index e4ee998c00f3bc..5aaad037384e44 100644 --- a/mm/page_isolation.c +++ b/mm/page_isolation.c @@ -419,10 +419,28 @@ static int isolate_single_pageblock(unsigned long boundary_pfn, if (PageCompound(page)) { struct page *head = compound_head(page); unsigned long head_pfn = page_to_pfn(head); - unsigned long nr_pages = compound_nr(head); + unsigned int order = compound_order(head); + unsigned long nr_pages; + + /* compound_order() is racy. Cap it at MAX_FOLIO_ORDER. */ + if (order > MAX_FOLIO_ORDER) + goto failed; + + nr_pages = 1UL << order; + + /* + * compound_head() is also racy, so the derived head_pfn + * needs additional checks to make sure it is valid. + * Otherwise, just fail the check. pfn comes from + * __first_valid_page() as a legitimate PFN, so use it to + * check head_pfn. + */ + if (head_pfn > pfn || !IS_ALIGNED(head_pfn, nr_pages) || + pfn - head_pfn >= nr_pages) + goto failed; if (head_pfn + nr_pages <= boundary_pfn || - PageHuge(page)) { + PageHuge(head)) { pfn = head_pfn + nr_pages; continue; } From 2a61596db288a50e90b1827ed0299401ee1b0644 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 25 Aug 2026 20:30:23 +0800 Subject: [PATCH 562/857] selftests/mm: fix incorrect skip output in pkey_sighandler_tests When pkeys is not supported, ksft_exit_skip() runs with ksft_plan already set, which takes the "ok N # SKIP" branch intended for skipping a single test case. The result is a TAP plan of 5 but only one result line. $ ./pkey_sighandler_tests TAP version 13 1..5 ok 1 # SKIP pkeys not supported # 1 skipped test(s) detected. Consider enabling relevant config options to improve coverage. # Planned tests != run tests (5 != 1) # Totals: pass:0 fail:0 xfail:0 xpass:0 skip:1 error:0 Move ksft_set_plan() after the skip check so ksft_exit_skip() takes the "1..0 # SKIP" branch, the correct TAP representation for skipping an entire test file. $ ./pkey_sighandler_tests TAP version 13 1..0 # SKIP pkeys not supported Link: https://lore.kernel.org/20260825123023.64418-1-zenghui.yu@linux.dev Signed-off-by: Zenghui Yu (Huawei) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Zenghui Yu Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/pkey_sighandler_tests.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/pkey_sighandler_tests.c b/tools/testing/selftests/mm/pkey_sighandler_tests.c index c218d0510a2a42..74bf79a5399dac 100644 --- a/tools/testing/selftests/mm/pkey_sighandler_tests.c +++ b/tools/testing/selftests/mm/pkey_sighandler_tests.c @@ -543,11 +543,12 @@ static void (*pkey_tests[])(void) = { int main(int argc, char *argv[]) { ksft_print_header(); - ksft_set_plan(ARRAY_SIZE(pkey_tests)); if (!is_pkeys_supported()) ksft_exit_skip("pkeys not supported\n"); + ksft_set_plan(ARRAY_SIZE(pkey_tests)); + for (test_nr = 0; test_nr < ARRAY_SIZE(pkey_tests); test_nr++) { tracing_on(); (*pkey_tests[test_nr])(); From 81025f00161fd8582d1413071af577ffa36aba4a Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:45:52 +0800 Subject: [PATCH 563/857] mm/sparse: relax struct mem_section size constraints Patch series "mm: Introduce section-based vmemmap optimization for HugeTLB", v5. HugeTLB vmemmap optimization currently has its own early boot setup path. It pre-populates optimized vmemmap mappings before the normal sparse-vmemmap population code runs, and sparsemem carries SPARSEMEM_VMEMMAP_PREINIT only to support that special case. That makes the HugeTLB vmemmap optimization path harder to share with other users of sparse-vmemmap optimization and leaves a fair amount of HugeTLB-specific boot-time state in the generic memory initialization flow. This series introduces section-based vmemmap optimization support in the sparse-vmemmap code and switches HugeTLB bootmem pages over to it. Instead of having HugeTLB pre-populate optimized vmemmap mappings itself, HugeTLB now records the compound page order in the corresponding memory sections. The generic sparse-vmemmap population path can then allocate or reuse shared tail vmemmap pages based on section metadata. The patches are organized as follows: - patches 1-2 prepare sparsemem and vmemmap optimization metadata - patches 3-8 teach the common sparse-vmemmap paths to use that state - patches 9-10 switch HugeTLB bootmem optimization to the section-based path - patches 11-17 clean up sparsemem and HugeTLB bootmem code that is no longer needed after the conversion This is intended to be the second smaller step toward the broader HVO generalization [1]. The device DAX conversion and the wider HVO consolidation are left for follow-up series. This patch (of 17): struct mem_section is currently forced to a power-of-2 size so the section-to-root lookup can use a mask instead of a modulo. That requirement makes future extensions harder than necessary: adding a small field can require configuration-dependent padding or layout checks just to preserve the lookup scheme. Keep the lookup correct for any struct mem_section size by using a plain modulo instead. Do not leave the layout entirely unconstrained, though. Keep struct mem_section double-word aligned so modest size changes, such as adding another word-sized field on 64-bit systems, still keep a compact and efficient layout. If future fields grow the structure beyond that sweet spot, the lookup remains correct; only the exact layout efficiency changes. Link: https://lore.kernel.org/20260825084608.47437-1-songmuchun@bytedance.com Link: https://lore.kernel.org/20260825084608.47437-2-songmuchun@bytedance.com Link: https://lore.kernel.org/linux-mm/20260513130542.35604-1-songmuchun@bytedance.com/ [1] Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Qi Zheng Signed-off-by: Andrew Morton --- include/linux/mmzone.h | 11 +++-------- mm/sparse.c | 2 -- scripts/gdb/linux/mm.py | 6 ++---- 3 files changed, 5 insertions(+), 14 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 94f9c3ff541604..0a2428714108c4 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2027,13 +2027,9 @@ struct mem_section { * section. (see page_ext.h about this.) */ struct page_ext *page_ext; - unsigned long pad; #endif - /* - * WARNING: mem_section must be a power-of-2 in size for the - * calculation and use of SECTION_ROOT_MASK to make sense. - */ -}; +/* Sacrifice minor padding space for efficient lookup. */ +} __aligned(2 * sizeof(unsigned long)); #ifdef CONFIG_SPARSEMEM_EXTREME #define SECTIONS_PER_ROOT (PAGE_SIZE / sizeof (struct mem_section)) @@ -2043,7 +2039,6 @@ struct mem_section { #define SECTION_NR_TO_ROOT(sec) ((sec) / SECTIONS_PER_ROOT) #define NR_SECTION_ROOTS DIV_ROUND_UP(NR_MEM_SECTIONS, SECTIONS_PER_ROOT) -#define SECTION_ROOT_MASK (SECTIONS_PER_ROOT - 1) #ifdef CONFIG_SPARSEMEM_EXTREME extern struct mem_section **mem_section; @@ -2067,7 +2062,7 @@ static inline struct mem_section *__nr_to_section(unsigned long nr) if (!mem_section || !mem_section[root]) return NULL; #endif - return &mem_section[root][nr & SECTION_ROOT_MASK]; + return &mem_section[root][nr % SECTIONS_PER_ROOT]; } /* diff --git a/mm/sparse.c b/mm/sparse.c index 7c15406e77f5d2..c84b4c7b8c7069 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -322,8 +322,6 @@ void __init sparse_init(void) unsigned long pnum_end, pnum_begin, map_count = 1; int nid_begin; - /* see include/linux/mmzone.h 'struct mem_section' definition */ - BUILD_BUG_ON(!is_power_of_2(sizeof(struct mem_section))); memblocks_present(); if (compound_info_has_mask()) { diff --git a/scripts/gdb/linux/mm.py b/scripts/gdb/linux/mm.py index 28d33624c38bd6..193a88d763abf7 100644 --- a/scripts/gdb/linux/mm.py +++ b/scripts/gdb/linux/mm.py @@ -70,7 +70,6 @@ def __init__(self): self.SECTIONS_PER_ROOT = 1 self.NR_SECTION_ROOTS = DIV_ROUND_UP(self.NR_MEM_SECTIONS, self.SECTIONS_PER_ROOT) - self.SECTION_ROOT_MASK = self.SECTIONS_PER_ROOT - 1 try: self.SECTION_HAS_MEM_MAP = 1 << int(gdb.parse_and_eval('SECTION_HAS_MEM_MAP_BIT')) @@ -100,7 +99,7 @@ def SECTION_NR_TO_ROOT(self, sec): def __nr_to_section(self, nr): root = self.SECTION_NR_TO_ROOT(nr) mem_section = gdb.parse_and_eval("mem_section") - return mem_section[root][nr & self.SECTION_ROOT_MASK] + return mem_section[root][nr % self.SECTIONS_PER_ROOT] def pfn_to_section_nr(self, pfn): return pfn >> self.PFN_SECTION_SHIFT @@ -249,7 +248,6 @@ def __init__(self): self.SECTIONS_PER_ROOT = 1 self.NR_SECTION_ROOTS = DIV_ROUND_UP(self.NR_MEM_SECTIONS, self.SECTIONS_PER_ROOT) - self.SECTION_ROOT_MASK = self.SECTIONS_PER_ROOT - 1 self.SUBSECTION_SHIFT = 21 self.SEBSECTION_SIZE = 1 << self.SUBSECTION_SHIFT self.PFN_SUBSECTION_SHIFT = self.SUBSECTION_SHIFT - self.PAGE_SHIFT @@ -304,7 +302,7 @@ def SECTION_NR_TO_ROOT(self, sec): def __nr_to_section(self, nr): root = self.SECTION_NR_TO_ROOT(nr) mem_section = gdb.parse_and_eval("mem_section") - return mem_section[root][nr & self.SECTION_ROOT_MASK] + return mem_section[root][nr % self.SECTIONS_PER_ROOT] def pfn_to_section_nr(self, pfn): return pfn >> self.PFN_SECTION_SHIFT From 076ace93f23e5759ef0da1dbe24a8b5f65849562 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:45:53 +0800 Subject: [PATCH 564/857] mm/sparse-vmemmap: rename HVO order macros The macros VMEMMAP_TAIL_MIN_ORDER and NR_VMEMMAP_TAILS describe the order range where HVO can be applied, but their names tie that range to the tail-page cache implementation. Rename them with a VMEMMAP_OPTIMIZATION prefix and use the new names in the HVO paths. This makes the code describe the optimization requirements rather than the tail-page cache implementation detail. No functional change intended. Link: https://lore.kernel.org/20260825084608.47437-3-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/mmzone.h | 17 +++++++++-------- mm/hugetlb.c | 4 ++-- mm/hugetlb_vmemmap.c | 2 +- mm/sparse-vmemmap.c | 4 ++-- 4 files changed, 14 insertions(+), 13 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 0a2428714108c4..5fb9b37819d550 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -107,13 +107,14 @@ is_power_of_2(sizeof(struct page)) ? \ MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) -/* - * vmemmap optimization (like HVO) is only possible for page orders that fill - * two or more pages with struct pages. - */ -#define VMEMMAP_TAIL_MIN_ORDER (ilog2(2 * PAGE_SIZE / sizeof(struct page))) -#define __NR_VMEMMAP_TAILS (MAX_FOLIO_ORDER - VMEMMAP_TAIL_MIN_ORDER + 1) -#define NR_VMEMMAP_TAILS (__NR_VMEMMAP_TAILS > 0 ? __NR_VMEMMAP_TAILS : 0) +/* The number of struct pages covered by the retained vmemmap pages with HVO enabled. */ +#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES (PAGE_SIZE / sizeof(struct page)) +#define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) + +#define __VMEMMAP_OPTIMIZATION_NR_ORDERS \ + (MAX_FOLIO_ORDER - VMEMMAP_OPTIMIZATION_MIN_ORDER + 1) +#define VMEMMAP_OPTIMIZATION_NR_ORDERS \ + (__VMEMMAP_OPTIMIZATION_NR_ORDERS > 0 ? __VMEMMAP_OPTIMIZATION_NR_ORDERS : 0) enum migratetype { MIGRATE_UNMOVABLE, @@ -1158,7 +1159,7 @@ struct zone { atomic_long_t vm_stat[NR_VM_ZONE_STAT_ITEMS]; atomic_long_t vm_numa_event[NR_VM_NUMA_EVENT_ITEMS]; #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct page *vmemmap_tails[NR_VMEMMAP_TAILS]; + struct page *vmemmap_tails[VMEMMAP_OPTIMIZATION_NR_ORDERS]; #endif } ____cacheline_internodealigned_in_smp; diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 871b81670cbe9c..f99b1d9d079eef 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3357,7 +3357,7 @@ void __init hugetlb_bootmem_struct_page_init(void) struct zone *zone; for_each_zone(zone) { - for (int i = 0; i < NR_VMEMMAP_TAILS; i++) { + for (int i = 0; i < VMEMMAP_OPTIMIZATION_NR_ORDERS; i++) { struct page *tail, *p; unsigned int order; @@ -3365,7 +3365,7 @@ void __init hugetlb_bootmem_struct_page_init(void) if (!tail) continue; - order = i + VMEMMAP_TAIL_MIN_ORDER; + order = i + VMEMMAP_OPTIMIZATION_MIN_ORDER; p = page_to_virt(tail); /* * prep_and_add_bootmem_folios() can access pageblock diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index 917db0984143c2..ae8fdaa4211891 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -494,7 +494,7 @@ static bool vmemmap_should_optimize_folio(const struct hstate *h, struct folio * static struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) { - const unsigned int idx = order - VMEMMAP_TAIL_MIN_ORDER; + const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; struct page *tail, *p; int node = zone_to_nid(zone); diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 5a2469fb1838c7..aa6a4a2fae9886 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -329,12 +329,12 @@ static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone * unsigned int idx; int node = zone_to_nid(zone); - if (WARN_ON_ONCE(order < VMEMMAP_TAIL_MIN_ORDER)) + if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) return NULL; if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) return NULL; - idx = order - VMEMMAP_TAIL_MIN_ORDER; + idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; tail = zone->vmemmap_tails[idx]; if (tail) return tail; From b93077c66d58d316b90c21988303d08e78a400da Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:45:54 +0800 Subject: [PATCH 565/857] mm/mm_init: skip initializing shared vmemmap tail pages memmap_init_range() initializes every struct page in the target range. For compound pages with vmemmap optimization, the tail struct pages are backed by a shared vmemmap page. Initializing those tail struct pages would overwrite the shared vmemmap page contents, requiring users such as HugeTLB to restore the metadata afterwards. Track the compound order for HVO-backed sections and use that metadata to detect struct pages that fall into the shared tail vmemmap range. Skip those shared tail pages in memmap_init_range(), then initialize pageblock migratetypes for the processed range with a helper after the per-page initialization loop. Keep direct mem_section access inside sparse helpers by exposing pfn_to_section_order() to users that only need the order associated with a PFN. This lets memmap_init_range() skip shared tail vmemmap pages without exposing __pfn_to_section() to !SPARSEMEM builds. This is a preparatory change for consolidating handling across users of vmemmap optimization, and it also avoids redundant initialization of shared tail vmemmap pages during early boot. That early-boot benefit appears only once HugeTLB is switched to this common handling, since HugeTLB is the early-boot user that creates those shared tail vmemmap pages. Link: https://lore.kernel.org/20260825084608.47437-4-songmuchun@bytedance.com Signed-off-by: Muchun Song Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/mmzone.h | 8 ++++++++ mm/mm_init.c | 40 +++++++++++++++++++++++----------------- mm/sparse.h | 33 +++++++++++++++++++++++++++++++++ 3 files changed, 64 insertions(+), 17 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 5fb9b37819d550..df31cac123119a 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2022,6 +2022,14 @@ struct mem_section { unsigned long section_mem_map; struct mem_section_usage *usage; +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP + /* + * Normally, sections hold regular (order-0) pages. However, for + * sections with HVO enabled, this tracks the compound page order + * to enable deduplication of redundant vmemmap pages. + */ + unsigned int order; +#endif #ifdef CONFIG_PAGE_EXTENSION /* * If SPARSEMEM, pgdat doesn't have page_ext pointer. We use diff --git a/mm/mm_init.c b/mm/mm_init.c index 1533aebafb6885..f801fd486085e0 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -29,6 +29,7 @@ #include #include #include +#include #include #include #include @@ -677,21 +678,19 @@ static inline void fixup_hashdist(void) static inline void fixup_hashdist(void) {} #endif /* CONFIG_NUMA */ -#if defined(CONFIG_ZONE_DEVICE) || defined(CONFIG_DEFERRED_STRUCT_PAGE_INIT) static __meminit void pageblock_migratetype_init_range(unsigned long pfn, - unsigned long nr_pages, int migratetype, bool atomic) + unsigned long nr_pages, int migratetype, bool isolate, bool atomic) { const unsigned long end = pfn + nr_pages; for (pfn = pageblock_align(pfn); pfn < end; pfn += pageblock_nr_pages) { enum migratetype mt = kho_scratch_migratetype(pfn, migratetype); - init_pageblock_migratetype(pfn_to_page(pfn), mt, false); - if (!atomic && IS_ALIGNED(pfn, PAGES_PER_SECTION)) + init_pageblock_migratetype(pfn_to_page(pfn), mt, isolate); + if (!atomic && IS_ALIGNED(pfn, PFN_DOWN(SZ_1G))) cond_resched(); } } -#endif #ifdef CONFIG_DEFERRED_STRUCT_PAGE_INIT static inline void pgdat_set_deferred_range(pg_data_t *pgdat) @@ -886,6 +885,17 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone } } + /* + * Vmemmap-optimizable PFNs are backed by shared tail struct pages, + * which have already been initialized during vmemmap population. + */ + if (vmemmap_optimizable_pfn(pfn)) { + unsigned int order = pfn_to_section_order(pfn); + + pfn = min(ALIGN(pfn, 1UL << order), end_pfn); + continue; + } + page = pfn_to_page(pfn); __init_single_page(page, pfn, zone, nid); if (context == MEMINIT_HOTPLUG) { @@ -897,19 +907,13 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone __SetPageOffline(page); } - /* - * Usually, we want to mark the pageblock MIGRATE_MOVABLE, - * such that unmovable allocations won't be scattered all - * over the place during system boot. - */ - if (pageblock_aligned(pfn)) { - enum migratetype mt = kho_scratch_migratetype(pfn, migratetype); - - init_pageblock_migratetype(page, mt, isolate_pageblock); + if (pageblock_aligned(pfn)) cond_resched(); - } pfn++; } + + pageblock_migratetype_init_range(start_pfn, pfn - start_pfn, migratetype, + isolate_pageblock, /* atomic */ false); } static void __init memmap_init_zone_range(struct zone *zone, @@ -1112,7 +1116,8 @@ void __ref memmap_init_zone_device(struct zone *zone, compound_nr_pages(pfn, altmap, pgmap)); } - pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, false); + pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, + /* isolate */ false, /* atomic */ false); pr_debug("%s initialised %lu pages in %ums\n", __func__, nr_pages, jiffies_to_msecs(jiffies - start)); @@ -1921,7 +1926,8 @@ static void __init deferred_free_pages(unsigned long pfn, if (!nr_pages) return; - pageblock_migratetype_init_range(pfn, nr_pages, MIGRATE_MOVABLE, true); + pageblock_migratetype_init_range(pfn, nr_pages, MIGRATE_MOVABLE, + /* isolate */ false, /* atomic */ true); page = pfn_to_page(pfn); diff --git a/mm/sparse.h b/mm/sparse.h index 3b744667a7e624..1fcda8a1c270d7 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -10,6 +10,39 @@ #include +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +static inline unsigned int section_order(const struct mem_section *section) +{ + return section->order; +} + +static inline unsigned int pfn_to_section_order(unsigned long pfn) +{ + return section_order(__pfn_to_section(pfn)); +} +#else +static inline unsigned int section_order(const struct mem_section *section) +{ + return 0; +} + +static inline unsigned int pfn_to_section_order(unsigned long pfn) +{ + return 0; +} +#endif + +static inline bool vmemmap_optimizable_pfn(unsigned long pfn) +{ + const unsigned int order = pfn_to_section_order(pfn); + const unsigned long nr_pages = 1UL << order; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; +} + /* * mm/sparse.c */ From 44660556736a2769d53478682c51286425c2c887 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:45:55 +0800 Subject: [PATCH 566/857] mm/sparse-vmemmap: initialize shared tail vmemmap pages on allocation The shared tail vmemmap page allocated in vmemmap_get_tail() used to be left uninitialized, because memmap_init_range() would later overwrite it. That forced users such as HugeTLB to defer the initialization to their own setup paths. Now that memmap_init_range() skips shared tail vmemmap pages, initialize them immediately in vmemmap_get_tail() with init_compound_tail() instead. This moves the initialization to the point where the shared tail page is allocated and avoids relying on deferred handling in individual users. The remaining deferred initialization in HugeTLB will be removed once it switches to the section-based vmemmap optimization mechanism. Link: https://lore.kernel.org/20260825084608.47437-5-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/sparse-vmemmap.c | 12 ++---------- 1 file changed, 2 insertions(+), 10 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index aa6a4a2fae9886..107215cf8488ce 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -338,19 +338,11 @@ static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone * tail = zone->vmemmap_tails[idx]; if (tail) return tail; - - /* - * Only allocate the page, but do not initialize it. - * - * Any initialization done here will be overwritten by memmap_init(). - * - * hugetlb_bootmem_struct_page_init() will take care of initialization - * after memmap_init(). - */ - p = vmemmap_alloc_block_zero(PAGE_SIZE, node); if (!p) return NULL; + for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) + init_compound_tail(p + i, NULL, order, zone); tail = virt_to_page(p); zone->vmemmap_tails[idx] = tail; From fa43a48c7f57c81323561c21fd88fb66468dc794 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:45:56 +0800 Subject: [PATCH 567/857] mm/sparse-vmemmap: support section-based vmemmap accounting section_nr_vmemmap_pages() can account ordinary sections and DAX sections, but section-based vmemmap optimization keeps its compound order in struct mem_section and retains a different number of vmemmap pages. Teach section_nr_vmemmap_pages() to recognize section-based optimized sections and calculate their vmemmap page count from the section order and the HVO retained page count. Link: https://lore.kernel.org/20260825084608.47437-6-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/mmzone.h | 6 ++++-- mm/sparse-vmemmap.c | 10 ++++++---- mm/sparse.h | 16 ++++++++++++++++ 3 files changed, 26 insertions(+), 6 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index df31cac123119a..177455d980646e 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -107,8 +107,10 @@ is_power_of_2(sizeof(struct page)) ? \ MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) -/* The number of struct pages covered by the retained vmemmap pages with HVO enabled. */ -#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES (PAGE_SIZE / sizeof(struct page)) +/* The number of retained vmemmap pages with HVO enabled. */ +#define VMEMMAP_OPTIMIZATION_PAGES 1 +#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES \ + (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) #define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) #define __VMEMMAP_OPTIMIZATION_NR_ORDERS \ diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 107215cf8488ce..b7abc5494bb992 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -649,24 +649,26 @@ void offline_mem_sections(unsigned long start_pfn, unsigned long end_pfn) static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { - const unsigned int order = pgmap ? pgmap->vmemmap_shift : 0; + const struct mem_section *ms = __pfn_to_section(pfn); + const int order = pgmap ? pgmap->vmemmap_shift : section_order(ms); + const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; const unsigned long pages_per_compound = 1UL << order; VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); - if (!vmemmap_can_optimize(altmap, pgmap)) + if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); if (order < PFN_SECTION_SHIFT) { VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); - return VMEMMAP_RESERVE_NR * nr_pages / pages_per_compound; + return vmemmap_pages * nr_pages / pages_per_compound; } VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); if (IS_ALIGNED(pfn, pages_per_compound)) - return VMEMMAP_RESERVE_NR; + return vmemmap_pages; return 0; } diff --git a/mm/sparse.h b/mm/sparse.h index 1fcda8a1c270d7..02ed0f34eac8e5 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -43,6 +43,17 @@ static inline bool vmemmap_optimizable_pfn(unsigned long pfn) return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; } +static inline bool vmemmap_optimizable_order(unsigned int order) +{ + if (!IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP)) + return false; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return order >= VMEMMAP_OPTIMIZATION_MIN_ORDER; +} + /* * mm/sparse.c */ @@ -86,6 +97,11 @@ static inline size_t mem_section_usage_size(void) return struct_size_t(struct mem_section_usage, pageblock_flags, BITS_TO_LONGS(SECTION_BLOCKFLAGS_BITS)); } + +static inline bool section_vmemmap_optimizable(const struct mem_section *ms) +{ + return vmemmap_optimizable_order(section_order(ms)); +} #else static inline void sparse_init(void) {} #endif /* CONFIG_SPARSEMEM */ From 51923b156d06897acc587d19d6a73d3aef272756 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:45:57 +0800 Subject: [PATCH 568/857] mm/mm_init: factor out pfn_to_zone() pfn_to_zone() in hugetlb_vmemmap.c duplicates the zone lookup logic in __init_deferred_page(). Move it to mm_init.c, declare it in mm/mm_init.h, and reuse it from __init_deferred_page() and HugeTLB early vmemmap initialization instead of open-coding the zone walk there. Link: https://lore.kernel.org/20260825084608.47437-7-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/hugetlb_vmemmap.c | 17 ++--------------- mm/mm_init.c | 28 ++++++++++++++++++---------- mm/mm_init.h | 1 + 3 files changed, 21 insertions(+), 25 deletions(-) diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index ae8fdaa4211891..c48fcea076a522 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -19,6 +19,7 @@ #include #include "hugetlb_vmemmap.h" #include "internal.h" +#include "mm_init.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -744,20 +745,6 @@ static bool vmemmap_should_optimize_bootmem_page(struct huge_bootmem_page *m) return true; } -static struct zone *pfn_to_zone(unsigned nid, unsigned long pfn) -{ - struct zone *zone; - enum zone_type zone_type; - - for (zone_type = 0; zone_type < MAX_NR_ZONES; zone_type++) { - zone = &NODE_DATA(nid)->node_zones[zone_type]; - if (zone_spans_pfn(zone, pfn)) - return zone; - } - - return NULL; -} - /* * Initialize memmap section for a gigantic page, HVO-style. */ @@ -787,7 +774,7 @@ void __init hugetlb_vmemmap_init_early(int nid) map = pfn_to_page(pfn); start = (unsigned long)map; end = start + hugetlb_vmemmap_size(m->hstate); - zone = pfn_to_zone(nid, pfn); + zone = pfn_to_zone(pfn, nid); if (vmemmap_populate_hvo(start, end, huge_page_order(m->hstate), zone, HUGETLB_VMEMMAP_RESERVE_SIZE)) diff --git a/mm/mm_init.c b/mm/mm_init.c index f801fd486085e0..0317fb781e77d6 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -692,6 +692,20 @@ static __meminit void pageblock_migratetype_init_range(unsigned long pfn, } } +struct zone __meminit *pfn_to_zone(unsigned long pfn, int nid) +{ + pg_data_t *pgdat = NODE_DATA(nid); + + for (enum zone_type zone_type = 0; zone_type < MAX_NR_ZONES; zone_type++) { + struct zone *zone = &pgdat->node_zones[zone_type]; + + if (zone_spans_pfn(zone, pfn)) + return zone; + } + + return NULL; +} + #ifdef CONFIG_DEFERRED_STRUCT_PAGE_INIT static inline void pgdat_set_deferred_range(pg_data_t *pgdat) { @@ -750,20 +764,14 @@ defer_init(int nid, unsigned long pfn, unsigned long end_pfn) static void __meminit __init_deferred_page(unsigned long pfn, int nid) { - pg_data_t *pgdat = NODE_DATA(nid); - int zid; + struct zone *zone; if (early_page_initialised(pfn, nid)) return; - for (zid = 0; zid < MAX_NR_ZONES; zid++) { - struct zone *zone = &pgdat->node_zones[zid]; - - if (zone_spans_pfn(zone, pfn)) - break; - } - __init_single_page(pfn_to_page(pfn), pfn, zid, nid); - + zone = pfn_to_zone(pfn, nid); + __init_single_page(pfn_to_page(pfn), pfn, + zone ? zone_idx(zone) : MAX_NR_ZONES, nid); if (pageblock_aligned(pfn)) { enum migratetype mt = kho_scratch_migratetype(pfn, MIGRATE_MOVABLE); diff --git a/mm/mm_init.h b/mm/mm_init.h index 39f75df9be1c26..c9fc35e7e9f1f3 100644 --- a/mm/mm_init.h +++ b/mm/mm_init.h @@ -39,6 +39,7 @@ void memmap_init_range(unsigned long size, int nid, unsigned long zone, enum meminit_context context, struct vmem_altmap *altmap, int migratetype, bool isolate_pageblock); +struct zone *pfn_to_zone(unsigned long pfn, int nid); #if defined CONFIG_COMPACTION || defined CONFIG_CMA /* Free whole pageblock and set its migration type to MIGRATE_CMA. */ From 2e2698eaefa40034587a3c39d1023f300c3bc41a Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:45:58 +0800 Subject: [PATCH 569/857] mm/sparse-vmemmap: move helpers ahead of future callers Prepare for section-based vmemmap optimization by moving helpers that follow-up changes will use. vmemmap_get_tail() will be called from the PTE population path. section_nr_vmemmap_pages() will be made visible outside the memory hotplug code and called from sparse_init_nid(). Move vmemmap_alloc_block_zero() together with vmemmap_get_tail(), since the tail helper depends on it. Move section_nr_vmemmap_pages() earlier into its own CONFIG_MEMORY_HOTPLUG block. That lets the later patch change its visibility and callers without also moving the function body. No functional change is intended. Link: https://lore.kernel.org/20260825084608.47437-8-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/sparse-vmemmap.c | 134 +++++++++++++++++++++++--------------------- 1 file changed, 69 insertions(+), 65 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index b7abc5494bb992..ea3cacec0a798f 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -148,6 +148,75 @@ void __meminit vmemmap_verify(pte_t *pte, int node, start, end - 1); } +#ifdef CONFIG_MEMORY_HOTPLUG +static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, + struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +{ + const struct mem_section *ms = __pfn_to_section(pfn); + const int order = pgmap ? pgmap->vmemmap_shift : section_order(ms); + const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; + const unsigned long pages_per_compound = 1UL << order; + + VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); + VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); + + if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) + return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); + + if (order < PFN_SECTION_SHIFT) { + VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); + return vmemmap_pages * nr_pages / pages_per_compound; + } + + VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); + + if (IS_ALIGNED(pfn, pages_per_compound)) + return vmemmap_pages; + + return 0; +} +#endif + +static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) +{ + void *p = vmemmap_alloc_block(size, node); + + if (!p) + return NULL; + memset(p, 0, size); + + return p; +} + +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +{ + struct page *p, *tail; + unsigned int idx; + int node = zone_to_nid(zone); + + if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) + return NULL; + if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) + return NULL; + + idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; + tail = zone->vmemmap_tails[idx]; + if (tail) + return tail; + p = vmemmap_alloc_block_zero(PAGE_SIZE, node); + if (!p) + return NULL; + for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) + init_compound_tail(p + i, NULL, order, zone); + + tail = virt_to_page(p); + zone->vmemmap_tails[idx] = tail; + + return tail; +} +#endif + static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, int node, struct vmem_altmap *altmap, unsigned long ptpfn, unsigned long flags) @@ -181,17 +250,6 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in return pte; } -static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) -{ - void *p = vmemmap_alloc_block(size, node); - - if (!p) - return NULL; - memset(p, 0, size); - - return p; -} - static pmd_t * __meminit vmemmap_pmd_populate(pud_t *pud, unsigned long addr, int node) { pmd_t *pmd = pmd_offset(pud, addr); @@ -323,33 +381,6 @@ void vmemmap_wrprotect_hvo(unsigned long addr, unsigned long end, } #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP -static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) -{ - struct page *p, *tail; - unsigned int idx; - int node = zone_to_nid(zone); - - if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) - return NULL; - if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) - return NULL; - - idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; - tail = zone->vmemmap_tails[idx]; - if (tail) - return tail; - p = vmemmap_alloc_block_zero(PAGE_SIZE, node); - if (!p) - return NULL; - for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) - init_compound_tail(p + i, NULL, order, zone); - - tail = virt_to_page(p); - zone->vmemmap_tails[idx] = tail; - - return tail; -} - int __meminit vmemmap_populate_hvo(unsigned long addr, unsigned long end, unsigned int order, struct zone *zone, unsigned long headsize) @@ -646,33 +677,6 @@ void offline_mem_sections(unsigned long start_pfn, unsigned long end_pfn) } } -static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap) -{ - const struct mem_section *ms = __pfn_to_section(pfn); - const int order = pgmap ? pgmap->vmemmap_shift : section_order(ms); - const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; - const unsigned long pages_per_compound = 1UL << order; - - VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); - VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); - - if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) - return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); - - if (order < PFN_SECTION_SHIFT) { - VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); - return vmemmap_pages * nr_pages / pages_per_compound; - } - - VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); - - if (IS_ALIGNED(pfn, pages_per_compound)) - return vmemmap_pages; - - return 0; -} - static struct page * __meminit populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) From af835f6e7a156e158b29e67b3c3615b494a39956 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:45:59 +0800 Subject: [PATCH 570/857] mm/sparse-vmemmap: support section-based vmemmap optimization Teach sparse-vmemmap population code to use the compound page order when deciding whether a vmemmap page can be optimized. With this information, the common sparse-vmemmap population path can allocate or reuse shared tail vmemmap pages directly instead of relying on HugeTLB-specific handling. This centralizes vmemmap optimization logic in the sparse-vmemmap code, based on section metadata, and prepares for sharing the same mechanism across different users of vmemmap optimization, including HugeTLB and DAX. Link: https://lore.kernel.org/20260825084608.47437-9-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/sparse-vmemmap.c | 46 +++++++++++++++++++++++++++++++++++++-------- mm/sparse.c | 4 ++-- mm/sparse.h | 7 +++++++ 3 files changed, 47 insertions(+), 10 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index ea3cacec0a798f..8d39196f6d936e 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -148,8 +148,7 @@ void __meminit vmemmap_verify(pte_t *pte, int node, start, end - 1); } -#ifdef CONFIG_MEMORY_HOTPLUG -static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, +int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { const struct mem_section *ms = __pfn_to_section(pfn); @@ -175,7 +174,6 @@ static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long n return 0; } -#endif static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) { @@ -215,19 +213,44 @@ static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone * return tail; } +#else +static inline struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +{ + return NULL; +} #endif +static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, + struct vmem_altmap *altmap) +{ + struct zone *zone; + struct page *page; + const unsigned int order = pfn_to_section_order(pfn); + + if (!vmemmap_optimizable_pfn(pfn)) + return vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); + + zone = pfn_to_zone(pfn, node); + page = vmemmap_get_tail(order, zone); + if (!page) + return NULL; + + return page_address(page); +} + static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, int node, struct vmem_altmap *altmap, unsigned long ptpfn, unsigned long flags) { pte_t *pte = pte_offset_kernel(pmd, addr); + unsigned long pfn = page_to_pfn((struct page *)addr); + if (pte_none(ptep_get(pte))) { pte_t entry; - void *p; if (ptpfn == (unsigned long)-1) { - p = vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); + void *p = vmemmap_alloc_pte(pfn, node, altmap); + if (!p) return NULL; ptpfn = PHYS_PFN(__pa(p)); @@ -246,7 +269,8 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in } entry = pfn_pte(ptpfn, PAGE_KERNEL); set_pte_at(&init_mm, addr, pte, entry); - } + } else if (WARN_ON_ONCE(vmemmap_optimizable_pfn(pfn))) + return NULL; return pte; } @@ -435,6 +459,9 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, pmd_t *pmd; for (addr = start; addr < end; addr = next) { + unsigned long pfn = page_to_pfn((struct page *)addr); + const struct mem_section *ms = __pfn_to_section(pfn); + next = pmd_addr_end(addr, end); pgd = vmemmap_pgd_populate(addr, node); @@ -450,7 +477,7 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, return -ENOMEM; pmd = pmd_offset(pud, addr); - if (pmd_none(pmdp_get(pmd))) { + if (pmd_none(pmdp_get(pmd)) && !section_vmemmap_optimizable(ms)) { void *p; p = vmemmap_alloc_block_buf(PMD_SIZE, node, altmap); @@ -468,8 +495,11 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, */ return -ENOMEM; } - } else if (vmemmap_check_pmd(pmd, node, addr, next)) + } else if (vmemmap_check_pmd(pmd, node, addr, next)) { + if (WARN_ON_ONCE(section_vmemmap_optimizable(ms))) + return -EOPNOTSUPP; continue; + } if (vmemmap_populate_basepages(addr, next, node, altmap)) return -ENOMEM; } diff --git a/mm/sparse.c b/mm/sparse.c index c84b4c7b8c7069..e6cb67ca9c8d10 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -305,8 +305,8 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, nid, NULL, NULL); if (!map) panic("Failed to allocate memmap for section %lu\n", pnum); - memmap_boot_pages_add(DIV_ROUND_UP(PAGES_PER_SECTION * sizeof(struct page), - PAGE_SIZE)); + memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, + NULL, NULL)); sparse_init_early_section(nid, map, pnum, 0); } } diff --git a/mm/sparse.h b/mm/sparse.h index 02ed0f34eac8e5..e427d72e9c2d98 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -111,8 +111,15 @@ static inline void sparse_init(void) {} */ #ifdef CONFIG_SPARSEMEM_VMEMMAP void sparse_init_subsection_map(void); +int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, + struct vmem_altmap *altmap, struct dev_pagemap *pgmap); #else static inline void sparse_init_subsection_map(void) {} +static inline int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, + struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +{ + return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); +} #endif /* CONFIG_SPARSEMEM_VMEMMAP */ #endif /* __MM_SPARSE_H */ From 20d5f30dca4c3fa342617658c5227f967bebfac1 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:46:00 +0800 Subject: [PATCH 571/857] mm/sparse: initialize memory sections earlier Upcoming HugeTLB bootmem changes need sparsemem section metadata before the HugeTLB bootmem allocation path runs. The memory sections are initialized from sparse_init(), which is called too late for that setup. Move the code that initializes sparsemem section metadata for memblock ranges into mm_core_init_early(), before free_area_init() and the HugeTLB bootmem setup. Rename the helper to sparse_sections_init() so the new caller describes the sparsemem-specific initialization step. This is a preparatory change. Link: https://lore.kernel.org/20260825084608.47437-10-songmuchun@bytedance.com Signed-off-by: Muchun Song Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/mm_init.c | 1 + mm/sparse.c | 10 ++-------- mm/sparse.h | 2 ++ 3 files changed, 5 insertions(+), 8 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 0317fb781e77d6..e2a16d83363557 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -2642,6 +2642,7 @@ void __init mm_core_init_early(void) { kho_memory_init_early(); + sparse_sections_init(); free_area_init(); hugetlb_cma_reserve(); diff --git a/mm/sparse.c b/mm/sparse.c index e6cb67ca9c8d10..439802e6a6ad78 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -191,12 +191,8 @@ static void __init memory_present(int nid, unsigned long start, unsigned long en } } -/* - * Mark all memblocks as present using memory_present(). - * This is a convenience function that is useful to mark all of the systems - * memory as present during initialization. - */ -static void __init memblocks_present(void) +/* Initialize memory section metadata for all system memory. */ +void __init sparse_sections_init(void) { unsigned long start, end; int i, nid; @@ -322,8 +318,6 @@ void __init sparse_init(void) unsigned long pnum_end, pnum_begin, map_count = 1; int nid_begin; - memblocks_present(); - if (compound_info_has_mask()) { VM_WARN_ON_ONCE(!IS_ALIGNED((unsigned long) pfn_to_page(0), MAX_FOLIO_VMEMMAP_ALIGN)); diff --git a/mm/sparse.h b/mm/sparse.h index e427d72e9c2d98..e4617f9c8876c5 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -59,6 +59,7 @@ static inline bool vmemmap_optimizable_order(unsigned int order) */ #ifdef CONFIG_SPARSEMEM void sparse_init(void); +void sparse_sections_init(void); int sparse_index_init(unsigned long section_nr, int nid); static inline void sparse_init_one_section(struct mem_section *ms, @@ -104,6 +105,7 @@ static inline bool section_vmemmap_optimizable(const struct mem_section *ms) } #else static inline void sparse_init(void) {} +static inline void sparse_sections_init(void) {} #endif /* CONFIG_SPARSEMEM */ /* From 93f1908e031055e5c5111dd771560d28b45b8e6f Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:46:01 +0800 Subject: [PATCH 572/857] mm/hugetlb: switch HugeTLB to section-based vmemmap optimization HugeTLB bootmem vmemmap optimization still carries its own early setup path, including pre-populating optimized mappings before the generic sparse-vmemmap code runs. Now that the section-based vmemmap optimization can derive HugeTLB vmemmap deduplication from section metadata, HugeTLB only needs to mark the bootmem huge page range with the appropriate order. The generic sparse-vmemmap population path can then allocate and map the shared tail vmemmap pages without any HugeTLB-specific early population code. Do that by setting the section order when a bootmem huge page is allocated and dropping the dedicated pre-HVO helpers and related special-casing. This removes duplicate early setup logic and switches HugeTLB to the section-based vmemmap optimization path. Link: https://lore.kernel.org/20260825084608.47437-11-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/hugetlb.h | 1 - include/linux/mm.h | 3 -- mm/hugetlb.c | 30 ++------------ mm/hugetlb_vmemmap.c | 90 +++-------------------------------------- mm/hugetlb_vmemmap.h | 14 +++---- mm/sparse-vmemmap.c | 31 -------------- mm/sparse.h | 27 +++++++++++++ 7 files changed, 42 insertions(+), 154 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 16c4c4caa126ce..fe28f98e1b220e 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -171,7 +171,6 @@ struct address_space *hugetlb_folio_mapping_lock_write(struct folio *folio); extern int movable_gigantic_pages __read_mostly; extern int sysctl_hugetlb_shm_group __read_mostly; -extern struct list_head huge_boot_pages[MAX_NUMNODES]; void hugetlb_bootmem_struct_page_init(void); void hugetlb_bootmem_alloc(void); diff --git a/include/linux/mm.h b/include/linux/mm.h index dd09c438fa23ec..441bd39eab7343 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5159,9 +5159,6 @@ int vmemmap_populate_hugepages(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); int vmemmap_populate(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); -int vmemmap_populate_hvo(unsigned long start, unsigned long end, - unsigned int order, struct zone *zone, - unsigned long headsize); void vmemmap_wrprotect_hvo(unsigned long start, unsigned long end, int node, unsigned long headsize); void vmemmap_populate_print_last(void); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index f99b1d9d079eef..524ca0c2b3ecea 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -52,6 +52,7 @@ #include "hugetlb_cma.h" #include "hugetlb_internal.h" #include "mm_init.h" +#include "sparse.h" #include int hugetlb_max_hstate __read_mostly; @@ -59,7 +60,7 @@ unsigned int default_hstate_idx; struct hstate hstates[HUGE_MAX_HSTATE]; __initdata nodemask_t hugetlb_bootmem_nodes; -__initdata struct list_head huge_boot_pages[MAX_NUMNODES]; +static struct list_head huge_boot_pages[MAX_NUMNODES] __initdata; /* * Due to ordering constraints across the init code for various @@ -3150,6 +3151,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) } else { list_add_tail(&m->list, &huge_boot_pages[nid]); m->flags |= HUGE_BOOTMEM_ZONES_VALID; + hugetlb_vmemmap_optimize_bootmem_page(m); /* * Only initialize the head struct page in memmap_init_reserved_pages, * rest of the struct pages will be initialized by the HugeTLB @@ -3310,6 +3312,7 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) * this folio. */ folio_set_hugetlb_vmemmap_optimized(folio); + section_set_order_range(folio_pfn(folio), folio_nr_pages(folio), 0); if (hugetlb_bootmem_page_earlycma(m)) folio_set_hugetlb_cma(folio); @@ -3353,31 +3356,6 @@ void __init hugetlb_bootmem_struct_page_init(void) .max_threads = num_node_state(N_MEMORY), .numa_aware = true, }; -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct zone *zone; - - for_each_zone(zone) { - for (int i = 0; i < VMEMMAP_OPTIMIZATION_NR_ORDERS; i++) { - struct page *tail, *p; - unsigned int order; - - tail = zone->vmemmap_tails[i]; - if (!tail) - continue; - - order = i + VMEMMAP_OPTIMIZATION_MIN_ORDER; - p = page_to_virt(tail); - /* - * prep_and_add_bootmem_folios() can access pageblock - * flags on bootmem HugeTLB pages, so initialize the - * shared tail struct pages here before bootmem folios - * start using them. - */ - for (int j = 0; j < PAGE_SIZE / sizeof(struct page); j++) - init_compound_tail(p + j, NULL, order, zone); - } - } -#endif padata_do_multithreaded(&job); } diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index c48fcea076a522..7293706b532f38 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -18,8 +18,7 @@ #include #include "hugetlb_vmemmap.h" -#include "internal.h" -#include "mm_init.h" +#include "sparse.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -706,95 +705,18 @@ void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head __hugetlb_vmemmap_optimize_folios(h, folio_list, true); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT - -/* Return true of a bootmem allocated HugeTLB page should be pre-HVO-ed */ -static bool vmemmap_should_optimize_bootmem_page(struct huge_bootmem_page *m) +void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) { - unsigned long section_size, psize, pmd_vmemmap_size; - phys_addr_t paddr; - - if (!READ_ONCE(vmemmap_optimize_enabled)) - return false; - - if (!hugetlb_vmemmap_optimizable(m->hstate)) - return false; - - psize = huge_page_size(m->hstate); - paddr = virt_to_phys(m); - - /* - * Pre-HVO only works if the bootmem huge page - * is aligned to the section size. - */ - section_size = (1UL << PA_SECTION_SHIFT); - if (!IS_ALIGNED(paddr, section_size) || - !IS_ALIGNED(psize, section_size)) - return false; - - /* - * The pre-HVO code does not deal with splitting PMDS, - * so the bootmem page must be aligned to the number - * of base pages that can be mapped with one vmemmap PMD. - */ - pmd_vmemmap_size = (PMD_SIZE / (sizeof(struct page))) << PAGE_SHIFT; - if (!IS_ALIGNED(paddr, pmd_vmemmap_size) || - !IS_ALIGNED(psize, pmd_vmemmap_size)) - return false; - - return true; -} - -/* - * Initialize memmap section for a gigantic page, HVO-style. - */ -void __init hugetlb_vmemmap_init_early(int nid) -{ - unsigned long psize, paddr, section_size; - unsigned long ns, i, pnum, pfn, nr_pages; - unsigned long start, end; - struct huge_bootmem_page *m = NULL; - void *map; + struct hstate *h = m->hstate; + unsigned long pfn = PHYS_PFN(__pa(m)); if (!READ_ONCE(vmemmap_optimize_enabled)) return; - section_size = (1UL << PA_SECTION_SHIFT); - - list_for_each_entry(m, &huge_boot_pages[nid], list) { - struct zone *zone; - - if (!vmemmap_should_optimize_bootmem_page(m)) - continue; - - nr_pages = pages_per_huge_page(m->hstate); - psize = nr_pages << PAGE_SHIFT; - paddr = virt_to_phys(m); - pfn = PHYS_PFN(paddr); - map = pfn_to_page(pfn); - start = (unsigned long)map; - end = start + hugetlb_vmemmap_size(m->hstate); - zone = pfn_to_zone(pfn, nid); - - if (vmemmap_populate_hvo(start, end, huge_page_order(m->hstate), - zone, HUGETLB_VMEMMAP_RESERVE_SIZE)) - panic("Failed to allocate memmap for HugeTLB page\n"); - memmap_boot_pages_add(DIV_ROUND_UP(HUGETLB_VMEMMAP_RESERVE_SIZE, PAGE_SIZE)); - - pnum = pfn_to_section_nr(pfn); - ns = psize / section_size; - - for (i = 0; i < ns; i++) { - sparse_init_early_section(nid, map, pnum, - SECTION_IS_VMEMMAP_PREINIT); - map += section_map_size(); - pnum++; - } - + section_set_order_range(pfn, pages_per_huge_page(h), huge_page_order(h)); + if (vmemmap_optimizable_order(pfn_to_section_order(pfn))) m->flags |= HUGE_BOOTMEM_HVO; - } } -#endif static const struct ctl_table hugetlb_vmemmap_sysctls[] = { { diff --git a/mm/hugetlb_vmemmap.h b/mm/hugetlb_vmemmap.h index 7ac49c52457ddf..20eb03df542a43 100644 --- a/mm/hugetlb_vmemmap.h +++ b/mm/hugetlb_vmemmap.h @@ -9,8 +9,7 @@ #ifndef _LINUX_HUGETLB_VMEMMAP_H #define _LINUX_HUGETLB_VMEMMAP_H #include -#include -#include +#include "internal.h" /* * Reserve one vmemmap page, all vmemmap addresses are mapped to it. See @@ -27,10 +26,7 @@ long hugetlb_vmemmap_restore_folios(const struct hstate *h, void hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio); void hugetlb_vmemmap_optimize_folios(struct hstate *h, struct list_head *folio_list); void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head *folio_list); -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -void hugetlb_vmemmap_init_early(int nid); -#endif - +void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m); static inline unsigned int hugetlb_vmemmap_size(const struct hstate *h) { @@ -76,13 +72,13 @@ static inline void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, { } -static inline void hugetlb_vmemmap_init_early(int nid) +static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h) { + return 0; } -static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h) +static inline void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) { - return 0; } #endif /* CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP */ diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 8d39196f6d936e..e48758805a8ff7 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -32,8 +32,6 @@ #include #include -#include "hugetlb_vmemmap.h" - /* * Flags for vmemmap_populate_range and friends. */ @@ -404,34 +402,6 @@ void vmemmap_wrprotect_hvo(unsigned long addr, unsigned long end, } } -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP -int __meminit vmemmap_populate_hvo(unsigned long addr, unsigned long end, - unsigned int order, struct zone *zone, - unsigned long headsize) -{ - unsigned long maddr; - struct page *tail; - pte_t *pte; - int node = zone_to_nid(zone); - - tail = vmemmap_get_tail(order, zone); - if (!tail) - return -ENOMEM; - - for (maddr = addr; maddr < addr + headsize; maddr += PAGE_SIZE) { - pte = vmemmap_populate_address(maddr, node, NULL, -1, 0); - if (!pte) - return -ENOMEM; - } - - /* - * Reuse the last page struct page mapped above for the rest. - */ - return vmemmap_populate_range(maddr, end, node, NULL, - page_to_pfn(tail), 0); -} -#endif - void __weak __meminit vmemmap_set_pmd(pmd_t *pmd, void *p, int node, unsigned long addr, unsigned long next) { @@ -634,7 +604,6 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, */ void __init sparse_vmemmap_init_nid_early(int nid) { - hugetlb_vmemmap_init_early(nid); } #endif diff --git a/mm/sparse.h b/mm/sparse.h index e4617f9c8876c5..049272aba84e56 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -16,6 +16,24 @@ static inline unsigned int section_order(const struct mem_section *section) return section->order; } +static inline void section_set_order(struct mem_section *section, unsigned int order) +{ + VM_WARN_ON(section_order(section) && order && section_order(section) != order); + section->order = order; +} + +static inline void section_set_order_range(unsigned long pfn, unsigned long nr_pages, + unsigned int order) +{ + unsigned long section_nr = pfn_to_section_nr(pfn); + + if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)) + return; + + for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++) + section_set_order(__nr_to_section(section_nr + i), order); +} + static inline unsigned int pfn_to_section_order(unsigned long pfn) { return section_order(__pfn_to_section(pfn)); @@ -26,6 +44,15 @@ static inline unsigned int section_order(const struct mem_section *section) return 0; } +static inline void section_set_order(struct mem_section *section, unsigned int order) +{ +} + +static inline void section_set_order_range(unsigned long pfn, unsigned long nr_pages, + unsigned int order) +{ +} + static inline unsigned int pfn_to_section_order(unsigned long pfn) { return 0; From 0a34fee1e9733a7b4dfec59a58f1bd8b4fa3dd97 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:46:02 +0800 Subject: [PATCH 573/857] mm/sparse-vmemmap: remove SPARSEMEM_VMEMMAP_PREINIT support SPARSEMEM_VMEMMAP_PREINIT existed only to support HugeTLB's early vmemmap optimization setup. Now that HugeTLB bootmem vmemmap optimization uses the common section-based sparse-vmemmap path, the sparse initialization code no longer needs a separate pre-initialization mechanism for vmemmap population. Remove the related Kconfig symbols, section flag, and empty early hook, so present sections always go through the normal sparse setup path. Link: https://lore.kernel.org/20260825084608.47437-12-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- arch/x86/Kconfig | 1 - fs/Kconfig | 1 - include/linux/mmzone.h | 25 ------------------------- mm/Kconfig | 5 ----- mm/sparse-vmemmap.c | 13 ------------- mm/sparse.c | 23 ++++++++--------------- 6 files changed, 8 insertions(+), 60 deletions(-) diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index 15fd9ec5ecacb7..7aa74bcc72f9db 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -150,7 +150,6 @@ config X86 select ARCH_WANT_LD_ORPHAN_WARN select ARCH_WANT_OPTIMIZE_DAX_VMEMMAP if X86_64 select ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP if X86_64 - select ARCH_WANT_HUGETLB_VMEMMAP_PREINIT if X86_64 select ARCH_WANTS_THP_SWAP if X86_64 select ARCH_HAS_PARANOID_L1D_FLUSH select ARCH_WANT_IRQS_OFF_ACTIVATE_MM diff --git a/fs/Kconfig b/fs/Kconfig index e05917adcd608e..d1c210c6508f0a 100644 --- a/fs/Kconfig +++ b/fs/Kconfig @@ -278,7 +278,6 @@ config HUGETLB_PAGE_OPTIMIZE_VMEMMAP def_bool HUGETLB_PAGE depends on ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP depends on SPARSEMEM_VMEMMAP - select SPARSEMEM_VMEMMAP_PREINIT if ARCH_WANT_HUGETLB_VMEMMAP_PREINIT config HUGETLB_PMD_PAGE_TABLE_SHARING def_bool HUGETLB_PAGE diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 177455d980646e..8bcb522645ba9e 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2095,9 +2095,6 @@ enum { SECTION_IS_EARLY_BIT, #ifdef CONFIG_ZONE_DEVICE SECTION_TAINT_ZONE_DEVICE_BIT, -#endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT - SECTION_IS_VMEMMAP_PREINIT_BIT, #endif SECTION_MAP_LAST_BIT, }; @@ -2109,9 +2106,6 @@ enum { #ifdef CONFIG_ZONE_DEVICE #define SECTION_TAINT_ZONE_DEVICE BIT(SECTION_TAINT_ZONE_DEVICE_BIT) #endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -#define SECTION_IS_VMEMMAP_PREINIT BIT(SECTION_IS_VMEMMAP_PREINIT_BIT) -#endif #define SECTION_MAP_MASK (~(BIT(SECTION_MAP_LAST_BIT) - 1)) #define SECTION_NID_SHIFT SECTION_MAP_LAST_BIT @@ -2166,24 +2160,6 @@ static inline int online_device_section(const struct mem_section *section) } #endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -static inline int preinited_vmemmap_section(const struct mem_section *section) -{ - return (section && - (section->section_mem_map & SECTION_IS_VMEMMAP_PREINIT)); -} - -void sparse_vmemmap_init_nid_early(int nid); -#else -static inline int preinited_vmemmap_section(const struct mem_section *section) -{ - return 0; -} -static inline void sparse_vmemmap_init_nid_early(int nid) -{ -} -#endif - static inline int online_section_nr(unsigned long nr) { return online_section(__nr_to_section(nr)); @@ -2385,7 +2361,6 @@ static inline unsigned long next_present_section_nr(unsigned long section_nr) #endif #else -#define sparse_vmemmap_init_nid_early(_nid) do {} while (0) #define pfn_in_present_section pfn_valid #endif /* CONFIG_SPARSEMEM */ diff --git a/mm/Kconfig b/mm/Kconfig index 604c58199acbf8..2c385f8b29445e 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -461,8 +461,6 @@ config SPARSEMEM_VMEMMAP pfn_to_page and page_to_pfn operations. This is the most efficient option when sufficient kernel resources are available. -config SPARSEMEM_VMEMMAP_PREINIT - bool # # Select this config option from the architecture Kconfig, if it is preferred # to enable the feature of HugeTLB/dev_dax vmemmap optimization. @@ -473,9 +471,6 @@ config ARCH_WANT_OPTIMIZE_DAX_VMEMMAP config ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP bool -config ARCH_WANT_HUGETLB_VMEMMAP_PREINIT - bool - config HAVE_MEMBLOCK_PHYS_MAP bool diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index e48758805a8ff7..e62e6aa07f1262 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -594,19 +594,6 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, return pfn_to_page(pfn); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -/* - * This is called just before initializing sections for a NUMA node. - * Any special initialization that needs to be done before the - * generic initialization can be done from here. Sections that - * are initialized in hooks called from here will be skipped by - * the generic initialization. - */ -void __init sparse_vmemmap_init_nid_early(int nid) -{ -} -#endif - static void subsection_mask_set(unsigned long *map, unsigned long pfn, unsigned long nr_pages) { diff --git a/mm/sparse.c b/mm/sparse.c index 439802e6a6ad78..948839621f83d4 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -284,27 +284,20 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, if (sparse_usage_init(nid, map_count)) panic("Failed to allocate usemap for node %d\n", nid); - sparse_vmemmap_init_nid_early(nid); - for_each_present_section_nr(pnum_begin, pnum) { - struct mem_section *ms; unsigned long pfn = section_nr_to_pfn(pnum); + struct page *map; if (pnum >= pnum_end) break; - ms = __nr_to_section(pnum); - if (!preinited_vmemmap_section(ms)) { - struct page *map; - - map = __populate_section_memmap(pfn, PAGES_PER_SECTION, - nid, NULL, NULL); - if (!map) - panic("Failed to allocate memmap for section %lu\n", pnum); - memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, - NULL, NULL)); - sparse_init_early_section(nid, map, pnum, 0); - } + map = __populate_section_memmap(pfn, PAGES_PER_SECTION, + nid, NULL, NULL); + if (!map) + panic("Failed to allocate memmap for section %lu\n", pnum); + memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, + NULL, NULL)); + sparse_init_early_section(nid, map, pnum, 0); } sparse_usage_fini(); } From 5605bc184d9f786d91e8346ac4ea6a3f8b36b952 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:46:03 +0800 Subject: [PATCH 574/857] mm/sparse: inline usemap allocation into sparse_init_nid() After removing SPARSEMEM_VMEMMAP_PREINIT, sparse_init_nid() no longer needs the transient sparse_usagebuf state and its helper wrappers. Allocate the usemap buffer directly in sparse_init_nid(), pass it to sparse_init_one_section(), and drop sparse_usage_init(), sparse_usage_fini(), and sparse_init_early_section(). Link: https://lore.kernel.org/20260825084608.47437-13-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Qi Zheng Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/mmzone.h | 3 --- mm/sparse.c | 46 +++++++----------------------------------- 2 files changed, 7 insertions(+), 42 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 8bcb522645ba9e..67c84a8a72583e 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2223,9 +2223,6 @@ static inline bool pfn_section_first_valid(struct mem_section *ms, unsigned long } #endif -void sparse_init_early_section(int nid, struct page *map, unsigned long pnum, - unsigned long flags); - #ifndef CONFIG_HAVE_ARCH_PFN_VALID /** * pfn_valid - check if there is a valid memory map entry for a PFN diff --git a/mm/sparse.c b/mm/sparse.c index 948839621f83d4..7096d8e1380267 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -235,42 +235,6 @@ void __weak __meminit vmemmap_populate_print_last(void) { } -static void *sparse_usagebuf __initdata; -static void *sparse_usagebuf_end __initdata; - -/* - * Helper function that is used for generic section initialization, and - * can also be used by any hooks added above. - */ -void __init sparse_init_early_section(int nid, struct page *map, - unsigned long pnum, unsigned long flags) -{ - BUG_ON(!sparse_usagebuf || sparse_usagebuf >= sparse_usagebuf_end); - sparse_init_one_section(__nr_to_section(pnum), pnum, map, - sparse_usagebuf, SECTION_IS_EARLY | flags); - sparse_usagebuf = (void *)sparse_usagebuf + mem_section_usage_size(); -} - -static int __init sparse_usage_init(int nid, unsigned long map_count) -{ - unsigned long size; - - size = mem_section_usage_size() * map_count; - sparse_usagebuf = memblock_alloc_node(size, SMP_CACHE_BYTES, nid); - if (!sparse_usagebuf) { - sparse_usagebuf_end = NULL; - return -ENOMEM; - } - - sparse_usagebuf_end = sparse_usagebuf + size; - return 0; -} - -static void __init sparse_usage_fini(void) -{ - sparse_usagebuf = sparse_usagebuf_end = NULL; -} - /* * Initialize sparse on a specific node. The node spans [pnum_begin, pnum_end) * And number of present sections in this node is map_count. @@ -280,8 +244,11 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, unsigned long map_count) { unsigned long pnum; + struct mem_section_usage *usage; - if (sparse_usage_init(nid, map_count)) + usage = memblock_alloc_node(map_count * mem_section_usage_size(), + SMP_CACHE_BYTES, nid); + if (!usage) panic("Failed to allocate usemap for node %d\n", nid); for_each_present_section_nr(pnum_begin, pnum) { @@ -297,9 +264,10 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, panic("Failed to allocate memmap for section %lu\n", pnum); memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, NULL, NULL)); - sparse_init_early_section(nid, map, pnum, 0); + sparse_init_one_section(__nr_to_section(pnum), pnum, map, usage, + SECTION_IS_EARLY); + usage = (void *)usage + mem_section_usage_size(); } - sparse_usage_fini(); } /* From c9b86a757aa1dcec7ebc487121e43658da612d09 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:46:04 +0800 Subject: [PATCH 575/857] mm/sparse: remove section_map_size() section_map_size() no longer provides any shared logic. After the sparse-vmemmap changes, its only remaining user is the !CONFIG_SPARSEMEM_VMEMMAP path in __populate_section_memmap(), which can compute the size inline with PAGE_ALIGN(sizeof(struct page) * PAGES_PER_SECTION). Remove section_map_size() and inline the remaining calculation. Link: https://lore.kernel.org/20260825084608.47437-14-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Qi Zheng Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/mm.h | 1 - mm/sparse.c | 15 ++------------- 2 files changed, 2 insertions(+), 14 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 441bd39eab7343..b19711b6dbc69a 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5140,7 +5140,6 @@ static inline void print_vma_addr(char *prefix, unsigned long rip) } #endif -unsigned long section_map_size(void); struct page * __populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap); diff --git a/mm/sparse.c b/mm/sparse.c index 7096d8e1380267..9349ed6326c015 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -209,23 +209,12 @@ void __init sparse_sections_init(void) memory_present(nid, start, end); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP -unsigned long __init section_map_size(void) -{ - return ALIGN(sizeof(struct page) * PAGES_PER_SECTION, PMD_SIZE); -} - -#else -unsigned long __init section_map_size(void) -{ - return PAGE_ALIGN(sizeof(struct page) * PAGES_PER_SECTION); -} - +#ifndef CONFIG_SPARSEMEM_VMEMMAP struct page __init *__populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { - unsigned long size = section_map_size(); + unsigned long size = PAGE_ALIGN(sizeof(struct page) * PAGES_PER_SECTION); return memmap_alloc(size, size, __pa(MAX_DMA_ADDRESS), nid, false); } From aeb1be5ad95b79fce89fbc5b57b0ff2a17e88497 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:46:05 +0800 Subject: [PATCH 576/857] mm/hugetlb: remove HUGE_BOOTMEM_HVO The HUGE_BOOTMEM_HVO flag tracked whether a bootmem huge page had already gone through the old early vmemmap optimization path. Now that HugeTLB uses section-based vmemmap optimization, that state is already reflected in the section order. Remove HUGE_BOOTMEM_HVO and its helper, and use the section state directly when deciding whether to mark a folio as vmemmap-optimized. Link: https://lore.kernel.org/20260825084608.47437-15-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/hugetlb.h | 5 ++--- mm/hugetlb.c | 12 +----------- mm/hugetlb_vmemmap.c | 2 -- 3 files changed, 3 insertions(+), 16 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index fe28f98e1b220e..3559041a5a5780 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -675,9 +675,8 @@ struct hstate { char name[HSTATE_NAME_LEN]; }; -#define HUGE_BOOTMEM_HVO 0x0001 -#define HUGE_BOOTMEM_ZONES_VALID 0x0002 -#define HUGE_BOOTMEM_CMA 0x0004 +#define HUGE_BOOTMEM_ZONES_VALID BIT(0) +#define HUGE_BOOTMEM_CMA BIT(1) int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 524ca0c2b3ecea..7e9f69996a33ea 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3209,11 +3209,6 @@ static void __init hugetlb_folio_init_vmemmap(struct folio *folio, prep_compound_head(&folio->page, huge_page_order(h)); } -static bool __init hugetlb_bootmem_page_prehvo(struct huge_bootmem_page *m) -{ - return m->flags & HUGE_BOOTMEM_HVO; -} - static bool __init hugetlb_bootmem_page_earlycma(struct huge_bootmem_page *m) { return m->flags & HUGE_BOOTMEM_CMA; @@ -3305,12 +3300,7 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) HUGETLB_VMEMMAP_RESERVE_PAGES); init_new_hugetlb_folio(folio); - if (hugetlb_bootmem_page_prehvo(m)) - /* - * If pre-HVO was done, just set the - * flag, the HVO code will then skip - * this folio. - */ + if (vmemmap_optimizable_order(pfn_to_section_order(folio_pfn(folio)))) folio_set_hugetlb_vmemmap_optimized(folio); section_set_order_range(folio_pfn(folio), folio_nr_pages(folio), 0); diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index 7293706b532f38..a25adc4743514f 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -714,8 +714,6 @@ void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) return; section_set_order_range(pfn, pages_per_huge_page(h), huge_page_order(h)); - if (vmemmap_optimizable_order(pfn_to_section_order(pfn))) - m->flags |= HUGE_BOOTMEM_HVO; } static const struct ctl_table hugetlb_vmemmap_sysctls[] = { From 35000b56e71c492446ec97ea05d7017a5b12a0bd Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:46:06 +0800 Subject: [PATCH 577/857] mm/hugetlb: remove HUGE_BOOTMEM_CMA Track early CMA hugetlb pages from the hstate instead of storing a redundant bootmem flag. This removes the unused helper and keeps the bootmem metadata minimal. Link: https://lore.kernel.org/20260825084608.47437-16-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/hugetlb.h | 1 - mm/hugetlb.c | 14 ++++---------- 2 files changed, 4 insertions(+), 11 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 3559041a5a5780..255a258f11d13c 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -676,7 +676,6 @@ struct hstate { }; #define HUGE_BOOTMEM_ZONES_VALID BIT(0) -#define HUGE_BOOTMEM_CMA BIT(1) int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 7e9f69996a33ea..1a2804f9ff9066 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3133,7 +3133,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) */ INIT_LIST_HEAD(&m->list); m->hstate = h; - m->flags = hugetlb_early_cma(h) ? HUGE_BOOTMEM_CMA : 0; + m->flags = 0; /* CMA pages: zone-crossing is validated in hugetlb_cma_reserve(). */ if (!hugetlb_early_cma(h) && @@ -3209,11 +3209,6 @@ static void __init hugetlb_folio_init_vmemmap(struct folio *folio, prep_compound_head(&folio->page, huge_page_order(h)); } -static bool __init hugetlb_bootmem_page_earlycma(struct huge_bootmem_page *m) -{ - return m->flags & HUGE_BOOTMEM_CMA; -} - /* * memblock-allocated pageblocks might not have the migrate type set * if marked with the 'noinit' flag. Set it to the default (MIGRATE_MOVABLE) @@ -3304,9 +3299,6 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) folio_set_hugetlb_vmemmap_optimized(folio); section_set_order_range(folio_pfn(folio), folio_nr_pages(folio), 0); - if (hugetlb_bootmem_page_earlycma(m)) - folio_set_hugetlb_cma(folio); - list_add(&folio->lru, &folio_list); /* @@ -3317,7 +3309,9 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) * For CMA pages, this is done in init_cma_pageblock * (via hugetlb_bootmem_init_migratetype), so skip it here. */ - if (!folio_test_hugetlb_cma(folio)) + if (hugetlb_early_cma(h)) + folio_set_hugetlb_cma(folio); + else adjust_managed_page_count(page, pages_per_huge_page(h)); cond_resched(); } From 71c282d565fcd4cb079e3b54ab7d176953c179c4 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:46:07 +0800 Subject: [PATCH 578/857] mm/hugetlb: localize struct huge_bootmem_page struct huge_bootmem_page is only used by hugetlb boot-time allocation code, but its definition currently lives in mm/internal.h because hugetlb_vmemmap_optimize_bootmem_page() takes it as an argument. This exposes a hugetlb-specific internal type more broadly than needed. Change hugetlb_vmemmap_optimize_bootmem_page() to take the information it actually needs. With that interface, mm/hugetlb_vmemmap.h no longer needs to include mm/internal.h, and struct huge_bootmem_page can move into mm/hugetlb.c. Link: https://lore.kernel.org/20260825084608.47437-17-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/hugetlb.c | 8 +++++++- mm/hugetlb_vmemmap.c | 8 +++----- mm/hugetlb_vmemmap.h | 5 ++--- mm/internal.h | 7 ------- 4 files changed, 12 insertions(+), 16 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 1a2804f9ff9066..7b541b83ec15f8 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -55,6 +55,12 @@ #include "sparse.h" #include +struct huge_bootmem_page { + struct list_head list; + struct hstate *hstate; + unsigned long flags; +}; + int hugetlb_max_hstate __read_mostly; unsigned int default_hstate_idx; struct hstate hstates[HUGE_MAX_HSTATE]; @@ -3151,7 +3157,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) } else { list_add_tail(&m->list, &huge_boot_pages[nid]); m->flags |= HUGE_BOOTMEM_ZONES_VALID; - hugetlb_vmemmap_optimize_bootmem_page(m); + hugetlb_vmemmap_optimize_bootmem_page(pfn, huge_page_order(h)); /* * Only initialize the head struct page in memmap_init_reserved_pages, * rest of the struct pages will be initialized by the HugeTLB diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index a25adc4743514f..eb339c4a71f403 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -19,6 +19,7 @@ #include #include "hugetlb_vmemmap.h" #include "sparse.h" +#include "internal.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -705,15 +706,12 @@ void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head __hugetlb_vmemmap_optimize_folios(h, folio_list, true); } -void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) +void __init hugetlb_vmemmap_optimize_bootmem_page(unsigned long pfn, unsigned int order) { - struct hstate *h = m->hstate; - unsigned long pfn = PHYS_PFN(__pa(m)); - if (!READ_ONCE(vmemmap_optimize_enabled)) return; - section_set_order_range(pfn, pages_per_huge_page(h), huge_page_order(h)); + section_set_order_range(pfn, 1UL << order, order); } static const struct ctl_table hugetlb_vmemmap_sysctls[] = { diff --git a/mm/hugetlb_vmemmap.h b/mm/hugetlb_vmemmap.h index 20eb03df542a43..464192e32decf4 100644 --- a/mm/hugetlb_vmemmap.h +++ b/mm/hugetlb_vmemmap.h @@ -9,7 +9,6 @@ #ifndef _LINUX_HUGETLB_VMEMMAP_H #define _LINUX_HUGETLB_VMEMMAP_H #include -#include "internal.h" /* * Reserve one vmemmap page, all vmemmap addresses are mapped to it. See @@ -26,7 +25,7 @@ long hugetlb_vmemmap_restore_folios(const struct hstate *h, void hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio); void hugetlb_vmemmap_optimize_folios(struct hstate *h, struct list_head *folio_list); void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head *folio_list); -void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m); +void hugetlb_vmemmap_optimize_bootmem_page(unsigned long pfn, unsigned int order); static inline unsigned int hugetlb_vmemmap_size(const struct hstate *h) { @@ -77,7 +76,7 @@ static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate return 0; } -static inline void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) +static inline void hugetlb_vmemmap_optimize_bootmem_page(unsigned long pfn, unsigned int order) { } #endif /* CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP */ diff --git a/mm/internal.h b/mm/internal.h index da833cafcd599d..5cc220db907668 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -23,13 +23,6 @@ #include "vma.h" struct folio_batch; -struct hstate; - -struct huge_bootmem_page { - struct list_head list; - struct hstate *hstate; - unsigned long flags; -}; /* mm/workingset.c */ bool workingset_test_recent(void *shadow, bool file, bool *workingset, From 5ba6fc539c1ba10dafe39bbdf12f9a328784fe34 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 25 Aug 2026 16:46:08 +0800 Subject: [PATCH 579/857] mm/hugetlb: localize HUGE_BOOTMEM_ZONES_VALID HUGE_BOOTMEM_ZONES_VALID is only used by the huge_bootmem_page flag handling in mm/hugetlb.c. Keep the definition next to that private data structure instead of exposing it through the public hugetlb header. No functional change is intended. Link: https://lore.kernel.org/20260825084608.47437-18-songmuchun@bytedance.com Signed-off-by: Muchun Song Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/hugetlb.h | 2 -- mm/hugetlb.c | 2 ++ 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 255a258f11d13c..900c95e346b2e3 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -675,8 +675,6 @@ struct hstate { char name[HSTATE_NAME_LEN]; }; -#define HUGE_BOOTMEM_ZONES_VALID BIT(0) - int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); void wait_for_freed_hugetlb_folios(void); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 7b541b83ec15f8..47cf45320ffbcd 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -55,6 +55,8 @@ #include "sparse.h" #include +#define HUGE_BOOTMEM_ZONES_VALID BIT(0) + struct huge_bootmem_page { struct list_head list; struct hstate *hstate; From 55f9a95fbffd658063ef6eeb56221d8c3fb0d07f Mon Sep 17 00:00:00 2001 From: Song Hu Date: Tue, 25 Aug 2026 16:57:54 +0800 Subject: [PATCH 580/857] selftests/mm: emit TAP header in uffd-wp-mremap Patch series "selftests/mm: TAP output and global-state fixes", v4. uffd-wp-mremap and mremap_test never print the TAP header (and mremap_test skips with a bare exit(KSFT_SKIP) rather than a KTAP skip), so their output is not valid KTAP; hugetlb-soft-offline toggles enable_soft_offline during the run and leaves it disabled afterwards. This patch (of 3): uffd-wp-mremap calls ksft_set_plan() without ksft_print_header(), so its output is not valid KTAP. Add the header, like the sibling uffd tests (uffd-stress, uffd-unit-tests). Link: https://lore.kernel.org/20260825085756.63030-1-husong@kylinos.cn Link: https://lore.kernel.org/20260825085756.63030-2-husong@kylinos.cn Signed-off-by: Song Hu Acked-by: Mike Rapoport (Microsoft) Reviewed-by: Sarthak Sharma Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Muhammad Usama Anjum Tested-by: Muhammad Usama Anjum Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Peter Xu Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/uffd-wp-mremap.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/mm/uffd-wp-mremap.c b/tools/testing/selftests/mm/uffd-wp-mremap.c index c973d6722720c3..572c2516e874d7 100644 --- a/tools/testing/selftests/mm/uffd-wp-mremap.c +++ b/tools/testing/selftests/mm/uffd-wp-mremap.c @@ -347,6 +347,8 @@ int main(int argc, char **argv) struct thp_settings settings; int i, j, plan = 0; + ksft_print_header(); + hugepage_save_settings(true, true); check_uffd_wp_feature_supported(); From 1c38a84fb6542539ac094a886511fd6e3928edcf Mon Sep 17 00:00:00 2001 From: Song Hu Date: Tue, 25 Aug 2026 16:57:55 +0800 Subject: [PATCH 581/857] selftests/mm: emit TAP header and use TAP skip in mremap_test mremap_test calls ksft_set_plan() without ksft_print_header(), and its get_mmap_min_addr() skip path uses a bare exit(KSFT_SKIP) that prints no TAP line, so its output is not valid KTAP. Add the header and switch the skip to ksft_exit_skip(). Also fix two more KTAP compliance issues spotted in review: - get_mmap_min_addr() calls strerror(errno) after fclose(), which may clobber errno; save errno before fclose() instead. - Some ksft_*() messages embed "\n\t", so the text after each embedded newline is printed without the "# " prefix. Split those into separate messages. And cache mmap_min_addr in main() before ksft_set_plan(), so that the skip paths in get_mmap_min_addr() are taken before the plan is set; a skip after the plan leaves the run with fewer tests than planned. Link: https://lore.kernel.org/20260825085756.63030-3-husong@kylinos.cn Signed-off-by: Song Hu Acked-by: Mike Rapoport (Microsoft) Reviewed-by: Sarthak Sharma Reviewed-by: Muhammad Usama Anjum Tested-by: Muhammad Usama Anjum Acked-by: Lorenzo Stoakes (ARM) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Peter Xu Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/mremap_test.c | 43 ++++++++++++++---------- 1 file changed, 25 insertions(+), 18 deletions(-) diff --git a/tools/testing/selftests/mm/mremap_test.c b/tools/testing/selftests/mm/mremap_test.c index 779ef2d5f1b9f7..97abf4713cc595 100644 --- a/tools/testing/selftests/mm/mremap_test.c +++ b/tools/testing/selftests/mm/mremap_test.c @@ -111,18 +111,17 @@ static unsigned long long get_mmap_min_addr(void) return addr; fp = fopen("/proc/sys/vm/mmap_min_addr", "r"); - if (fp == NULL) { - ksft_print_msg("Failed to open /proc/sys/vm/mmap_min_addr: %s\n", - strerror(errno)); - exit(KSFT_SKIP); - } + if (!fp) + ksft_exit_skip("Failed to open /proc/sys/vm/mmap_min_addr: %s\n", + strerror(errno)); n_matched = fscanf(fp, "%llu", &addr); if (n_matched != 1) { - ksft_print_msg("Failed to read /proc/sys/vm/mmap_min_addr: %s\n", - strerror(errno)); + int err = errno; + fclose(fp); - exit(KSFT_SKIP); + ksft_exit_skip("Failed to read /proc/sys/vm/mmap_min_addr: %s\n", + strerror(err)); } fclose(fp); @@ -1165,10 +1164,11 @@ static void run_mremap_test_case(struct test test_case, int *failures, rand_addr); if (remap_time < 0) { - if (test_case.expect_failure) - ksft_test_result_xfail("%s\n\tExpected mremap failure\n", - test_case.name); - else { + if (test_case.expect_failure) { + ksft_print_msg("%s: expected mremap failure\n", + test_case.name); + ksft_test_result_xfail("%s\n", test_case.name); + } else { ksft_test_result_fail("%s\n", test_case.name); *failures += 1; } @@ -1178,11 +1178,13 @@ static void run_mremap_test_case(struct test test_case, int *failures, * was faulted in. */ if (threshold_mb == VALIDATION_NO_THRESHOLD || - test_case.config.region_size <= threshold_mb * _1MB) - ksft_test_result_pass("%s\n\tmremap time: %12lldns\n", - test_case.name, remap_time); - else + test_case.config.region_size <= threshold_mb * _1MB) { + ksft_print_msg("%s: mremap time: %12lldns\n", + test_case.name, remap_time); ksft_test_result_pass("%s\n", test_case.name); + } else { + ksft_test_result_pass("%s\n", test_case.name); + } } } @@ -1251,13 +1253,18 @@ int main(int argc, char **argv) time_t t; FILE *maps_fp; + ksft_print_header(); + + get_mmap_min_addr(); + pattern_seed = (unsigned int) time(&t); if (parse_args(argc, argv, &threshold_mb, &pattern_seed) < 0) exit(EXIT_FAILURE); - ksft_print_msg("Test configs:\n\tthreshold_mb=%u\n\tpattern_seed=%u\n\n", - threshold_mb, pattern_seed); + ksft_print_msg("Test configs:\n"); + ksft_print_msg("threshold_mb=%u\n", threshold_mb); + ksft_print_msg("pattern_seed=%u\n", pattern_seed); /* * set preallocated random array according to test configs; see the From a2cb3f5545a3a72079a7f3a891606f865086d0c8 Mon Sep 17 00:00:00 2001 From: Song Hu Date: Tue, 25 Aug 2026 16:57:56 +0800 Subject: [PATCH 582/857] selftests/mm: restore enable_soft_offline in hugetlb-soft-offline hugetlb-soft-offline toggles /proc/sys/vm/enable_soft_offline between 1 and 0 (test_soft_offline_common(1) then (0)) and leaves it at 0 when it finishes, silently disabling soft offlining for the whole system after the run. Save the original value before the test and restore it from an atexit() handler, as hugepage_restore_settings_atexit() in hugepage_settings.c already does. Use read_num()/write_num() from vm_util instead of hand-rolled popen()/fopen() helpers. The restore handler must not call write_num(): on failure it re-enters exit() through ksft_exit_fail_msg(), which is undefined behavior from inside an atexit handler. A non-root run hits it directly - the restore write fails the same way the write that triggered the exit did. Restore with plain open()/write(), best effort. Link: https://lore.kernel.org/20260825085756.63030-4-husong@kylinos.cn Signed-off-by: Song Hu Reviewed-by: Muhammad Usama Anjum Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Peter Xu Cc: Sarthak Sharma Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- .../selftests/mm/hugetlb-soft-offline.c | 49 +++++++++++-------- 1 file changed, 28 insertions(+), 21 deletions(-) diff --git a/tools/testing/selftests/mm/hugetlb-soft-offline.c b/tools/testing/selftests/mm/hugetlb-soft-offline.c index bc202e4ed2bda7..4af9d3db7b5b6f 100644 --- a/tools/testing/selftests/mm/hugetlb-soft-offline.c +++ b/tools/testing/selftests/mm/hugetlb-soft-offline.c @@ -11,6 +11,7 @@ #define _GNU_SOURCE #include +#include #include #include #include @@ -23,6 +24,7 @@ #include #include "kselftest.h" +#include "vm_util.h" #include "hugepage_settings.h" #ifndef MADV_SOFT_OFFLINE @@ -31,6 +33,8 @@ #define EPREFIX " !!! " +#define ENABLE_SOFT_OFFLINE_PATH "/proc/sys/vm/enable_soft_offline" + static int do_soft_offline(int fd, size_t len, int expect_errno) { char *filemap = NULL; @@ -77,26 +81,29 @@ static int do_soft_offline(int fd, size_t len, int expect_errno) return ret; } -static int set_enable_soft_offline(int value) -{ - char cmd[256] = {0}; - FILE *cmdfile = NULL; - - if (value != 0 && value != 1) - return -EINVAL; +static unsigned long orig_enable_soft_offline = -1UL; - sprintf(cmd, "echo %d > /proc/sys/vm/enable_soft_offline", value); - cmdfile = popen(cmd, "r"); +/* + * Runs from an atexit handler, so it must not call anything that + * exits on failure: write_num() would re-enter exit() through + * ksft_exit_fail_msg(). + */ +static void restore_enable_soft_offline(void) +{ + char buf[24]; + int fd, len; - if (cmdfile) - ksft_print_msg("enable_soft_offline => %d\n", value); - else { - ksft_perror(EPREFIX "failed to set enable_soft_offline"); - return errno; - } + if (orig_enable_soft_offline == -1UL) + return; - pclose(cmdfile); - return 0; + len = snprintf(buf, sizeof(buf), "%lu", orig_enable_soft_offline); + fd = open(ENABLE_SOFT_OFFLINE_PATH, O_WRONLY); + if (fd < 0) + return; + if (write(fd, buf, len) != len) + ksft_print_msg("failed to restore enable_soft_offline: %s\n", + strerror(errno)); + close(fd); } static int create_hugetlbfs_file(struct statfs *file_stat) @@ -145,10 +152,7 @@ static void test_soft_offline_common(int enable_soft_offline) hugepagesize_kb = file_stat.f_bsize / 1024; ksft_print_msg("Hugepagesize is %ldkB\n", hugepagesize_kb); - if (set_enable_soft_offline(enable_soft_offline) != 0) { - close(fd); - ksft_exit_fail_msg("Failed to set enable_soft_offline\n"); - } + write_num(ENABLE_SOFT_OFFLINE_PATH, enable_soft_offline); nr_hugepages_before = hugetlb_nr_default_pages(); @@ -192,6 +196,9 @@ int main(int argc, char **argv) ksft_set_plan(2); + orig_enable_soft_offline = read_num(ENABLE_SOFT_OFFLINE_PATH); + atexit(restore_enable_soft_offline); + test_soft_offline_common(1); test_soft_offline_common(0); From 8499b4645c7f199707510461779f82d85732e742 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Sun, 23 Aug 2026 12:40:52 +0800 Subject: [PATCH 583/857] mm/hugetlb: warn instead of silently bailing gigantic pages without runtime support remove_hugetlb_folio() and __update_and_free_hugetlb_folio() silently return for gigantic hstates that lack runtime freeing support. All callers should already filter such hstates upstream, so turn the silent bail into a VM_WARN_ON_ONCE to catch caller regressions instead of masking them. Link: https://lore.kernel.org/20260823044118.1097121-3-xialonglong2025@163.com Signed-off-by: Longlong Xia Assisted-by: Codex:gpt-5.6-sol Acked-by: Muchun Song Cc: David Hildenbrand Cc: Miaohe Lin Cc: Michal Hocko Cc: Oscar Salvador Signed-off-by: Andrew Morton --- mm/hugetlb.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 47cf45320ffbcd..8fa1bafa03d91b 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -1396,8 +1396,11 @@ void remove_hugetlb_folio(struct hstate *h, struct folio *folio, VM_BUG_ON_FOLIO(hugetlb_cgroup_from_folio_rsvd(folio), folio); lockdep_assert_held(&hugetlb_lock); - if (hstate_is_gigantic_no_runtime(h)) + if (hstate_is_gigantic_no_runtime(h)) { + /* Callers must filter gigantic_no_runtime upstream. */ + VM_WARN_ON_ONCE(1); return; + } list_del(&folio->lru); @@ -1458,8 +1461,11 @@ static void __update_and_free_hugetlb_folio(struct hstate *h, { bool clear_flag = folio_test_hugetlb_vmemmap_optimized(folio); - if (hstate_is_gigantic_no_runtime(h)) + if (hstate_is_gigantic_no_runtime(h)) { + /* Callers must filter gigantic_no_runtime upstream. */ + VM_WARN_ON_ONCE(1); return; + } /* * If we don't know which subpages are hwpoisoned, we can't free From a6f3c83b9eec172e0156adbc512267adb806b497 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sun, 23 Aug 2026 14:46:12 +0300 Subject: [PATCH 584/857] set_memory: add number of pages parameter to set_direct_map APIs Patch series "arch, mm/execmem: resolve confusion about set_direct_map_valid_noflush()", v2. Recent discussion about implementation of execmem's ROX caches on arm64 [1] revealed a confusion about how set_direct_map_valid_noflush() implemented on different architectures. On arm64 it sets or clears the PTE_VALID bit marking a PTE as present or not present. On other architectures it's a range version of set_direct_map_invalid_noflush() and set_direct_map_default_noflush() Unlike arm64::set_direct_map_valid_noflush(), set_direct_map_default_noflush() not only marks PTE as present, but also sets its default protection mode. Other than that, initial design of execmem ROX caches didn't rely on restoration of large mappings that's now available on x86, but completely removed the memory allocated for the ROX cache from the direct map to ensure that large mappings are not split. This precluded usage of VM_FLUSH_RESET_PERMS for the ROX cache allocations and required execmem to implement manipulation of the direct map alias. Current implementation of ROX caches does not remove the direct map alias but simply calls set_memory_rox() that updates the permissions in both vmalloc address space and the direct map and relies on collapse_large_pages() in x86 CPA to keep large mappings. This allow using VM_FLUSH_RESET_PERMS for execmem ROX cache allocations with small adjustments to set_direct_map APIs and vmalloc::reset_perms() behaviour: adding number of pages parameter to set_direct_map APIs and making resetting of the direct map permissions in vmalloc VMAP_HUGE friendly. Implement these adjustments, make execmem always use VM_FLUSH_RESET_PERMS and revert set_direct_map_valid_noflush() changes. This patch (of 6): When set_direct_map APIs were introduced by the commit d253ca0c3865 ("x86/mm/cpa: Add set_direct_map_*() functions") the single page parameter made sense because the initial callers (vmalloc and hibernation) had sets of unsorted struct pages that required changes of their mappings in the direct map. Since there is an increasing demand for direct map manipulation and it is also desirable to be able to update larger physically contiguous mappings, for example an entire large folio, extend set_direct_map APIs to receive number of pages parameter. As there is still only a handful of callers, change the existing functions directly and update all the call sites rather than adding wrappers for single page case. Link: https://lore.kernel.org/20260823-execmem-set-vm-perms-v0-2-v2-1-b013a37d84b3@kernel.org Link: https://lore.kernel.org/all/20260611130144.1385343-4-abarnas@google.com [1] Signed-off-by: Mike Rapoport (Microsoft) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Signed-off-by: Andrew Morton --- arch/arm64/include/asm/set_memory.h | 4 ++-- arch/arm64/mm/pageattr.c | 8 ++++---- arch/loongarch/include/asm/set_memory.h | 4 ++-- arch/loongarch/mm/pageattr.c | 8 ++++---- arch/riscv/include/asm/set_memory.h | 4 ++-- arch/riscv/mm/pageattr.c | 8 ++++---- arch/s390/include/asm/set_memory.h | 4 ++-- arch/s390/mm/pageattr.c | 8 ++++---- arch/x86/include/asm/set_memory.h | 4 ++-- arch/x86/mm/pat/set_memory.c | 8 ++++---- include/linux/set_memory.h | 6 ++++-- kernel/power/snapshot.c | 4 ++-- mm/secretmem.c | 6 +++--- mm/vmalloc.c | 5 +++-- 14 files changed, 42 insertions(+), 39 deletions(-) diff --git a/arch/arm64/include/asm/set_memory.h b/arch/arm64/include/asm/set_memory.h index 90f61b17275e1b..b07fd4e026eac0 100644 --- a/arch/arm64/include/asm/set_memory.h +++ b/arch/arm64/include/asm/set_memory.h @@ -11,8 +11,8 @@ bool can_set_direct_map(void); int set_memory_valid(unsigned long addr, int numpages, int enable); -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int numpages); +int set_direct_map_default_noflush(struct page *page, unsigned int numpages); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/arm64/mm/pageattr.c b/arch/arm64/mm/pageattr.c index bbe98ac9ad8c67..db8d60a84d1441 100644 --- a/arch/arm64/mm/pageattr.c +++ b/arch/arm64/mm/pageattr.c @@ -251,7 +251,7 @@ int set_memory_valid(unsigned long addr, int numpages, int enable) __pgprot(PTE_PRESENT_VALID_KERNEL)); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int numpages) { pgprot_t clear_mask = __pgprot(PTE_PRESENT_VALID_KERNEL); pgprot_t set_mask = __pgprot(PTE_PRESENT_INVALID); @@ -260,10 +260,10 @@ int set_direct_map_invalid_noflush(struct page *page) return 0; return update_range_prot((unsigned long)page_address(page), - PAGE_SIZE, set_mask, clear_mask); + PAGE_SIZE * numpages, set_mask, clear_mask); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int numpages) { pgprot_t set_mask = __pgprot(PTE_PRESENT_VALID_KERNEL | PTE_WRITE); pgprot_t clear_mask = __pgprot(PTE_PRESENT_INVALID | PTE_RDONLY); @@ -272,7 +272,7 @@ int set_direct_map_default_noflush(struct page *page) return 0; return update_range_prot((unsigned long)page_address(page), - PAGE_SIZE, set_mask, clear_mask); + PAGE_SIZE * numpages, set_mask, clear_mask); } static int __set_memory_enc_dec(unsigned long addr, diff --git a/arch/loongarch/include/asm/set_memory.h b/arch/loongarch/include/asm/set_memory.h index 55dfaefd02c8a6..563aab92896e9b 100644 --- a/arch/loongarch/include/asm/set_memory.h +++ b/arch/loongarch/include/asm/set_memory.h @@ -15,8 +15,8 @@ int set_memory_ro(unsigned long addr, int numpages); int set_memory_rw(unsigned long addr, int numpages); bool kernel_page_present(struct page *page); -int set_direct_map_default_noflush(struct page *page); -int set_direct_map_invalid_noflush(struct page *page); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); #endif /* _ASM_LOONGARCH_SET_MEMORY_H */ diff --git a/arch/loongarch/mm/pageattr.c b/arch/loongarch/mm/pageattr.c index 614ccc7afccbea..43ad2a104f19df 100644 --- a/arch/loongarch/mm/pageattr.c +++ b/arch/loongarch/mm/pageattr.c @@ -198,24 +198,24 @@ bool kernel_page_present(struct page *page) return pte_present(ptep_get(pte)); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { unsigned long addr = (unsigned long)page_address(page); if (addr < vm_map_base) return 0; - return __set_memory(addr, 1, PAGE_KERNEL, __pgprot(0)); + return __set_memory(addr, nr, PAGE_KERNEL, __pgprot(0)); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { unsigned long addr = (unsigned long)page_address(page); if (addr < vm_map_base) return 0; - return __set_memory(addr, 1, __pgprot(0), __pgprot(_PAGE_PRESENT | _PAGE_VALID)); + return __set_memory(addr, nr, __pgprot(0), __pgprot(_PAGE_PRESENT | _PAGE_VALID)); } int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) diff --git a/arch/riscv/include/asm/set_memory.h b/arch/riscv/include/asm/set_memory.h index ef59e1716a2cfd..db1d0ed82b6962 100644 --- a/arch/riscv/include/asm/set_memory.h +++ b/arch/riscv/include/asm/set_memory.h @@ -40,8 +40,8 @@ static inline int set_kernel_memory(char *startp, char *endp, } #endif -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/riscv/mm/pageattr.c b/arch/riscv/mm/pageattr.c index 3f76db3d276992..20ef95b1d0c36e 100644 --- a/arch/riscv/mm/pageattr.c +++ b/arch/riscv/mm/pageattr.c @@ -374,15 +374,15 @@ int set_memory_nx(unsigned long addr, int numpages) return __set_memory(addr, numpages, __pgprot(0), __pgprot(_PAGE_EXEC)); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_address(page), 1, + return __set_memory((unsigned long)page_address(page), nr, __pgprot(0), __pgprot(_PAGE_PRESENT)); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_address(page), 1, + return __set_memory((unsigned long)page_address(page), nr, PAGE_KERNEL, __pgprot(_PAGE_EXEC)); } diff --git a/arch/s390/include/asm/set_memory.h b/arch/s390/include/asm/set_memory.h index 94092f4ae76499..6b0aa9147ed8e6 100644 --- a/arch/s390/include/asm/set_memory.h +++ b/arch/s390/include/asm/set_memory.h @@ -60,8 +60,8 @@ __SET_MEMORY_FUNC(set_memory_rox, SET_MEMORY_RO | SET_MEMORY_X) __SET_MEMORY_FUNC(set_memory_rwnx, SET_MEMORY_RW | SET_MEMORY_NX) __SET_MEMORY_FUNC(set_memory_4k, SET_MEMORY_4K) -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/s390/mm/pageattr.c b/arch/s390/mm/pageattr.c index 1e202e3d08e75f..7549543d624125 100644 --- a/arch/s390/mm/pageattr.c +++ b/arch/s390/mm/pageattr.c @@ -382,14 +382,14 @@ int __set_memory(unsigned long addr, unsigned long numpages, unsigned long flags return rc; } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_to_virt(page), 1, SET_MEMORY_INV); + return __set_memory((unsigned long)page_to_virt(page), nr, SET_MEMORY_INV); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_to_virt(page), 1, SET_MEMORY_DEF); + return __set_memory((unsigned long)page_to_virt(page), nr, SET_MEMORY_DEF); } int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) diff --git a/arch/x86/include/asm/set_memory.h b/arch/x86/include/asm/set_memory.h index 4362c26aa992db..0c4235d159f483 100644 --- a/arch/x86/include/asm/set_memory.h +++ b/arch/x86/include/asm/set_memory.h @@ -86,8 +86,8 @@ int set_pages_wb(struct page *page, int numpages); int set_pages_ro(struct page *page, int numpages); int set_pages_rw(struct page *page, int numpages); -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/x86/mm/pat/set_memory.c b/arch/x86/mm/pat/set_memory.c index c38faf39ce152d..4d07a9fbc43a78 100644 --- a/arch/x86/mm/pat/set_memory.c +++ b/arch/x86/mm/pat/set_memory.c @@ -2656,14 +2656,14 @@ static int __set_pages_np(struct page *page, int numpages, unsigned int cpa_flag return __change_page_attr_set_clr(&cpa, 1); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { - return __set_pages_np(page, 1, 0); + return __set_pages_np(page, nr, 0); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { - return __set_pages_p(page, 1, 0); + return __set_pages_p(page, nr, 0); } int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h index 3030d9245f5ac8..0b77f1d7d8b9ce 100644 --- a/include/linux/set_memory.h +++ b/include/linux/set_memory.h @@ -25,11 +25,13 @@ static inline int set_memory_rox(unsigned long addr, int numpages) #endif #ifndef CONFIG_ARCH_HAS_SET_DIRECT_MAP -static inline int set_direct_map_invalid_noflush(struct page *page) +static inline int set_direct_map_invalid_noflush(struct page *page, + unsigned int nr) { return 0; } -static inline int set_direct_map_default_noflush(struct page *page) +static inline int set_direct_map_default_noflush(struct page *page, + unsigned int nr) { return 0; } diff --git a/kernel/power/snapshot.c b/kernel/power/snapshot.c index b209712cb2c3ac..d5dba0e50b2eb6 100644 --- a/kernel/power/snapshot.c +++ b/kernel/power/snapshot.c @@ -88,7 +88,7 @@ static inline int hibernate_restore_unprotect_page(void *page_address) {return 0 static inline void hibernate_map_page(struct page *page) { if (IS_ENABLED(CONFIG_ARCH_HAS_SET_DIRECT_MAP)) { - int ret = set_direct_map_default_noflush(page); + int ret = set_direct_map_default_noflush(page, 1); if (ret) pr_warn_once("Failed to remap page\n"); @@ -101,7 +101,7 @@ static inline void hibernate_unmap_page(struct page *page) { if (IS_ENABLED(CONFIG_ARCH_HAS_SET_DIRECT_MAP)) { unsigned long addr = (unsigned long)page_address(page); - int ret = set_direct_map_invalid_noflush(page); + int ret = set_direct_map_invalid_noflush(page, 1); if (ret) pr_warn_once("Failed to remap page\n"); diff --git a/mm/secretmem.c b/mm/secretmem.c index 384f5cfc457f9e..6cbb8efc994a4d 100644 --- a/mm/secretmem.c +++ b/mm/secretmem.c @@ -139,7 +139,7 @@ static vm_fault_t secretmem_fault(struct vm_fault *vmf) goto out; } - err = set_direct_map_invalid_noflush(folio_page(folio, 0)); + err = set_direct_map_invalid_noflush(folio_page(folio, 0), 1); if (err) { secretmem_unaccount_folio(state, folio); folio_put(folio); @@ -156,7 +156,7 @@ static vm_fault_t secretmem_fault(struct vm_fault *vmf) * already happened when we marked the page invalid * which guarantees that this call won't fail */ - set_direct_map_default_noflush(folio_page(folio, 0)); + set_direct_map_default_noflush(folio_page(folio, 0), 1); folio_put(folio); if (err == -EEXIST) goto retry; @@ -228,7 +228,7 @@ static int secretmem_migrate_folio(struct address_space *mapping, static void secretmem_free_folio(struct folio *folio) { - set_direct_map_default_noflush(folio_page(folio, 0)); + set_direct_map_default_noflush(folio_page(folio, 0), 1); folio_zero_segment(folio, 0, folio_size(folio)); } diff --git a/mm/vmalloc.c b/mm/vmalloc.c index cfeac79856e8bc..fcefa4795701fd 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3363,14 +3363,15 @@ struct vm_struct *remove_vm_area(const void *addr) } static inline void set_area_direct_map(const struct vm_struct *area, - int (*set_direct_map)(struct page *page)) + int (*set_direct_map)(struct page *page, + unsigned int nr)) { unsigned long i; /* HUGE_VMALLOC passes small pages to set_direct_map */ for (i = 0; i < area->nr_pages; i++) if (page_address(area->pages[i])) - set_direct_map(area->pages[i]); + set_direct_map(area->pages[i], 1); } /* From 3bf40008a2e7be8e93d2b8e0ebac9b66ed4cae41 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sun, 23 Aug 2026 14:46:13 +0300 Subject: [PATCH 585/857] mm/vmalloc: set area's page_order after allocation succeeds __vmalloc_area_node() calls set_vm_area_page_order() to set area's page_order before actually allocating pages to populate the area. If allocation of large pages in HUGE_VMAP case fails midway, this leaves the area with elevated page_order throughout the cleanup path. There is no actual issue with this because the only place that currently relies on area->page_order on the cleanup path is the loop calculating the direct map alias range in vm_reset_perms() and it anyway skips unpopulated pages. But having set_vm_area_page_order() in the middle of __vmalloc_area_node() makes things very obscure, hard to reason about and error prone against future changes of the cleanup path. Move the call to set_vm_area_page_order() just before the successful return from __vmalloc_area_node() where page order is guaranteed. While on it, initialize local page_order variable with its declaration. Link: https://lore.kernel.org/20260823-execmem-set-vm-perms-v0-2-v2-2-b013a37d84b3@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Reviewed-by: Uladzislau Rezki (Sony) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Signed-off-by: Andrew Morton --- mm/vmalloc.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index fcefa4795701fd..358494a91a3911 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3878,7 +3878,7 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask, unsigned long size = get_vm_area_size(area); unsigned long array_size; unsigned long nr_small_pages = size >> PAGE_SHIFT; - unsigned int page_order; + unsigned int page_order = page_shift - PAGE_SHIFT; unsigned int flags; int ret; @@ -3906,9 +3906,6 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask, goto fail; } - set_vm_area_page_order(area, page_shift - PAGE_SHIFT); - page_order = vm_area_page_order(area); - /* * High-order nofail allocations are really expensive and * potentially dangerous (pre-mature OOM, disruptive reclaim @@ -3963,6 +3960,7 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask, goto fail; } + set_vm_area_page_order(area, page_order); return area->addr; fail: From 4279d9efe0714a21d1aee037dc8882a713db8034 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sun, 23 Aug 2026 14:46:14 +0300 Subject: [PATCH 586/857] mm/vmalloc: constify vm parameter of get_vm_area_page_order() get_vm_area_page_order() and vm_area_page_order() do not need to modify struct vm_struct passed to them. Constify the parameter. Link: https://lore.kernel.org/20260823-execmem-set-vm-perms-v0-2-v2-3-b013a37d84b3@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Reviewed-by: Uladzislau Rezki (Sony) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Signed-off-by: Andrew Morton --- mm/vmalloc.c | 4 ++-- mm/vmalloc.h | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 358494a91a3911..2a202e8fe7f56e 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3129,7 +3129,7 @@ EXPORT_SYMBOL(vm_map_ram); static struct vm_struct *vmlist __initdata; -static inline unsigned int vm_area_page_order(struct vm_struct *vm) +static inline unsigned int vm_area_page_order(const struct vm_struct *vm) { #ifdef CONFIG_HAVE_ARCH_HUGE_VMALLOC return vm->page_order; @@ -3138,7 +3138,7 @@ static inline unsigned int vm_area_page_order(struct vm_struct *vm) #endif } -unsigned int get_vm_area_page_order(struct vm_struct *vm) +unsigned int get_vm_area_page_order(const struct vm_struct *vm) { return vm_area_page_order(vm); } diff --git a/mm/vmalloc.h b/mm/vmalloc.h index 8866ddcff6681b..211869f365095f 100644 --- a/mm/vmalloc.h +++ b/mm/vmalloc.h @@ -12,7 +12,7 @@ void __init vmalloc_init(void); int __must_check vmap_pages_range_noflush(unsigned long addr, unsigned long end, pgprot_t prot, struct page **pages, unsigned int page_shift, gfp_t gfp_mask); -unsigned int get_vm_area_page_order(struct vm_struct *vm); +unsigned int get_vm_area_page_order(const struct vm_struct *vm); #else static inline void vmalloc_init(void) {} From 7bb948d8c8804d7f606179e88e816aae049147f6 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sun, 23 Aug 2026 14:46:15 +0300 Subject: [PATCH 587/857] mm/vmalloc: make set_area_direct_map HUGE_VMAP friendly set_area_direct_map() always updates direct map alias permissions in single page increments. For HUGE_VMAP areas it's suboptimal. Not only the loop in set_area_direct_map() needlessly has more iterations (e.g times 512 on x86), but it also causes fragmentation of the direct map that could be avoided for the HUGE_VMAP areas populated with large pages. All pages in an area are always of the same order: either same-order large pages when VM_ALLOW_HUGE_VMAP is set and all huge pages were successfully allocated, or order-0 page when VM_ALLOW_HUGE_VMAP is cleared or when huge pages allocation fails and fallback path is taken. Instead of updating the direct map permissions for every order-0 page in an area, use the area's page_order as the loop increment and update the large pages in one call to set_direct_map_{invalid,default}_noflush(). Link: https://lore.kernel.org/20260823-execmem-set-vm-perms-v0-2-v2-4-b013a37d84b3@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Signed-off-by: Andrew Morton --- mm/vmalloc.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 2a202e8fe7f56e..676fe96ae84145 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3366,12 +3366,15 @@ static inline void set_area_direct_map(const struct vm_struct *area, int (*set_direct_map)(struct page *page, unsigned int nr)) { - unsigned long i; + unsigned int nr = (1U << vm_area_page_order(area)); + + for (unsigned long i = 0; i < area->nr_pages; i += nr) { + if (page_address(area->pages[i])) { + int err = set_direct_map(area->pages[i], nr); - /* HUGE_VMALLOC passes small pages to set_direct_map */ - for (i = 0; i < area->nr_pages; i++) - if (page_address(area->pages[i])) - set_direct_map(area->pages[i], 1); + WARN_ON_ONCE(err); + } + } } /* From 0d21552f11fe9807955473709f61dd6ac2d490b3 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sun, 23 Aug 2026 14:46:16 +0300 Subject: [PATCH 588/857] mm/execmem: use VM_FLUSH_RESET_PERMS for ROX cache allocations Initially execmem completely removed direct map alias for the memory allocated for the ROX cache in PMD_SIZE chunks. When that memory was freed, its direct map was restored also in PMD_SIZE chunks to avoid fragmentation of the direct map caused by vmalloc::vm_reset_perms(). This required execmem to implement the wrappers for set_direct_map APIs for proper sequencing of removal and restoration of the direct map aliases. Since then x86's CPA gained support for collapsing the direct map page tables for ROX pages and execmem switched from removing ROX caches from the direct map to making them ROX there, so execmem only needs to update direct map alias permissions when freeing the ROX cache memory. vmalloc already handles those updates for areas with VM_FLUSH_RESET_PERMS set and vmalloc::vm_reset_perms() does not force split of the direct map for PMD_SIZE chunks. Make all execmem vmalloc allocations use VM_FLUSH_RESET_PERMS and remove custom wrappers for set_direct_map APIs. Link: https://lore.kernel.org/20260823-execmem-set-vm-perms-v0-2-v2-5-b013a37d84b3@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Signed-off-by: Andrew Morton --- mm/execmem.c | 42 +++++++----------------------------------- 1 file changed, 7 insertions(+), 35 deletions(-) diff --git a/mm/execmem.c b/mm/execmem.c index 74a178a87e7581..d35f1d0ea54a41 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -36,6 +36,7 @@ static void *execmem_vmalloc(struct execmem_range *range, size_t size, unsigned long end = range->end; void *p; + vm_flags |= VM_FLUSH_RESET_PERMS; if (kasan) vm_flags |= VM_DEFER_KMEMLEAK; @@ -113,28 +114,6 @@ static inline unsigned long mas_range_len(struct ma_state *mas) return mas->last - mas->index + 1; } -static int execmem_set_direct_map_valid(struct vm_struct *vm, bool valid) -{ - unsigned int nr = (1 << get_vm_area_page_order(vm)); - unsigned int updated = 0; - int err = 0; - - for (int i = 0; i < vm->nr_pages; i += nr) { - err = set_direct_map_valid_noflush(vm->pages[i], nr, valid); - if (err) - goto err_restore; - updated += nr; - } - - return 0; - -err_restore: - for (int i = 0; i < updated; i += nr) - set_direct_map_valid_noflush(vm->pages[i], nr, !valid); - - return err; -} - static int execmem_force_rw(void *ptr, size_t size) { unsigned int nr = PAGE_ALIGN(size) >> PAGE_SHIFT; @@ -169,9 +148,6 @@ static void execmem_cache_clean(struct work_struct *work) if (IS_ALIGNED(size, PMD_SIZE) && IS_ALIGNED(mas.index, PMD_SIZE)) { - struct vm_struct *vm = find_vm_area(area); - - execmem_set_direct_map_valid(vm, true); mas_store_gfp(&mas, NULL, GFP_KERNEL); vfree(area); } @@ -312,18 +288,15 @@ static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t si */ mutex_lock(mutex); err = execmem_cache_add_locked(p, alloc_size, GFP_KERNEL); - if (err) - goto err_reset_direct_map; - - p = execmem_cache_alloc_locked(range, size); - + if (!err) + p = execmem_cache_alloc_locked(range, size); mutex_unlock(mutex); + if (err) + goto err_free_mem; + return p; -err_reset_direct_map: - mutex_unlock(mutex); - execmem_set_direct_map_valid(vm, true); err_free_mem: vfree(p); return NULL; @@ -466,7 +439,6 @@ void *execmem_alloc(enum execmem_type type, size_t size) { struct execmem_range *range = &execmem_info->ranges[type]; bool use_cache = range->flags & EXECMEM_ROX_CACHE; - unsigned long vm_flags = VM_FLUSH_RESET_PERMS; pgprot_t pgprot = range->pgprot; void *p = NULL; @@ -475,7 +447,7 @@ void *execmem_alloc(enum execmem_type type, size_t size) if (use_cache) p = execmem_cache_alloc(range, size); else - p = execmem_vmalloc(range, size, pgprot, vm_flags); + p = execmem_vmalloc(range, size, pgprot, 0); return kasan_reset_tag(p); } From 9c3df37aa2784047773341d6790bdcfeb559e195 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sun, 23 Aug 2026 14:46:17 +0300 Subject: [PATCH 589/857] Revert "arch: introduce set_direct_map_valid_noflush()" Commit 0c6378a71574 ("arch: introduce set_direct_map_valid_noflush()") added set_direct_map_valid_noflush() to allow updating the direct map for a physically contiguous range in execmem. As Brendan recently pointed out [1], this API is confusing because on arm64 it means that is sets VALID bit in ptes, while on other architectures it is an analog of set_direct_map_default_noflush(). The only user of set_direct_map_valid_noflush() was execmem's ROX cache freeing path and it was switched to utilize VM_FLUSH_RESET_PERMS for resetting permissions of the direct map alias. With the last user gone and with set_direct_map_{invalid,default}_noflush() accepting number of pages as a parameter, set_direct_map_valid_noflush() become a copy of set_memory_valid() on arm64 and a duplicate of set_direct_map_{invalid,default}_noflush() on other architecture, it is safe to remove set_direct_map_valid_noflush(). Also drop a stale comment in arm64::__kernel_map_pages() that Linus bothered to add when merging changes containing set_direct_map_valid_noflush() to his tree. This reverts commit 0c6378a71574daa6cd1534ad42a956e3262756c7. Link: https://lore.kernel.org/20260823-execmem-set-vm-perms-v0-2-v2-6-b013a37d84b3@kernel.org Link: https://lore.kernel.org/all/DJ69RCVRBO0Y.3JCYSW50IC4RC@linux.dev [1] Signed-off-by: Mike Rapoport (Microsoft) Reviewed-by: Brendan Jackman Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Signed-off-by: Andrew Morton --- arch/arm64/include/asm/set_memory.h | 1 - arch/arm64/mm/pageattr.c | 16 ---------------- arch/loongarch/include/asm/set_memory.h | 1 - arch/loongarch/mm/pageattr.c | 19 ------------------- arch/riscv/include/asm/set_memory.h | 1 - arch/riscv/mm/pageattr.c | 15 --------------- arch/s390/include/asm/set_memory.h | 1 - arch/s390/mm/pageattr.c | 12 ------------ arch/x86/include/asm/set_memory.h | 1 - arch/x86/mm/pat/set_memory.c | 8 -------- include/linux/set_memory.h | 6 ------ 11 files changed, 81 deletions(-) diff --git a/arch/arm64/include/asm/set_memory.h b/arch/arm64/include/asm/set_memory.h index b07fd4e026eac0..0091ba12200e68 100644 --- a/arch/arm64/include/asm/set_memory.h +++ b/arch/arm64/include/asm/set_memory.h @@ -13,7 +13,6 @@ int set_memory_valid(unsigned long addr, int numpages, int enable); int set_direct_map_invalid_noflush(struct page *page, unsigned int numpages); int set_direct_map_default_noflush(struct page *page, unsigned int numpages); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); int set_memory_encrypted(unsigned long addr, int numpages); diff --git a/arch/arm64/mm/pageattr.c b/arch/arm64/mm/pageattr.c index db8d60a84d1441..132938b32eb16f 100644 --- a/arch/arm64/mm/pageattr.c +++ b/arch/arm64/mm/pageattr.c @@ -355,23 +355,7 @@ int realm_register_memory_enc_ops(void) return arm64_mem_crypt_ops_register(&realm_crypt_ops); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - unsigned long addr = (unsigned long)page_address(page); - - if (!can_set_direct_map()) - return 0; - - return set_memory_valid(addr, nr, valid); -} - #ifdef CONFIG_DEBUG_PAGEALLOC -/* - * This is - apart from the return value - doing the same - * thing as the new set_direct_map_valid_noflush() function. - * - * Unify? Explain the conceptual differences? - */ void __kernel_map_pages(struct page *page, int numpages, int enable) { if (!can_set_direct_map()) diff --git a/arch/loongarch/include/asm/set_memory.h b/arch/loongarch/include/asm/set_memory.h index 563aab92896e9b..4bb01172fbc245 100644 --- a/arch/loongarch/include/asm/set_memory.h +++ b/arch/loongarch/include/asm/set_memory.h @@ -17,6 +17,5 @@ int set_memory_rw(unsigned long addr, int numpages); bool kernel_page_present(struct page *page); int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); #endif /* _ASM_LOONGARCH_SET_MEMORY_H */ diff --git a/arch/loongarch/mm/pageattr.c b/arch/loongarch/mm/pageattr.c index 43ad2a104f19df..a7dcff40f75982 100644 --- a/arch/loongarch/mm/pageattr.c +++ b/arch/loongarch/mm/pageattr.c @@ -217,22 +217,3 @@ int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) return __set_memory(addr, nr, __pgprot(0), __pgprot(_PAGE_PRESENT | _PAGE_VALID)); } - -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - unsigned long addr = (unsigned long)page_address(page); - pgprot_t set, clear; - - if (addr < vm_map_base) - return 0; - - if (valid) { - set = PAGE_KERNEL; - clear = __pgprot(0); - } else { - set = __pgprot(0); - clear = __pgprot(_PAGE_PRESENT | _PAGE_VALID); - } - - return __set_memory(addr, nr, set, clear); -} diff --git a/arch/riscv/include/asm/set_memory.h b/arch/riscv/include/asm/set_memory.h index db1d0ed82b6962..e9f9960c194772 100644 --- a/arch/riscv/include/asm/set_memory.h +++ b/arch/riscv/include/asm/set_memory.h @@ -42,7 +42,6 @@ static inline int set_kernel_memory(char *startp, char *endp, int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_default_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); #endif /* __ASSEMBLER__ */ diff --git a/arch/riscv/mm/pageattr.c b/arch/riscv/mm/pageattr.c index 20ef95b1d0c36e..5b3cf326455db1 100644 --- a/arch/riscv/mm/pageattr.c +++ b/arch/riscv/mm/pageattr.c @@ -386,21 +386,6 @@ int set_direct_map_default_noflush(struct page *page, unsigned int nr) PAGE_KERNEL, __pgprot(_PAGE_EXEC)); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - pgprot_t set, clear; - - if (valid) { - set = PAGE_KERNEL; - clear = __pgprot(_PAGE_EXEC); - } else { - set = __pgprot(0); - clear = __pgprot(_PAGE_PRESENT); - } - - return __set_memory((unsigned long)page_address(page), nr, set, clear); -} - #ifdef CONFIG_DEBUG_PAGEALLOC static int debug_pagealloc_set_page(pte_t *pte, unsigned long addr, void *data) { diff --git a/arch/s390/include/asm/set_memory.h b/arch/s390/include/asm/set_memory.h index 6b0aa9147ed8e6..e3562bf0c1aa5e 100644 --- a/arch/s390/include/asm/set_memory.h +++ b/arch/s390/include/asm/set_memory.h @@ -62,7 +62,6 @@ __SET_MEMORY_FUNC(set_memory_4k, SET_MEMORY_4K) int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_default_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); #endif diff --git a/arch/s390/mm/pageattr.c b/arch/s390/mm/pageattr.c index 7549543d624125..80e834e8b8e1b5 100644 --- a/arch/s390/mm/pageattr.c +++ b/arch/s390/mm/pageattr.c @@ -392,18 +392,6 @@ int set_direct_map_default_noflush(struct page *page, unsigned int nr) return __set_memory((unsigned long)page_to_virt(page), nr, SET_MEMORY_DEF); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - unsigned long flags; - - if (valid) - flags = SET_MEMORY_DEF; - else - flags = SET_MEMORY_INV; - - return __set_memory((unsigned long)page_to_virt(page), nr, flags); -} - bool kernel_page_present(struct page *page) { unsigned long addr; diff --git a/arch/x86/include/asm/set_memory.h b/arch/x86/include/asm/set_memory.h index 0c4235d159f483..39271a5ea92527 100644 --- a/arch/x86/include/asm/set_memory.h +++ b/arch/x86/include/asm/set_memory.h @@ -88,7 +88,6 @@ int set_pages_rw(struct page *page, int numpages); int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_default_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); extern int kernel_set_to_readonly; diff --git a/arch/x86/mm/pat/set_memory.c b/arch/x86/mm/pat/set_memory.c index 4d07a9fbc43a78..a1a061d995b315 100644 --- a/arch/x86/mm/pat/set_memory.c +++ b/arch/x86/mm/pat/set_memory.c @@ -2666,14 +2666,6 @@ int set_direct_map_default_noflush(struct page *page, unsigned int nr) return __set_pages_p(page, nr, 0); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - if (valid) - return __set_pages_p(page, nr, 0); - - return __set_pages_np(page, nr, 0); -} - #ifdef CONFIG_DEBUG_PAGEALLOC void __kernel_map_pages(struct page *page, int numpages, int enable) { diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h index 0b77f1d7d8b9ce..3fe293cfed8cc8 100644 --- a/include/linux/set_memory.h +++ b/include/linux/set_memory.h @@ -36,12 +36,6 @@ static inline int set_direct_map_default_noflush(struct page *page, return 0; } -static inline int set_direct_map_valid_noflush(struct page *page, - unsigned nr, bool valid) -{ - return 0; -} - static inline bool kernel_page_present(struct page *page) { return true; From 7069903c06ad74476864a6d6141cc80a7fa64396 Mon Sep 17 00:00:00 2001 From: Hao Jia Date: Fri, 28 Aug 2026 16:31:49 +0800 Subject: [PATCH 590/857] zram: fix idle age_sec underflow in idle_store() After commit 2e8ff2f51dde ("zram: use u32 for entry ac_time tracking"), idle_store() computes the idle cutoff as: cutoff = ktime_sub((u32)ktime_get_boottime_seconds(), age_sec); Because the left operand is cast to u32, when age_sec exceeds the current uptime the subtraction wraps modulo 2^32 and the huge result is zero-extended into the s64 cutoff. mark_idle() then marks every entry as idle instead of matching nothing. For instance, running echo 86400 > /sys/block/zramX/idle on a machine up for only two minutes marks all newly written pages idle and hands them to idle writeback and recompression. No slot can have been accessed before the system booted, so an age_sec that reaches back past uptime cannot match any slot. Return early in that case, without walking the table or taking any slot locks. Track the cutoff as time64_t rather than ktime_t. Both cutoff and ac_time are boot-time values in seconds, so a plain arithmetic comparison against ac_time in mark_idle() is correct and no ktime helpers are needed. Link: https://lore.kernel.org/20260828083149.45760-1-jiahao.kernel@gmail.com Fixes: 2e8ff2f51dde ("zram: use u32 for entry ac_time tracking") Signed-off-by: Hao Jia Suggested-by: Sergey Senozhatsky Cc: Brian Geffon Cc: Jens Axboe Cc: Minchan Kim Cc: Signed-off-by: Andrew Morton --- drivers/block/zram/zram_drv.c | 21 +++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c index a9b3bb1d3bef35..4ba0f77b2abd80 100644 --- a/drivers/block/zram/zram_drv.c +++ b/drivers/block/zram/zram_drv.c @@ -415,7 +415,7 @@ static ssize_t mem_used_max_store(struct device *dev, * Mark all pages which are older than or equal to cutoff as IDLE. * Callers should hold the zram init lock in read mode */ -static void mark_idle(struct zram *zram, ktime_t cutoff) +static void mark_idle(struct zram *zram, time64_t cutoff) { int is_idle = 1; unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; @@ -439,7 +439,7 @@ static void mark_idle(struct zram *zram, ktime_t cutoff) #ifdef CONFIG_ZRAM_TRACK_ENTRY_ACTIME is_idle = !cutoff || - ktime_after(cutoff, zram->table[index].attr.ac_time); + cutoff > zram->table[index].attr.ac_time; #endif if (is_idle) set_slot_flag(zram, index, ZRAM_IDLE); @@ -453,21 +453,26 @@ static ssize_t idle_store(struct device *dev, struct device_attribute *attr, const char *buf, size_t len) { struct zram *zram = dev_to_zram(dev); - ktime_t cutoff = 0; + time64_t cutoff = 0; if (!sysfs_streq(buf, "all")) { /* * If it did not parse as 'all' try to treat it as an integer * when we have memory tracking enabled. */ + time64_t uptime; u32 age_sec; - if (IS_ENABLED(CONFIG_ZRAM_TRACK_ENTRY_ACTIME) && - !kstrtouint(buf, 0, &age_sec)) - cutoff = ktime_sub((u32)ktime_get_boottime_seconds(), - age_sec); - else + if (!IS_ENABLED(CONFIG_ZRAM_TRACK_ENTRY_ACTIME) || + kstrtouint(buf, 0, &age_sec)) return -EINVAL; + + /* No slot can be older than the system uptime */ + uptime = ktime_get_boottime_seconds(); + if (age_sec >= uptime) + return len; + + cutoff = uptime - age_sec; } guard(rwsem_read)(&zram->dev_lock); From b8aaed6cbb9cf186fbde025bd30b41ce883586ed Mon Sep 17 00:00:00 2001 From: Vernon Yang Date: Fri, 28 Aug 2026 13:59:24 +0800 Subject: [PATCH 591/857] mm: khugepaged: fix swap entry value to folio_pfn() Patch series "mm: khugepaged: fix tracepoint UAF", v4. The khugepaged tracepoints take a folio pointer and call folio_pfn(), but by then the folio may no longer be valid: freed after folio_put(), folio_unlock() or pte_unmap_unlock(), or not a folio at all but an xarray-encoded swap entry. On classic SPARSEMEM, dereferencing it oopses khugepaged as soon as the trace event is enabled; on other memory models it merely prints a bogus pfn. Pass the pfn to the tracepoints directly, captured while the folio is still pinned, closing the use-after-free windows in mm_khugepaged_scan_file(), mm_khugepaged_scan_pmd() and mm_khugepaged_collapse_file(). This patch (of 3): When the swap entries found exceed max_ptes_swap, the loop is left via break with folio still holding the xarray value that encodes the swap entry, not valid folio pointer. That value is passed to trace_mm_khugepaged_scan_file(), which feeds it to folio_pfn(). On FLATMEM and SPARSEMEM_VMEMMAP, the page_to_pfn() is plain pointer arithmetic, so the trace event merely prints bogus scan_pfn. On classic SPARSEMEM, the page_to_pfn() reads page->flags, dereferencing the tiny encoded integer and oopsing khugepaged whenever the trace event is enabled. So when folio is the swap entry value, simply set pfn to -1, just like exhausted scan naturally. And the folio_put() has maybe dropped the last reference of folio. The trace_mm_khugepaged_scan_file() is left with a dangling folio pointer. so using the folio_pfn() before dropping the reference, closing use-after-free window. About calling the respective trace_xxx() functions separately on success and failure, refer to [1]. Link: https://lore.kernel.org/20260828055926.346744-1-vernon2gm@gmail.com Link: https://lore.kernel.org/20260828055926.346744-2-vernon2gm@gmail.com Link: https://lore.kernel.org/linux-mm/ao6jVbVHLUmuY2UA@gremlin/ [1] Fixes: d41fd2016ed0 ("mm/khugepaged: add tracepoint to hpage_collapse_scan_file()") Signed-off-by: Vernon Yang Cc: Barry Song Cc: David Hildenbrand Cc: Dev Jain Cc: Lance Yang Cc: Lorenzo Stoakes Cc: Ryan Roberts Cc: Zach O'Keefe Cc: Signed-off-by: Andrew Morton --- include/trace/events/huge_memory.h | 6 +++--- mm/khugepaged.c | 11 ++++++++++- 2 files changed, 13 insertions(+), 4 deletions(-) diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h index 5a48c5406cce49..7b526528f85b80 100644 --- a/include/trace/events/huge_memory.h +++ b/include/trace/events/huge_memory.h @@ -178,10 +178,10 @@ TRACE_EVENT(mm_collapse_huge_page_swapin, TRACE_EVENT(mm_khugepaged_scan_file, - TP_PROTO(struct mm_struct *mm, struct folio *folio, struct file *file, + TP_PROTO(struct mm_struct *mm, unsigned long pfn, struct file *file, int present, int swap, int result), - TP_ARGS(mm, folio, file, present, swap, result), + TP_ARGS(mm, pfn, file, present, swap, result), TP_STRUCT__entry( __field(struct mm_struct *, mm) @@ -194,7 +194,7 @@ TRACE_EVENT(mm_khugepaged_scan_file, TP_fast_assign( __entry->mm = mm; - __entry->pfn = folio ? folio_pfn(folio) : -1; + __entry->pfn = pfn; __assign_str(filename); __entry->present = present; __entry->swap = swap; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 75639298efc271..b597a3e686062d 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2683,6 +2683,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, int present, swap; int node = NUMA_NO_NODE; enum scan_result result = SCAN_SUCCEED; + unsigned long failed_pfn = -1; present = 0; swap = 0; @@ -2715,6 +2716,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, if (is_pmd_order(folio_order(folio))) { result = SCAN_PTE_MAPPED_HUGEPAGE; + failed_pfn = folio_pfn(folio); /* * PMD-sized THP implies that we can only try * retracting the PTE table. @@ -2726,6 +2728,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, node = folio_nid(folio); if (collapse_scan_abort(node, cc)) { result = SCAN_SCAN_ABORT; + failed_pfn = folio_pfn(folio); folio_put(folio); break; } @@ -2733,12 +2736,14 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, if (!folio_test_lru(folio)) { result = SCAN_PAGE_LRU; + failed_pfn = folio_pfn(folio); folio_put(folio); break; } if (folio_expected_ref_count(folio) + 1 != folio_ref_count(folio)) { result = SCAN_PAGE_COUNT; + failed_pfn = folio_pfn(folio); folio_put(folio); break; } @@ -2771,9 +2776,13 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, } else { result = collapse_file(mm, addr, file, start, cc); } + trace_mm_khugepaged_scan_file(mm, -1, file, present, swap, + SCAN_SUCCEED); + } else { + trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, + swap, result); } - trace_mm_khugepaged_scan_file(mm, folio, file, present, swap, result); return result; } From d27996011b7c57b347087aee866fc12147fe3c1e Mon Sep 17 00:00:00 2001 From: Vernon Yang Date: Fri, 28 Aug 2026 13:59:25 +0800 Subject: [PATCH 592/857] mm: khugepaged: fix folio is used after pte_unmap_unlock() After the page table lock has dropped, the folio can be freed concurrently. The trace_mm_khugepaged_scan_pmd() is left with a dangling folio pointer. So using the folio_pfn() before dropping the page table lock, closing use-after-free window. And other pre-existing bug, When the `for (i = 0; i < HPAGE_PMD_NR; i++)` iteration to terminate and the folio operation preceding is normal, but pfn will be incorrect. so we really only trace the PFN if it really was problematic. About calling the respective trace_xxx() functions separately on success and failure, refer to [1]. Link: https://lore.kernel.org/20260828055926.346744-3-vernon2gm@gmail.com Link: https://lore.kernel.org/linux-mm/ao6jVbVHLUmuY2UA@gremlin/ [1] Fixes: 7d2eba0557c1 ("mm: add tracepoint for scanning pages") Signed-off-by: Vernon Yang Acked-by: Lorenzo Stoakes (ARM) Cc: Barry Song Cc: David Hildenbrand Cc: Dev Jain Cc: Lance Yang Cc: Ryan Roberts Cc: Zach O'Keefe Cc: Signed-off-by: Andrew Morton --- include/trace/events/huge_memory.h | 6 +++--- mm/khugepaged.c | 17 ++++++++++++++--- 2 files changed, 17 insertions(+), 6 deletions(-) diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h index 7b526528f85b80..fa828967e1fb0e 100644 --- a/include/trace/events/huge_memory.h +++ b/include/trace/events/huge_memory.h @@ -55,10 +55,10 @@ SCAN_STATUS TRACE_EVENT(mm_khugepaged_scan_pmd, - TP_PROTO(struct mm_struct *mm, struct folio *folio, + TP_PROTO(struct mm_struct *mm, unsigned long pfn, int referenced, int none_or_zero, int status, int unmapped), - TP_ARGS(mm, folio, referenced, none_or_zero, status, unmapped), + TP_ARGS(mm, pfn, referenced, none_or_zero, status, unmapped), TP_STRUCT__entry( __field(struct mm_struct *, mm) @@ -71,7 +71,7 @@ TRACE_EVENT(mm_khugepaged_scan_pmd, TP_fast_assign( __entry->mm = mm; - __entry->pfn = folio ? folio_pfn(folio) : -1; + __entry->pfn = pfn; __entry->referenced = referenced; __entry->none_or_zero = none_or_zero; __entry->status = status; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index b597a3e686062d..4d360ae87769f5 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1612,6 +1612,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, enum scan_result result = SCAN_FAIL; struct page *page = NULL; struct folio *folio = NULL; + unsigned long failed_pfn = -1; unsigned long addr; unsigned long enabled_orders; spinlock_t *ptl; @@ -1706,11 +1707,13 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, if (cc->is_khugepaged && !(vma->vm_flags & VM_DROPPABLE) && folio_test_lazyfree(folio) && !pte_dirty(pteval)) { result = SCAN_PAGE_LAZYFREE; + failed_pfn = folio_pfn(folio); goto out_unmap; } if (!folio_test_anon(folio)) { result = SCAN_PAGE_ANON; + failed_pfn = folio_pfn(folio); goto out_unmap; } @@ -1721,6 +1724,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, if (folio_maybe_mapped_shared(folio)) { if (++shared > max_ptes_shared) { result = SCAN_EXCEED_SHARED_PTE; + failed_pfn = folio_pfn(folio); count_collapse_event(HPAGE_PMD_ORDER, THP_SCAN_EXCEED_SHARED_PTE, MTHP_STAT_COLLAPSE_EXCEED_SHARED); goto out_unmap; @@ -1738,15 +1742,18 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, node = folio_nid(folio); if (collapse_scan_abort(node, cc)) { result = SCAN_SCAN_ABORT; + failed_pfn = folio_pfn(folio); goto out_unmap; } cc->node_load[node]++; if (!folio_test_lru(folio)) { result = SCAN_PAGE_LRU; + failed_pfn = folio_pfn(folio); goto out_unmap; } if (folio_test_locked(folio)) { result = SCAN_PAGE_LOCK; + failed_pfn = folio_pfn(folio); goto out_unmap; } @@ -1759,6 +1766,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, */ if (folio_expected_ref_count(folio) != folio_ref_count(folio)) { result = SCAN_PAGE_COUNT; + failed_pfn = folio_pfn(folio); goto out_unmap; } @@ -1782,10 +1790,13 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, unmapped, cc, enabled_orders); /* mmap_lock was released above, set lock_dropped */ *lock_dropped = true; - } + trace_mm_khugepaged_scan_pmd(mm, -1, referenced, none_or_zero, + SCAN_SUCCEED, unmapped); + } else { out: - trace_mm_khugepaged_scan_pmd(mm, folio, referenced, - none_or_zero, result, unmapped); + trace_mm_khugepaged_scan_pmd(mm, failed_pfn, referenced, + none_or_zero, result, unmapped); + } return result; } From cea5843ecd7ae91406fc28f78c927301c04d9ddf Mon Sep 17 00:00:00 2001 From: Vernon Yang Date: Fri, 28 Aug 2026 13:59:26 +0800 Subject: [PATCH 593/857] mm: khugepaged: fix folio is used after folio_put/unlock() On the rollback path, folio_put() has already dropped the last reference of new_folio. On the success path, new_folio is already unlocked and can be freed concurrently. The trace_mm_khugepaged_collapse_file() is left with a dangling folio pointer. So using the folio_pfn() before dropping the reference, closing use-after-free window. Link: https://lore.kernel.org/20260828055926.346744-4-vernon2gm@gmail.com Fixes: 4c9473e87e75 ("mm/khugepaged: add tracepoint to collapse_file()") Signed-off-by: Vernon Yang Acked-by: Lorenzo Stoakes (ARM) Cc: Barry Song Cc: David Hildenbrand Cc: Dev Jain Cc: Lance Yang Cc: Ryan Roberts Cc: Zach O'Keefe Cc: Signed-off-by: Andrew Morton --- include/trace/events/huge_memory.h | 6 +++--- mm/khugepaged.c | 4 +++- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h index fa828967e1fb0e..5fb4d92cfd8408 100644 --- a/include/trace/events/huge_memory.h +++ b/include/trace/events/huge_memory.h @@ -211,10 +211,10 @@ TRACE_EVENT(mm_khugepaged_scan_file, ); TRACE_EVENT(mm_khugepaged_collapse_file, - TP_PROTO(struct mm_struct *mm, struct folio *new_folio, pgoff_t index, + TP_PROTO(struct mm_struct *mm, unsigned long new_pfn, pgoff_t index, unsigned long addr, bool is_shmem, struct file *file, int nr, int result), - TP_ARGS(mm, new_folio, index, addr, is_shmem, file, nr, result), + TP_ARGS(mm, new_pfn, index, addr, is_shmem, file, nr, result), TP_STRUCT__entry( __field(struct mm_struct *, mm) __field(unsigned long, hpfn) @@ -228,7 +228,7 @@ TRACE_EVENT(mm_khugepaged_collapse_file, TP_fast_assign( __entry->mm = mm; - __entry->hpfn = new_folio ? folio_pfn(new_folio) : -1; + __entry->hpfn = new_pfn; __entry->index = index; __entry->addr = addr; __entry->is_shmem = is_shmem; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 4d360ae87769f5..52b4476898d96f 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2256,6 +2256,7 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, struct address_space *mapping = file->f_mapping; struct page *dst; struct folio *folio, *tmp, *new_folio; + unsigned long new_pfn = -1; pgoff_t index = 0, end = start + HPAGE_PMD_NR; LIST_HEAD(pagelist); XA_STATE_ORDER(xas, &mapping->i_pages, start, HPAGE_PMD_ORDER); @@ -2275,6 +2276,7 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, result = alloc_charge_folio(&new_folio, mm, cc, HPAGE_PMD_ORDER); if (result != SCAN_SUCCEED) goto out; + new_pfn = folio_pfn(new_folio); mapping_set_update(&xas, mapping); @@ -2678,7 +2680,7 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, folio_put(new_folio); out: VM_BUG_ON(!list_empty(&pagelist)); - trace_mm_khugepaged_collapse_file(mm, new_folio, index, addr, is_shmem, file, HPAGE_PMD_NR, result); + trace_mm_khugepaged_collapse_file(mm, new_pfn, index, addr, is_shmem, file, HPAGE_PMD_NR, result); return result; } From 6576dc1dd4cdca6013c346832dde038986ba7842 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Fri, 28 Aug 2026 17:17:53 +0800 Subject: [PATCH 594/857] mm: vmalloc: fix vmap_purge_lock livelock under memory pressure The vmap_purge_lock mutex can be held for an extended period by __purge_vmap_area_lazy() which calls flush_work() to wait for purge_vmap_node workers while holding the lock. Under memory pressure, those workers may themselves be blocked in direct reclaim trying to acquire the same lock via the vmap_node_shrink_scan() shrinker callback, creating a circular dependency that deadlocks the entire system. Two places acquire vmap_purge_lock from paths that can be reached during direct reclaim: 1. vmap_node_shrink_scan(): replace blocking guard(mutex) with mutex_trylock(). This is a shrinker that only decays the vmap pool and returns SHRINK_STOP without freeing memory; skipping a decay cycle when the lock is contended is harmless and prevents tasks from piling up on the mutex in the direct reclaim path. 2. reclaim_and_purge_vmap_areas(): replace mutex_lock() with mutex_trylock(). This is called from the vmalloc allocation overflow path; if trylock fails, another thread is already purging and the allocator's retry will find freed space. The notifier chain provides a fallback if the retry still fails. Both trylock failures break the circular dependency: the lock holder's flush_work() can complete because workers are no longer blocked on vmap_purge_lock in the direct reclaim path. Link: https://lore.kernel.org/20260828091753.299295-1-ye.liu@linux.dev Fixes: 7679ba6b36db ("mm: vmalloc: add a shrinker to drain vmap pools") Signed-off-by: Ye Liu Suggested-by: Uladzislau Rezki Suggested-by: Dev Jain Reviewed-by: Uladzislau Rezki (Sony) Signed-off-by: Andrew Morton --- mm/vmalloc.c | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 676fe96ae84145..6ed6c160abed78 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -2445,7 +2445,8 @@ static bool __purge_vmap_area_lazy(unsigned long start, unsigned long end, static void reclaim_and_purge_vmap_areas(void) { - mutex_lock(&vmap_purge_lock); + if (!mutex_trylock(&vmap_purge_lock)) + return; purge_fragmented_blocks_allcpus(); __purge_vmap_area_lazy(ULONG_MAX, 0, true); mutex_unlock(&vmap_purge_lock); @@ -5526,10 +5527,20 @@ vmap_node_shrink_scan(struct shrinker *shrink, struct shrink_control *sc) { struct vmap_node *vn; - guard(mutex)(&vmap_purge_lock); + /* + * This shrinker is invoked from direct reclaim where memory + * pressure is already high. Blocking on vmap_purge_lock here + * can deadlock the system: the lock holder may be blocked in + * flush_work() waiting for a worker that is stuck in this same + * reclaim path trying to acquire the same lock. Use trylock + * to avoid this; skipping a pool decay cycle is harmless. + */ + if (!mutex_trylock(&vmap_purge_lock)) + return SHRINK_STOP; for_each_vmap_node(vn) decay_va_pool_node(vn, true); + mutex_unlock(&vmap_purge_lock); return SHRINK_STOP; } From 295333e25af9f7d2d82155cfae9de94c4cf29159 Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:26 +0800 Subject: [PATCH 595/857] mm: memcontrol: make obj_cgroup_memcg() handle NULL objcg Patch series "make unused huge shrinker memcg aware", v4. The shmem unused huge shrinker maintains a per-superblock list of inodes whose tail huge folio extends beyond i_size. Because this list is not memcg aware, reclaim triggered by memcg A can scan inodes across the entire superblock and split huge folios charged to unrelated memcg B, causing unexpected impact on it. In the worst case, memcg A has no reclaimable shmem at all, making the reclaim entirely useless and incurring unnecessary latency. We observed this in production, where page lock contention during split caused multi-hundred-millisecond stalls: tid 11340 comm scanner locked a page for 182264 us! kstack: unlock_page+1 split_huge_page_to_list+3135 shmem_unused_huge_shrink+767 super_cache_scan+329 do_shrink_slab+291 shrink_slab+533 shrink_node+400 do_try_to_free_pages+206 try_to_free_mem_cgroup_pages+262 try_charge_memcg+591 mem_cgroup_charge+136 __handle_mm_fault+2431 handle_mm_fault+194 do_user_addr_fault+462 __do_page_fault+176 do_page_fault+48 page_fault+62 Usama's recent patch [1] prevents the shmem unused shrinker from being invoked during memcg-level reclaim altogether, but this is overly conservative: we can do better by reclaiming only the shmem charged to the reclaiming memcg. This series converts the shrinker list to a memcg-aware list_lru, so that non-root memcg reclaim walks only candidates charged to the reclaiming memcg. Global reclaim, root memcg reclaim and shmem quota reclaim retain their existing global semantics. To avoid pinning a dying memcg through a long-lived CSS reference, each inode stores an obj_cgroup reference instead of a mem_cgroup reference. The list_lru add/delete paths resolve the current memcg from the objcg under RCU, staying consistent with list_lru's own memcg migration on offline. This patch (of 3): obj_cgroup_memcg() currently requires a non-NULL objcg, so callers that may hold a NULL objcg must guard the call with an explicit NULL check. This pattern is duplicated in folio_memcg(), folio_memcg_check(), mm/page_owner.c, and mm/zswap.c. Teach obj_cgroup_memcg() to accept NULL and return NULL in that case, then remove the redundant NULL checks at the call sites. Also remove the mem_cgroup_from_entry() wrapper in zswap, which existed solely to provide this NULL-safe behaviour, and replace its two callers with direct obj_cgroup_memcg() calls. No functional change intended. Link: https://lore.kernel.org/cover.1786955972.git.zhengqi.arch@bytedance.com Link: https://lore.kernel.org/09bcf74312246a6e4146be8a0cb9787f8beddb28.1786955972.git.zhengqi.arch@bytedance.com Signed-off-by: Qi Zheng Acked-by: Shakeel Butt Cc: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Signed-off-by: Andrew Morton --- include/linux/memcontrol.h | 11 ++++++++--- mm/page_owner.c | 2 +- mm/zswap.c | 17 ++--------------- 3 files changed, 11 insertions(+), 19 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 7d1c0ce189a887..da625d2edb3bab 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -380,7 +380,7 @@ enum objext_flags { static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) { lockdep_assert_once(rcu_read_lock_held() || lockdep_is_held(&cgroup_mutex)); - return READ_ONCE(objcg->memcg); + return objcg ? READ_ONCE(objcg->memcg) : NULL; } /* @@ -433,7 +433,7 @@ static inline struct mem_cgroup *folio_memcg(struct folio *folio) { struct obj_cgroup *objcg = folio_objcg(folio); - return objcg ? obj_cgroup_memcg(objcg) : NULL; + return obj_cgroup_memcg(objcg); } /* @@ -476,7 +476,7 @@ static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) objcg = (void *)(memcg_data & ~OBJEXTS_FLAGS_MASK); - return objcg ? obj_cgroup_memcg(objcg) : NULL; + return obj_cgroup_memcg(objcg); } static inline struct mem_cgroup *page_memcg_check(struct page *page) @@ -1050,6 +1050,11 @@ void mem_cgroup_flush_workqueue(void); extern int mem_cgroup_init(void); #else /* CONFIG_MEMCG */ +static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) +{ + return NULL; +} + #define MEM_CGROUP_ID_SHIFT 0 #define root_mem_cgroup (NULL) diff --git a/mm/page_owner.c b/mm/page_owner.c index fbbda7ba914ba5..3fc37d9b908ef0 100644 --- a/mm/page_owner.c +++ b/mm/page_owner.c @@ -575,7 +575,7 @@ static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret, } objcg = (void *)(memcg_data & ~OBJEXTS_FLAGS_MASK); - memcg = objcg ? obj_cgroup_memcg(objcg) : NULL; + memcg = obj_cgroup_memcg(objcg); if (!memcg) goto out_unlock; diff --git a/mm/zswap.c b/mm/zswap.c index 37f34e406c8e3b..c1dc60926bad99 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -647,19 +647,6 @@ static int zswap_enabled_param_set(const char *val, * lru functions **********************************/ -/* should be called under RCU */ -#ifdef CONFIG_MEMCG -static inline struct mem_cgroup *mem_cgroup_from_entry(struct zswap_entry *entry) -{ - return entry->objcg ? obj_cgroup_memcg(entry->objcg) : NULL; -} -#else -static inline struct mem_cgroup *mem_cgroup_from_entry(struct zswap_entry *entry) -{ - return NULL; -} -#endif - static inline int entry_to_nid(struct zswap_entry *entry) { return page_to_nid(virt_to_page(entry)); @@ -682,7 +669,7 @@ static void zswap_lru_add(struct zswap_entry *entry) * Similar reasoning holds for list_lru_del(). */ rcu_read_lock(); - memcg = mem_cgroup_from_entry(entry); + memcg = obj_cgroup_memcg(entry->objcg); /* will always succeed */ list_lru_add(&zswap_list_lru, &entry->lru, nid, memcg); rcu_read_unlock(); @@ -694,7 +681,7 @@ static void zswap_lru_del(struct zswap_entry *entry) struct mem_cgroup *memcg; rcu_read_lock(); - memcg = mem_cgroup_from_entry(entry); + memcg = obj_cgroup_memcg(entry->objcg); /* will always succeed */ list_lru_del(&zswap_list_lru, &entry->lru, nid, memcg); rcu_read_unlock(); From d624fb90fdb4825679f744bc3247e45ac6a731df Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:27 +0800 Subject: [PATCH 596/857] mm: shmem: move unused huge shrinklist queuing past the truncation check The shmem_get_folio_gfp() adds the inode to the unused huge shrinker list at the alloced label, but a subsequent truncation check may still fail and remove the folio, leaving the inode on the list with a stale folio. The original code works because the shrinker re-looks-up the folio and drops stale entries, but it is cleaner to queue the inode only after all checks that might remove the folio have passed. So just make the pure structural move with no functional change, and it serves as preparation for the memcg-aware shrinker conversion. Link: https://lore.kernel.org/17fcf64faec0dfbbe6cf8a97924e3f39cd3c51a4.1786955972.git.zhengqi.arch@bytedance.com Signed-off-by: Qi Zheng Reviewed-by: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Signed-off-by: Andrew Morton --- mm/shmem.c | 48 +++++++++++++++++++++++++++--------------------- 1 file changed, 27 insertions(+), 21 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index de144a9a9558b7..d49cbffbe7e16b 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2536,27 +2536,6 @@ static int shmem_get_folio_gfp(struct inode *inode, pgoff_t index, alloced: alloced = true; - if (folio_test_large(folio) && - DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE) < - folio_next_index(folio)) { - struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); - struct shmem_inode_info *info = SHMEM_I(inode); - /* - * Part of the large folio is beyond i_size: subject - * to shrink under memory pressure. - */ - spin_lock(&sbinfo->shrinklist_lock); - /* - * _careful to defend against unlocked access to - * ->shrink_list in shmem_unused_huge_shrink() - */ - if (list_empty_careful(&info->shrinklist)) { - list_add_tail(&info->shrinklist, - &sbinfo->shrinklist); - sbinfo->shrinklist_len++; - } - spin_unlock(&sbinfo->shrinklist_lock); - } if (sgp == SGP_WRITE) folio_set_referenced(folio); @@ -2586,6 +2565,33 @@ static int shmem_get_folio_gfp(struct inode *inode, pgoff_t index, error = -EINVAL; goto unlock; } + + /* + * Queue the inode on the shrink list only after all checks that might + * remove the folio have passed. Otherwise the inode could be left on + * the shrinker list with a stale folio. + */ + if (alloced && folio_test_large(folio) && + DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE) < folio_next_index(folio)) { + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + struct shmem_inode_info *info = SHMEM_I(inode); + /* + * Part of the large folio is beyond i_size: subject + * to shrink under memory pressure. + */ + spin_lock(&sbinfo->shrinklist_lock); + /* + * _careful to defend against unlocked access to + * ->shrink_list in shmem_unused_huge_shrink() + */ + if (list_empty_careful(&info->shrinklist)) { + list_add_tail(&info->shrinklist, + &sbinfo->shrinklist); + sbinfo->shrinklist_len++; + } + spin_unlock(&sbinfo->shrinklist_lock); + } + out: *foliop = folio; return 0; From e20e510cf72b8fb41e2d823ca6839d80fee53051 Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:28 +0800 Subject: [PATCH 597/857] mm: shmem: make unused huge shrinker memcg aware The shmem unused huge shrinker keeps a per-superblock list of inodes whose tail huge folio extends beyond i_size. Since that list is not memcg aware, reclaim triggered by one memcg can scan inodes from the whole superblock and split shmem huge folios charged to unrelated memcgs. Convert the shrink list to a memcg-aware list_lru. Queue each inode on the list_lru sublist matching the memcg and node of the current tail huge folio, so non-root memcg reclaim only walks candidates charged to the reclaiming memcg. Global reclaim, root memcg reclaim and shmem quota reclaim keep global semantics. Rather than pinning a struct mem_cgroup reference in shmem_inode_info, store a struct obj_cgroup reference instead. The list_lru add and delete paths resolve the current memcg from the objcg under RCU, so that memcg offline and list_lru entry migration remain consistent: list_lru migrates entries to the parent memcg sublist on offline, and obj_cgroup_memcg() follows the same reparenting, ensuring the correct sublist is always found at delete time. This avoids pinning a dying memcg through a long-lived CSS reference. The list_lru still tracks inodes while the actual split target is the current tail huge folio, so validate the folio memcg/node during scan. If the folio no longer matches the reclaim context or splitting cannot proceed, requeue the inode according to the current tail folio; if the inode is no longer shrinkable, drop the scan entry. This can be tested with the shrinker debugfs interface by allocating 32 tmpfs tail THPs in each of two memcgs, then scanning the sb-tmpfs shrinker with memcg A's cgroup id: before A scan after A scan base A=64M, B=64M A=64M, B=64M (per-memcg count is skipped) patched A=64M, B=64M A=0, B=64M Link: https://lore.kernel.org/94cc7fe1fd645254ae90effc1c0687678438d940.1786955972.git.zhengqi.arch@bytedance.com Signed-off-by: Qi Zheng Reviewed-by: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Qinyun Tan Signed-off-by: Andrew Morton --- include/linux/shmem_fs.h | 12 +- mm/shmem.c | 362 ++++++++++++++++++++++++++++++--------- 2 files changed, 289 insertions(+), 85 deletions(-) diff --git a/include/linux/shmem_fs.h b/include/linux/shmem_fs.h index 5663dff53186e2..a7c7a96a7cbf90 100644 --- a/include/linux/shmem_fs.h +++ b/include/linux/shmem_fs.h @@ -11,6 +11,7 @@ #include #include #include +#include /* inode in-kernel data */ @@ -54,6 +55,11 @@ struct shmem_inode_info { struct dquot __rcu *i_dquot[MAXQUOTAS]; #endif struct inode vfs_inode; + +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + struct obj_cgroup *shrinklist_objcg; + int shrinklist_nid; +#endif }; #define SHMEM_FL_USER_VISIBLE (FS_FL_USER_VISIBLE | FS_CASEFOLD_FL) @@ -83,9 +89,9 @@ struct shmem_sb_info { ino_t next_ino; /* The next per-sb inode number to use */ ino_t __percpu *ino_batch; /* The next per-cpu inode number to use */ struct mempolicy *mpol; /* default memory policy for mappings */ - spinlock_t shrinklist_lock; /* Protects shrinklist */ - struct list_head shrinklist; /* List of shinkable inodes */ - unsigned long shrinklist_len; /* Length of shrinklist */ +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + struct list_lru shrinklist; /* List of shrinkable inodes */ +#endif struct shmem_quota_limits qlimits; /* Default quota limits */ struct simple_xattr_cache xa_cache; }; diff --git a/mm/shmem.c b/mm/shmem.c index d49cbffbe7e16b..9c76032c396ee6 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -725,51 +725,258 @@ static const char *shmem_format_huge(int huge) } #endif -static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, - struct shrink_control *sc, unsigned long nr_to_free) +static bool is_shmem_unused_huge_isolated(struct shmem_inode_info *info) { - LIST_HEAD(list), *pos, *next; - struct inode *inode; + + return info->shrinklist_nid == -1; +} + +static void set_shmem_unused_huge_isolated(struct shmem_inode_info *info) +{ + info->shrinklist_nid = -1; +} + +static struct obj_cgroup *shmem_get_and_clear_objcg(struct shmem_inode_info *info) +{ + struct obj_cgroup *objcg = info->shrinklist_objcg; + + info->shrinklist_objcg = NULL; + + return objcg; +} + +#ifdef CONFIG_MEMCG +static struct obj_cgroup * +shmem_unused_huge_alloc_lru(struct shmem_sb_info *sbinfo, struct folio *folio, + gfp_t gfp) +{ + int ret; + + ret = folio_memcg_list_lru_alloc(folio, &sbinfo->shrinklist, gfp); + if (ret) + return ERR_PTR(ret); + + return get_obj_cgroup_from_folio(folio); +} +#else +static struct obj_cgroup * +shmem_unused_huge_alloc_lru(struct shmem_sb_info *sbinfo, struct folio *folio, + gfp_t gfp) +{ + return NULL; +} +#endif + +static void shmem_unused_huge_lru_add(struct shmem_sb_info *sbinfo, + struct list_head *item, int nid, + struct obj_cgroup *objcg) +{ + struct mem_cgroup *memcg; + + rcu_read_lock(); + memcg = obj_cgroup_memcg(objcg); + list_lru_add(&sbinfo->shrinklist, item, nid, memcg); + rcu_read_unlock(); +} + +static void shmem_unused_huge_lru_del(struct shmem_sb_info *sbinfo, + struct list_head *item, int nid, + struct obj_cgroup *objcg) +{ + struct mem_cgroup *memcg; + + rcu_read_lock(); + memcg = obj_cgroup_memcg(objcg); + list_lru_del(&sbinfo->shrinklist, item, nid, memcg); + rcu_read_unlock(); +} + +static void shmem_unused_huge_add(struct inode *inode, struct folio *folio, + gfp_t gfp) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + int nid = folio_nid(folio); + struct obj_cgroup *objcg = NULL, *old_objcg = NULL; + + objcg = shmem_unused_huge_alloc_lru(sbinfo, folio, gfp); + if (IS_ERR(objcg)) + return; + + spin_lock(&info->lock); + if (!list_empty(&info->shrinklist)) { + /* isolated on scan list, let shrink handle it */ + if (is_shmem_unused_huge_isolated(info)) + goto unlock; + + if (info->shrinklist_nid == nid && + info->shrinklist_objcg == objcg) + goto unlock; + + shmem_unused_huge_lru_del(sbinfo, &info->shrinklist, + info->shrinklist_nid, + info->shrinklist_objcg); + old_objcg = shmem_get_and_clear_objcg(info); + } + + info->shrinklist_objcg = objcg; + info->shrinklist_nid = nid; + shmem_unused_huge_lru_add(sbinfo, &info->shrinklist, nid, objcg); + objcg = NULL; +unlock: + spin_unlock(&info->lock); + obj_cgroup_put(old_objcg); + obj_cgroup_put(objcg); +} + +static void shmem_unused_huge_del(struct inode *inode) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + struct obj_cgroup *objcg = NULL; + + spin_lock(&info->lock); + if (!list_empty(&info->shrinklist)) { + shmem_unused_huge_lru_del(sbinfo, &info->shrinklist, + info->shrinklist_nid, + info->shrinklist_objcg); + objcg = shmem_get_and_clear_objcg(info); + } + spin_unlock(&info->lock); + + obj_cgroup_put(objcg); +} + +struct shmem_unused_huge_scan { + struct list_head list; + struct shrink_control *sc; +}; + +static enum lru_status shmem_unused_huge_isolate(struct list_head *item, + struct list_lru_one *lru, + void *arg) +{ + struct shmem_unused_huge_scan *scan = arg; struct shmem_inode_info *info; - struct folio *folio; - unsigned long batch = sc ? sc->nr_to_scan : 128; - unsigned long split = 0, freed = 0; + struct inode *inode; + struct obj_cgroup *objcg = NULL; - if (list_empty(&sbinfo->shrinklist)) - return SHRINK_STOP; + info = list_entry(item, struct shmem_inode_info, shrinklist); - spin_lock(&sbinfo->shrinklist_lock); - list_for_each_safe(pos, next, &sbinfo->shrinklist) { - info = list_entry(pos, struct shmem_inode_info, shrinklist); + /* + * Use trylock to avoid ABBA deadlock: add/del path takes info->lock + * before the list_lru bucket lock, while here the order is reversed. + */ + if (!spin_trylock(&info->lock)) + return LRU_SKIP; - /* pin the inode */ - inode = igrab(&info->vfs_inode); + /* pin the inode */ + inode = igrab(&info->vfs_inode); + /* inode is about to be evicted */ + if (!inode) { + list_lru_isolate(lru, item); + objcg = shmem_get_and_clear_objcg(info); + spin_unlock(&info->lock); + obj_cgroup_put(objcg); + return LRU_REMOVED; + } - /* inode is about to be evicted */ - if (!inode) { - list_del_init(&info->shrinklist); - goto next; - } + list_lru_isolate(lru, item); + objcg = shmem_get_and_clear_objcg(info); + set_shmem_unused_huge_isolated(info); + list_add_tail(&info->shrinklist, &scan->list); + spin_unlock(&info->lock); + obj_cgroup_put(objcg); - list_move(&info->shrinklist, &list); -next: - sbinfo->shrinklist_len--; - if (!--batch) - break; + return LRU_REMOVED; +} + +static bool is_shmem_unused_huge_match(struct folio *folio, + struct shrink_control *sc) +{ + struct mem_cgroup *memcg = NULL; + bool match; + + /* shmem quota reclaim has no NUMA node or memcg restriction */ + if (!sc) + return true; + + if (folio_nid(folio) != sc->nid) + return false; + + /* + * Only non-root memcg reclaim needs to match the folio charge against + * sc->memcg. Skip the folio memcg check for global shrinker reclaim and + * root memcg reclaim. + */ + if (!sc->memcg || mem_cgroup_is_root(sc->memcg)) + return true; + + memcg = get_mem_cgroup_from_folio(folio); + match = memcg == sc->memcg; + mem_cgroup_put(memcg); + + return match; +} + +static void shmem_unused_huge_drop(struct inode *inode) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + + spin_lock(&info->lock); + list_del_init(&info->shrinklist); + spin_unlock(&info->lock); +} + +static void shmem_unused_huge_requeue(struct inode *inode, struct folio *folio) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + struct obj_cgroup *objcg; + int nid = folio_nid(folio); + + objcg = shmem_unused_huge_alloc_lru(sbinfo, folio, GFP_NOWAIT); + if (IS_ERR(objcg)) { + shmem_unused_huge_drop(inode); + return; } - spin_unlock(&sbinfo->shrinklist_lock); - list_for_each_safe(pos, next, &list) { - pgoff_t next, end; + spin_lock(&info->lock); + /* Requeue the inode to shrinklist */ + list_del_init(&info->shrinklist); + shmem_unused_huge_lru_add(sbinfo, &info->shrinklist, nid, objcg); + info->shrinklist_objcg = objcg; + info->shrinklist_nid = nid; + spin_unlock(&info->lock); +} + +static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, + struct shrink_control *sc, unsigned long nr_to_free) +{ + struct shmem_unused_huge_scan scan; + struct inode *inode; + struct shmem_inode_info *info; + struct folio *folio; + struct list_head *pos, *next; + unsigned long split = 0, freed = 0; + + INIT_LIST_HEAD(&scan.list); + scan.sc = sc; + if (sc) + list_lru_shrink_walk(&sbinfo->shrinklist, sc, + shmem_unused_huge_isolate, &scan); + else + list_lru_walk(&sbinfo->shrinklist, shmem_unused_huge_isolate, + &scan, 128); + + list_for_each_safe(pos, next, &scan.list) { + pgoff_t folio_end, end; loff_t i_size; int ret; info = list_entry(pos, struct shmem_inode_info, shrinklist); inode = &info->vfs_inode; - if (nr_to_free && freed >= nr_to_free) - goto move_back; - i_size = i_size_read(inode); folio = filemap_get_entry(inode->i_mapping, i_size / PAGE_SIZE); if (!folio || xa_is_value(folio)) @@ -782,13 +989,19 @@ static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, } /* Check if there is anything to gain from splitting */ - next = folio_next_index(folio); + folio_end = folio_next_index(folio); end = shmem_fallocend(inode, DIV_ROUND_UP(i_size, PAGE_SIZE)); - if (end <= folio->index || end >= next) { + if (end <= folio->index || end >= folio_end) { folio_put(folio); goto drop; } + if (!is_shmem_unused_huge_match(folio, scan.sc)) + goto move_back; + + if (nr_to_free && freed >= nr_to_free) + goto move_back; + /* * Move the inode on the list back to shrinklist if we failed * to lock the page at this time. @@ -796,35 +1009,30 @@ static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, * Waiting for the lock may lead to deadlock in the * reclaim path. */ - if (!folio_trylock(folio)) { - folio_put(folio); + if (!folio_trylock(folio)) + goto move_back; + + if (!is_shmem_unused_huge_match(folio, scan.sc)) { + folio_unlock(folio); goto move_back; } ret = split_folio(folio); folio_unlock(folio); - folio_put(folio); /* If split failed move the inode on the list back to shrinklist */ if (ret) goto move_back; - freed += next - end; + freed += folio_end - end; split++; + folio_put(folio); drop: - list_del_init(&info->shrinklist); + shmem_unused_huge_drop(inode); goto put; move_back: - /* - * Make sure the inode is either on the global list or deleted - * from any local list before iput() since it could be deleted - * in another thread once we put the inode (then the local list - * is corrupted). - */ - spin_lock(&sbinfo->shrinklist_lock); - list_move(&info->shrinklist, &sbinfo->shrinklist); - sbinfo->shrinklist_len++; - spin_unlock(&sbinfo->shrinklist_lock); + shmem_unused_huge_requeue(inode, folio); + folio_put(folio); put: iput(inode); } @@ -837,7 +1045,7 @@ static long shmem_unused_huge_scan(struct super_block *sb, { struct shmem_sb_info *sbinfo = SHMEM_SB(sb); - if (!READ_ONCE(sbinfo->shrinklist_len)) + if (!list_lru_shrink_count(&sbinfo->shrinklist, sc)) return SHRINK_STOP; return shmem_unused_huge_shrink(sbinfo, sc, 0); @@ -848,21 +1056,21 @@ static long shmem_unused_huge_count(struct super_block *sb, { struct shmem_sb_info *sbinfo = SHMEM_SB(sb); - /* - * The per-superblock shrinklist is filesystem-global and does not - * honour sc->memcg, so it is only meaningful on the global (kswapd or - * root direct reclaim) shrink path. Skip the per-memcg iterations of - * shrink_slab_memcg() to avoid queueing duplicate global work. - */ - if (!mem_cgroup_shrink_is_root(sc)) - return 0; - - return READ_ONCE(sbinfo->shrinklist_len); + return list_lru_shrink_count(&sbinfo->shrinklist, sc); } #else /* !CONFIG_TRANSPARENT_HUGEPAGE */ #define shmem_huge SHMEM_HUGE_DENY +static void shmem_unused_huge_add(struct inode *inode, struct folio *folio, + gfp_t gfp) +{ +} + +static void shmem_unused_huge_del(struct inode *inode) +{ +} + static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, struct shrink_control *sc, unsigned long nr_to_free) { @@ -1418,14 +1626,7 @@ static void shmem_evict_inode(struct inode *inode) inode->i_size = 0; mapping_set_exiting(inode->i_mapping); shmem_truncate_range(inode, 0, (loff_t)-1); - if (!list_empty(&info->shrinklist)) { - spin_lock(&sbinfo->shrinklist_lock); - if (!list_empty(&info->shrinklist)) { - list_del_init(&info->shrinklist); - sbinfo->shrinklist_len--; - } - spin_unlock(&sbinfo->shrinklist_lock); - } + shmem_unused_huge_del(inode); while (!list_empty(&info->swaplist)) { /* Wait while shmem_unuse() is scanning this inode... */ wait_var_event(&info->stop_eviction, @@ -2573,25 +2774,12 @@ static int shmem_get_folio_gfp(struct inode *inode, pgoff_t index, */ if (alloced && folio_test_large(folio) && DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE) < folio_next_index(folio)) { - struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); - struct shmem_inode_info *info = SHMEM_I(inode); /* * Part of the large folio is beyond i_size: subject * to shrink under memory pressure. */ - spin_lock(&sbinfo->shrinklist_lock); - /* - * _careful to defend against unlocked access to - * ->shrink_list in shmem_unused_huge_shrink() - */ - if (list_empty_careful(&info->shrinklist)) { - list_add_tail(&info->shrinklist, - &sbinfo->shrinklist); - sbinfo->shrinklist_len++; - } - spin_unlock(&sbinfo->shrinklist_lock); + shmem_unused_huge_add(inode, folio, gfp); } - out: *foliop = folio; return 0; @@ -3068,6 +3256,10 @@ static struct inode *__shmem_get_inode(struct mnt_idmap *idmap, if (info->fsflags) shmem_set_inode_flags(inode, info->fsflags, NULL); INIT_LIST_HEAD(&info->shrinklist); +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + info->shrinklist_objcg = NULL; + info->shrinklist_nid = -1; +#endif INIT_LIST_HEAD(&info->swaplist); cache_no_acl(inode); if (sbinfo->noswap) @@ -4943,6 +5135,9 @@ static void shmem_put_super(struct super_block *sb) #endif free_percpu(sbinfo->ino_batch); percpu_counter_destroy(&sbinfo->used_blocks); +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + list_lru_destroy(&sbinfo->shrinklist); +#endif mpol_put(sbinfo->mpol); #ifdef CONFIG_TMPFS_XATTR simple_xattr_cache_cleanup(&sbinfo->xa_cache); @@ -5037,8 +5232,11 @@ static int shmem_fill_super(struct super_block *sb, struct fs_context *fc) raw_spin_lock_init(&sbinfo->stat_lock); if (percpu_counter_init(&sbinfo->used_blocks, 0, GFP_KERNEL)) goto failed; - spin_lock_init(&sbinfo->shrinklist_lock); - INIT_LIST_HEAD(&sbinfo->shrinklist); + +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + if (list_lru_init_memcg(&sbinfo->shrinklist, sb->s_shrink)) + goto failed; +#endif sb->s_maxbytes = MAX_LFS_FILESIZE; sb->s_blocksize = PAGE_SIZE; From b1cb889f6162b0d60d89f53d5ba83bd1175fca71 Mon Sep 17 00:00:00 2001 From: Qinyun Tan Date: Wed, 2 Sep 2026 17:32:02 +0800 Subject: [PATCH 598/857] mm/list_lru: disable memcg awareness under cgroup_disable=memory __list_lru_init() only collapses a memcg-aware list_lru into plain per-node lists when kmem accounting is disabled (cgroup.memory=nokmem). When the memory controller is disabled entirely (cgroup_disable=memory), mem_cgroup_kmem_disabled() is false, so the lru stays memcg aware even though no object will ever be charged to a memcg. This is more than a semantic inconsistency. folio_memcg_list_lru_alloc() trusts list_lru_memcg_aware() and dereferences the folio's memcg, which is always NULL with the controller disabled. The only mainline caller, folio_memcg_alloc_deferred(), papers over this with an explicit mem_cgroup_disabled() check. The shmem unused-huge shrinker conversion ("mm: shmem: make unused huge shrinker memcg aware") adds a second caller without such a guard, so booting with cgroup_disable=memory and writing to a huge=always tmpfs oopses: BUG: unable to handle page fault for address: 0000000000000488 RIP: 0010:folio_memcg_list_lru_alloc+0x41/0xf0 Call Trace: shmem_get_folio_gfp+0x1cd/0x7c0 shmem_write_begin+0x5d/0x100 generic_perform_write+0x89/0x2a0 shmem_file_write_iter+0x82/0x90 vfs_write+0x256/0x410 ksys_write+0x61/0xe0 do_syscall_64+0x8d/0x460 entry_SYSCALL_64_after_hwframe+0x76/0x7e The faulting address is the offset of mem_cgroup->kmemcg_id, dereferenced on a NULL memcg in memcg_list_lru_allocated(): folio_memcg_list_lru_alloc() list_lru_memcg_aware() <- true, only nokmem checked memcg = folio_memcg(folio) <- NULL memcg_list_lru_allocated(memcg, lru) memcg->kmemcg_id <- NULL pointer dereference Check mem_cgroup_disabled() in __list_lru_init() so that all list_lrus fall back to plain per-node lists when the controller is disabled, matching what the shrinker side already does (shrinker_memcg_alloc() bails out on mem_cgroup_disabled()). This makes the mem_cgroup_disabled() check in callers unnecessary rather than mandatory. Link: https://lore.kernel.org/20260902093202.609559-1-qinyuntan@linux.alibaba.com Signed-off-by: Qinyun Tan Reviewed-by: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Roman Gushchin Cc: Shakeel Butt Signed-off-by: Andrew Morton --- mm/list_lru.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/list_lru.c b/mm/list_lru.c index 36662d02ff9631..a4522ca93ebcb9 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -671,7 +671,7 @@ int __list_lru_init(struct list_lru *lru, bool memcg_aware, struct shrinker *shr else lru->shrinker_id = -1; - if (mem_cgroup_kmem_disabled()) + if (mem_cgroup_disabled() || mem_cgroup_kmem_disabled()) memcg_aware = false; #endif From f96f4776b0bae06088a820149a9062f02524ca33 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Fri, 28 Aug 2026 03:37:31 +0000 Subject: [PATCH 599/857] selftests/cgroup: test_zswap: wait for cgroup to unpopulate in test_zswap_writeback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "selftests/cgroup: fixes for test_zswap on single core VM", v4. This series fixes two test failures in test_zswap observed when running on a single-core VM (-smp 1) with 4GB of RAM. Patch 1 addresses a race condition in test_zswap_writeback() where waitpid() returns before the exiting child process is switched away by the kernel, causing an immediate write of "+memory" to cgroup.subtree_control to fail with -EBUSY. We fix this by waiting for cgroup.events to report "populated 0". Patch 2 fixes an implicit unsigned conversion bug in test_no_kmem_bypass() where small negative timing differences between debugfs stored_pages and cgroup zswapped bytes caused the comparison to falsely fail due to unsigned promotion. This patch (of 2): When running test_zswap on a single-core VM (-smp 1) with 4GB of RAM, test_zswap_writeback intermittently fails on the initial run after boot. In test_zswap_writeback(), after waitpid() reaps the child process created by test_zswap_writeback_one(), writing "+memory" to cgroup.subtree_control can fail with -EBUSY. Under cgroup v2, enabling domain subtree controllers is forbidden while any tasks remain in cgroup.procs. When a child process exits, exit_notify() wakes the parent process, allowing waitpid() to return immediately. However, the cgroup populated task count (nr_populated_csets) is only decremented when the exiting task is switched away via finish_task_switch() -> cgroup_task_dead(). On single-core systems, the parent runs before the dead child has been switched out, causing "+memory" to fail with -EBUSY if written immediately after waitpid() returns. Fix this by waiting for cgroup.events to report "populated 0\n" via cg_read_strcmp_wait() before enabling subtree control. Link: https://lore.kernel.org/20260828033741.2184560-1-wfelipe@google.com Link: https://lore.kernel.org/20260828033741.2184560-2-wfelipe@google.com Signed-off-by: Wilson Felipe Pereira Acked-by: Michal Koutný Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham Cc: Shuah Khan Cc: Tejun Heo Signed-off-by: Andrew Morton --- tools/testing/selftests/cgroup/test_zswap.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c index 609c48f3852410..8f2c9aa4776c02 100644 --- a/tools/testing/selftests/cgroup/test_zswap.c +++ b/tools/testing/selftests/cgroup/test_zswap.c @@ -408,6 +408,8 @@ static int test_zswap_writeback(const char *root, bool wb) * Thus, the parent's setting shall be what's in effect. */ if (cg_write(test_group, "memory.zswap.max", "max")) goto out; + if (cg_read_strcmp_wait(test_group, "cgroup.events", "populated 0\n")) + goto out; if (cg_write(test_group, "cgroup.subtree_control", "+memory")) goto out; From 83640ca642c0c8178e983a2255b4dc8018aa83b3 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Fri, 28 Aug 2026 03:37:32 +0000 Subject: [PATCH 600/857] selftests/cgroup: test_zswap: fix implicit unsigned promotion bug in test_no_kmem_bypass MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit In test_no_kmem_bypass(), delta (stored_pages * page_size - zswapped) is checked against stored_pages * page_size / 4 to verify that the pages pushed to zswap belong to the test memory cgroup. Due to slight stat update timing differences, delta can evaluate to a small negative number (e.g. -5MB out of 1GB). Because delta is declared as a signed int and stored_pages is an unsigned size_t, C's usual arithmetic conversions implicitly promote a negative delta to a large unsigned 64-bit integer, causing `delta < stored_pages * page_size / 4` to falsely evaluate to 0 and fail the test. Fix this by declaring zswapped and delta as signed long long and comparing against a signed threshold, ensuring negative deltas correctly evaluate to true. Link: https://lore.kernel.org/20260828033741.2184560-3-wfelipe@google.com Fixes: a549f9f31561a ("selftests: cgroup: add test_zswap with no kmem bypass test") Signed-off-by: Wilson Felipe Pereira Acked-by: Michal Koutný Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham Cc: Shuah Khan Cc: Tejun Heo Signed-off-by: Andrew Morton --- tools/testing/selftests/cgroup/test_zswap.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c index 8f2c9aa4776c02..9c5bd503c3f73a 100644 --- a/tools/testing/selftests/cgroup/test_zswap.c +++ b/tools/testing/selftests/cgroup/test_zswap.c @@ -630,11 +630,14 @@ static int test_no_kmem_bypass(const char *root) break; /* If memory was pushed to zswap, verify it belongs to memcg */ if (stored_pages > stored_pages_threshold) { - int zswapped = cg_read_key_long(test_group, "memory.stat", "zswapped "); - int delta = stored_pages * page_size - zswapped; - int result_ok = delta < stored_pages * page_size / 4; - - ret = result_ok ? KSFT_PASS : KSFT_FAIL; + long zswapped = cg_read_key_long( + test_group, "memory.stat", "zswapped "); + long long delta = + (long long)stored_pages * page_size - zswapped; + long long max_delta = + (long long)stored_pages * page_size / 4; + + ret = (delta < max_delta) ? KSFT_PASS : KSFT_FAIL; break; } } From b7990ab07a85af5c1900ec30d6bc1eb0ac2579c7 Mon Sep 17 00:00:00 2001 From: Rik van Riel Date: Fri, 28 Aug 2026 13:50:36 -0400 Subject: [PATCH 601/857] mm/memcontrol: fix stuck FLUSHING_CACHED_CHARGE bit on isolated cpus When drain_all_stock() sets FLUSHING_CACHED_CHARGE before checking isolation, schedule_drain_work() can drop the work in a separate RCU critical section, and housekeeping_update()'s synchronize_rcu() can race that second check, leaving the flag set. drain_local_stock() only clears the bit for work that ran, so the flag remains set and the stock is never drained again. Have schedule_drain_work() return whether the work was queued, and clear FLUSHING_CACHED_CHARGE in drain_all_stock() when the remote CPU is isolated, so future drains can retry. Link: https://lore.kernel.org/20260828135036.7d44361f@fangorn Fixes: 6a792697a53a ("memcg: do not drain charge pcp caches on remote isolated cpus") Suggested-by: Michal Hocko Suggested-by: Shakeel Butt Signed-off-by: Rik van Riel Acked-by: Shakeel Butt Acked-by: Michal Hocko Cc: Johannes Weiner Cc: Muchun Song Cc: Roman Gushchin Cc: Signed-off-by: Andrew Morton --- mm/memcontrol.c | 19 ++++++++++++------- 1 file changed, 12 insertions(+), 7 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1709ac96bbdec5..c6b85e3a5a0b3e 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2306,7 +2306,7 @@ static bool is_memcg_drain_needed(struct memcg_stock_pcp *stock, return flush; } -static void schedule_drain_work(int cpu, struct work_struct *work) +static bool schedule_drain_work(int cpu, struct work_struct *work) { /* * Protect housekeeping cpumask read and work enqueue together @@ -2315,8 +2315,11 @@ static void schedule_drain_work(int cpu, struct work_struct *work) * pending work on newly isolated CPUs. */ guard(rcu)(); - if (!cpu_is_isolated(cpu)) - queue_work_on(cpu, memcg_wq, work); + if (cpu_is_isolated(cpu)) + return false; + + queue_work_on(cpu, memcg_wq, work); + return true; } /* @@ -2348,8 +2351,9 @@ void drain_all_stock(struct mem_cgroup *root_memcg) &memcg_st->flags)) { if (cpu == curcpu) drain_local_memcg_stock(&memcg_st->work); - else - schedule_drain_work(cpu, &memcg_st->work); + else if (!schedule_drain_work(cpu, &memcg_st->work)) + clear_bit(FLUSHING_CACHED_CHARGE, + &memcg_st->flags); } if (!test_bit(FLUSHING_CACHED_CHARGE, &obj_st->flags) && @@ -2358,8 +2362,9 @@ void drain_all_stock(struct mem_cgroup *root_memcg) &obj_st->flags)) { if (cpu == curcpu) drain_local_obj_stock(&obj_st->work); - else - schedule_drain_work(cpu, &obj_st->work); + else if (!schedule_drain_work(cpu, &obj_st->work)) + clear_bit(FLUSHING_CACHED_CHARGE, + &obj_st->flags); } } migrate_enable(); From 1d46a35abab7cb100b08eb8ca31307a3f45838d1 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 28 Aug 2026 12:24:19 -0700 Subject: [PATCH 602/857] memcg: clear FLUSHING_CACHED_CHARGE on cpu offline Sashiko [1] reported that memcg_hotplug_cpu_dead() drains the stocks of the CPU which went away but leaves FLUSHING_CACHED_CHARGE alone. The flag can be set at that point: drain_all_stock() may have claimed the stock and queued the drain work shortly before the CPU went down. workqueue_offline_cpu() unbinds the per-cpu workers, so such a pending work item is executed by an unbound worker on some other CPU, where drain_local_memcg_stock() operates on this_cpu_ptr() and thus drains and clears the flag of that other CPU instead. Nothing clears the flag of the dead CPU, so drain_all_stock() would skip its stock forever once the CPU comes back online. Clear the flag of both stocks after draining them. Link: https://lore.kernel.org/20260828192419.3057939-1-shakeel.butt@linux.dev Link: https://sashiko.dev/#/patchset/20260828135036.7d44361f%40fangorn [1] Signed-off-by: Shakeel Butt Reviewed-by: Rik van Riel Reported-by: Sashiko Acked-by: Michal Hocko Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Signed-off-by: Andrew Morton --- mm/memcontrol.c | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index c6b85e3a5a0b3e..05f5338765a995 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2373,9 +2373,21 @@ void drain_all_stock(struct mem_cgroup *root_memcg) static int memcg_hotplug_cpu_dead(unsigned int cpu) { + struct memcg_stock_pcp *memcg_st = &per_cpu(memcg_stock, cpu); + struct obj_stock_pcp *obj_st = &per_cpu(obj_stock, cpu); + /* no need for the local lock */ - drain_obj_stock(&per_cpu(obj_stock, cpu)); - drain_stock_fully(&per_cpu(memcg_stock, cpu)); + drain_obj_stock(obj_st); + drain_stock_fully(memcg_st); + + /* + * A drain work queued before the CPU went away is executed by an + * unbound worker on some other CPU and clears that CPU's flag, so + * clear the flags here to make these stocks drainable again once + * the CPU comes back online. + */ + clear_bit(FLUSHING_CACHED_CHARGE, &memcg_st->flags); + clear_bit(FLUSHING_CACHED_CHARGE, &obj_st->flags); return 0; } From 15d9e2dd69c667fe037b6e3e157e4534691991db Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Fri, 28 Aug 2026 15:31:11 -0400 Subject: [PATCH 603/857] mm/mempolicy: take a cpuset cookie for the interleave node count alloc_pages_bulk_interleave() counts pol->nodes without a cpuset cookie: nodes = nodes_weight(pol->nodes); nr_pages_per_node = nr_pages / nodes; nodemask_t spans several words once MAX_NUMNODES exceeds BITS_PER_LONG, so a concurrent cpuset rebind can tear that read and yield an empty mask even though neither version of it was empty. The call then allocates nothing and returns 0. Some compilers will hoist the loop entry test above the division, because nr_pages_per_node is dead when the loop does not run. 682e: call ... <- nodes_weight() 6838: test %eax,%eax 683a: jle 692d <- nodes <= 0 skips the loop 684a: div %rcx So in most deployments, this div/0 is unreachable - but nothing in the source guarantees that, it's just not easily exercised. Take the cookie around the count and bail if the mask really is empty. Only the count needs it, interleave_nodes() takes the cookie itself so so a torn read there is already retried. A rebind landing mid-loop can still leave the count disagreeing with the mask, so the loop may revisit a node or skip one - but a rebind where nodes change causes migration, so a handful of misplaced pages isn't catastrophic in any sense. Measured on a 72 node VM (NODES_SHIFT=10) with a cgroup v2 cpuset flipping cpuset.mems between a word 0 and a word 1 node set, and the two word read artificially widened: 330 zero counts in 130414 calls without the cookie, and 401 retries with it. Link: https://lore.kernel.org/20260828193111.1023497-1-gourry@gourry.net Fixes: c00b6b961099 ("mm/vmalloc: introduce alloc_pages_bulk_array_mempolicy to accelerate memory allocation") Signed-off-by: Gregory Price (Meta) Reported-by: Chelsy Ratnawat Link: https://lore.kernel.org/all/20250907160829.91628-1-chelsyratnawat2001@gmail.com/ Assisted-by: Claude:claude-opus-5 Reviewed-by: Huang Ying Cc: Alistair Popple Cc: Byungchul Park Cc: Chenwandun Cc: David Hildenbrand Cc: Joshua Hahn Cc: Matthew Brost Cc: Rakie Kim Cc: "Uladzislau Rezki (Sony)" Cc: Zi Yan Cc: Signed-off-by: Andrew Morton --- mm/mempolicy.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 79053ece02cd48..060a0eb2691709 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2592,6 +2592,7 @@ static unsigned long alloc_pages_bulk_interleave(gfp_t gfp, struct mempolicy *pol, unsigned long nr_pages, struct page **page_array) { + unsigned int cpuset_mems_cookie; int nodes; unsigned long nr_pages_per_node; int delta; @@ -2599,7 +2600,16 @@ static unsigned long alloc_pages_bulk_interleave(gfp_t gfp, unsigned long nr_allocated; unsigned long total_allocated = 0; - nodes = nodes_weight(pol->nodes); + /* count the nodes, retry if a rebind happened during the read */ + do { + cpuset_mems_cookie = read_mems_allowed_begin(); + nodes = nodes_weight(pol->nodes); + } while (read_mems_allowed_retry(cpuset_mems_cookie)); + + /* if the nodemask has become invalid, we cannot do anything */ + if (!nodes) + return 0; + nr_pages_per_node = nr_pages / nodes; delta = nr_pages - nodes * nr_pages_per_node; From 638e43e66e5f27094269c21c86faf0fd5209d14f Mon Sep 17 00:00:00 2001 From: James Houghton Date: Fri, 28 Aug 2026 22:26:40 +0000 Subject: [PATCH 604/857] mm/khugepaged: don't install PMDs in uffd-minor-registered VMAs Userfaultfd minor faults provides userspace with the ability to manually install PTEs with UFFDIO_CONTINUE. Right now, khugepaged collapse can map holes in the VMA when a naturally-aligned THP is present without explicit action from userspace. This is a problem, as it bypasses userfaultfd minor faults that userspace is expecting to handle. If userspace implements post-copy live migration using userfaultfd minor faults, this situation is currently possible: 1. The VMA for guest memory is userfaultfd-minor-registered and nothing is mapped in the page tables. 2. A stale copy of a page is present in a naturally-aligned THP (from pre-copy live migration). 3. khugepaged collapses the mapping of the THP, installs a PMD. 4. The VM now has access to the stale contents => VM is broken. 5. After installing the correct contents, userspace attempts to map the page with UFFDIO_CONTINUE; it gets EEXIST, indicating that something unexpectedly mapped the page. The naturally-aligned THP case is the only case where this is a problem. khugepaged otherwise requires all PTEs to be present for userfaultfd-registered VMAs (i.e., max none PTEs is 0), which is correct. This check is essentially bypassed for naturally-aligned THPs. No changes are needed for file_backed_vma_is_retractable(), as zapping PTEs is safe. Userspace must already handle cases where PTEs are zapped without explicit action (e.g. due to reclaim). A reproducer for this issue is at https://gist.github.com/48ca/d399bf534158e80241fb4937ef1ff664 Link: https://lore.kernel.org/20260828222640.1638457-1-jthoughton@google.com Fixes: 58ac9a8993a1 ("mm/khugepaged: attempt to map file/shmem-backed pte-mapped THPs by pmds") Signed-off-by: James Houghton Suggested-by: Lance Yang Tested-by: Lance Yang Cc: Baolin Wang Cc: Barry Song Cc: David Hildenbrand Cc: Dev Jain Cc: Hugh Dickins Cc: Kiryl Shutsemau Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ryan Roberts Cc: Yang Shi Cc: Zach O'Keefe Cc: Zi Yan Cc: # 6.1 Signed-off-by: Andrew Morton --- mm/khugepaged.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 52b4476898d96f..f49a6710933b14 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1905,6 +1905,13 @@ static enum scan_result try_collapse_pte_mapped_thp(struct mm_struct *mm, unsign if (userfaultfd_protected(vma)) return SCAN_PTE_UFFD; + /* + * Userfaultfd-minor-registered VMAs should not be collapsed, as + * userspace is expecting to explicitly install PTEs. + */ + if (userfaultfd_minor(vma)) + return SCAN_PTE_UFFD; + folio = filemap_lock_folio(vma->vm_file->f_mapping, linear_page_index(vma, haddr)); if (IS_ERR(folio)) From 9e62357d2ee2cfc3c93b3edddbea32fd9cc58467 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:16:09 +0800 Subject: [PATCH 605/857] tools/mm/page_owner_sort: fix --sort option being silently ignored Patch series "tools/mm/page_owner_sort: fix --sort, add module filter, improve usage", v3. This series improves the page_owner_sort tool with a bug fix, a new module-name feature, and better usage text. Patch 1 fixes a long-standing bug where --sort was silently ignored when used without a short option (-a, -m, -p, etc.). The COMP_NO_FLAG case fell through to COMP_NUM and overwrote the sort conditions configured by parse_sort_args(). Patch 2 adds kernel module name support for sort, cull, and filter operations. Page owner stack traces already contain module names in the "function+0xNN/0xNN [module]" format produced by %pS, but page_owner_sort had no way to use them. Records without module frames are assigned "vmlinux". # Aggregate page usage per module ./page_owner_sort input.txt output.txt --cull=mod # Filter to records from xfs module only ./page_owner_sort input.txt output.txt --module xfs # Sort by module name, then by pid descending ./page_owner_sort input.txt output.txt --sort=mod,-pid Patch 3 lists all available sort keys with abbreviations and examples directly in the --sort help section so users no longer need to read the source to discover valid keys. This patch (of 3): When --sort is used without any short option (-a, -m, -p, etc.), compare_flag remains COMP_NO_FLAG. The switch (compare_flag) then falls through to the COMP_NUM case and calls set_single_cmp(), which unconditionally overwrites the sort conditions that parse_sort_args() already configured. This makes --sort silently ineffective unless a short option is also supplied. Split COMP_NO_FLAG out of the COMP_NUM fallthrough so that --sort is respected when no short option is present. Reproduction: # Before fix: ascending order (ignored --sort=-pid) ./page_owner_sort --sort=-pid input.txt output.txt # After fix: descending order as expected Link: https://lore.kernel.org/20260819021611.2910835-1-ye.liu@linux.dev Link: https://lore.kernel.org/20260819021611.2910835-2-ye.liu@linux.dev Signed-off-by: Ye Liu Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Liu Jing Cc: Yichong Chen Signed-off-by: Andrew Morton --- tools/mm/page_owner_sort.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index 22b3b500d33a47..f1dc0763c25cef 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -821,6 +821,9 @@ int main(int argc, char **argv) set_single_cmp(compare_stacktrace, SORT_ASC); break; case COMP_NO_FLAG: + if (sc.size > 0) + break; + /* fallthrough */ case COMP_NUM: set_single_cmp(compare_num, SORT_DESC); break; From 3719d3cc17368b2224e1745303662c04378d1a7c Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:16:10 +0800 Subject: [PATCH 606/857] tools/mm/page_owner_sort: add module name sort/cull/filter support Page owner stack traces already contain kernel module names in the "[module]" format produced by %pS, but page_owner_sort has no way to sort, cull, or filter by module. Extract the first module name from each record's stack trace using the regex \[([a-zA-Z0-9_]+)\]. Records whose stack traces contain no module frames are assigned "vmlinux". New options: -M Sort by module name --sort=mod Sort by module name (supports +/- prefix) --cull=mod Cull (aggregate) by module name --module Filter to records matching the given module(s) The module field is also printed in cull output when relevant. Link: https://lore.kernel.org/20260819021611.2910835-3-ye.liu@linux.dev Signed-off-by: Ye Liu Cc: David Hildenbrand (Arm) Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Liu Jing Cc: Yichong Chen Signed-off-by: Andrew Morton --- Documentation/mm/page_owner.rst | 8 ++- tools/mm/page_owner_sort.c | 122 +++++++++++++++++++++++++++----- 2 files changed, 110 insertions(+), 20 deletions(-) diff --git a/Documentation/mm/page_owner.rst b/Documentation/mm/page_owner.rst index a6bd3fe6423ad6..bd027377dff6ee 100644 --- a/Documentation/mm/page_owner.rst +++ b/Documentation/mm/page_owner.rst @@ -199,6 +199,7 @@ Usage -p Sort by pid. -P Sort by tgid. -n Sort by task command name. + -M Sort by module name. -r Sort by memory release time. -s Sort by stack trace. -t Sort by times (default). @@ -240,8 +241,10 @@ Usage group ID numbers appear in . --name Select by task command name. This selects the blocks whose task command name appear in . + --module Select by module name. This selects the blocks whose + module name appear in . - , , are single arguments in the form of a comma-separated list, + , , , are single arguments in the form of a comma-separated list, which offers a way to specify individual selecting rules. @@ -249,6 +252,7 @@ Usage ./page_owner_sort --pid=1 ./page_owner_sort --tgid=1,2,3 ./page_owner_sort --name name1,name2 + ./page_owner_sort --module xfs,ext4 STANDARD FORMAT SPECIFIERS ========================== @@ -265,6 +269,7 @@ STANDARD FORMAT SPECIFIERS ft free_ts timestamp of the page when it was released at alloc_ts timestamp of the page when it was allocated ator allocator memory allocator for pages + mod module kernel module name For --cull option: @@ -275,6 +280,7 @@ STANDARD FORMAT SPECIFIERS f free whether the page has been released or not st stacktrace stack trace of the page allocation ator allocator memory allocator for pages + mod module kernel module name Filtering page_owner output ============================ diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index f1dc0763c25cef..a5b61b4abc2f08 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -25,11 +25,13 @@ #include #define TASK_COMM_LEN 16 +#define MODULE_NAME_LEN 64 struct block_list { char *txt; char *comm; // task command name char *stacktrace; + char *module; // kernel module name __u64 ts_nsec; int len; int num; @@ -41,7 +43,8 @@ struct block_list { enum FILTER_BIT { FILTER_PID = 1<<1, FILTER_TGID = 1<<2, - FILTER_COMM = 1<<3 + FILTER_COMM = 1<<3, + FILTER_MODULE = 1<<4 }; enum FILTER_RESULT { @@ -55,7 +58,8 @@ enum CULL_BIT { CULL_TGID = 1<<2, CULL_COMM = 1<<3, CULL_STACKTRACE = 1<<4, - CULL_ALLOCATOR = 1<<5 + CULL_ALLOCATOR = 1<<5, + CULL_MODULE = 1<<6 }; enum ALLOCATOR_BIT { ALLOCATOR_CMA = 1<<1, @@ -65,7 +69,8 @@ enum ALLOCATOR_BIT { }; enum ARG_TYPE { ARG_TXT, ARG_COMM, ARG_STACKTRACE, ARG_ALLOC_TS, ARG_CULL_TIME, - ARG_PAGE_NUM, ARG_PID, ARG_TGID, ARG_UNKNOWN, ARG_ALLOCATOR + ARG_PAGE_NUM, ARG_PID, ARG_TGID, ARG_UNKNOWN, ARG_ALLOCATOR, + ARG_MODULE }; enum SORT_ORDER { SORT_ASC = 1, @@ -79,15 +84,18 @@ enum COMP_FLAG { COMP_STACK = 1<<3, COMP_NUM = 1<<4, COMP_TGID = 1<<5, - COMP_COMM = 1<<6 + COMP_COMM = 1<<6, + COMP_MODULE = 1<<7 }; struct filter_condition { pid_t *pids; pid_t *tgids; char **comms; + char **modules; int pids_size; int tgids_size; int comms_size; + int modules_size; }; struct sort_condition { int (**cmps)(const void *, const void *); @@ -101,6 +109,7 @@ static regex_t pid_pattern; static regex_t tgid_pattern; static regex_t comm_pattern; static regex_t ts_nsec_pattern; +static regex_t module_pattern; static struct block_list *list; static int list_size; static int max_size; @@ -184,6 +193,13 @@ static int compare_comm(const void *p1, const void *p2) return strcmp(l1->comm, l2->comm); } +static int compare_module(const void *p1, const void *p2) +{ + const struct block_list *l1 = p1, *l2 = p2; + + return strcmp(l1->module, l2->module); +} + static int compare_ts(const void *p1, const void *p2) { const struct block_list *l1 = p1, *l2 = p2; @@ -207,6 +223,8 @@ static int compare_cull_condition(const void *p1, const void *p2) return compare_tgid(p1, p2); if ((cull & CULL_COMM) && compare_comm(p1, p2)) return compare_comm(p1, p2); + if ((cull & CULL_MODULE) && compare_module(p1, p2)) + return compare_module(p1, p2); if ((cull & CULL_ALLOCATOR) && compare_allocator(p1, p2)) return compare_allocator(p1, p2); return 0; @@ -411,9 +429,33 @@ static char *get_comm(char *buf) return comm_str; } +static char *get_module(char *buf) +{ + char *module_str = malloc(MODULE_NAME_LEN); + regmatch_t pmatch[2]; + int val_len; + + if (!module_str) + return NULL; + memset(module_str, 0, MODULE_NAME_LEN); + if (regexec(&module_pattern, buf, 2, pmatch, REG_NOTBOL) != 0 || pmatch[1].rm_so == -1) { + strcpy(module_str, "vmlinux"); + return module_str; + } + + val_len = pmatch[1].rm_eo - pmatch[1].rm_so; + if ((size_t)val_len >= MODULE_NAME_LEN) + val_len = MODULE_NAME_LEN - 1; + memcpy(module_str, buf + pmatch[1].rm_so, val_len); + module_str[val_len] = '\0'; + + return module_str; +} + static void free_block_list(struct block_list *block) { free(block->comm); + free(block->module); free(block->txt); } @@ -433,6 +475,8 @@ static int get_arg_type(const char *arg) return ARG_ALLOC_TS; else if (!strcmp(arg, "allocator") || !strcmp(arg, "ator")) return ARG_ALLOCATOR; + else if (!strcmp(arg, "module") || !strcmp(arg, "mod")) + return ARG_MODULE; else { return ARG_UNKNOWN; } @@ -483,25 +527,36 @@ static bool match_str_list(const char *str, char **list, int list_size) static enum FILTER_RESULT filter_record(char *buf) { - char *comm; + char *comm, *module; if ((filter & FILTER_PID) && !match_num_list(get_pid(buf), fc.pids, fc.pids_size)) return FILTER_SKIP; if ((filter & FILTER_TGID) && !match_num_list(get_tgid(buf), fc.tgids, fc.tgids_size)) return FILTER_SKIP; - if (!(filter & FILTER_COMM)) + if (!(filter & (FILTER_COMM | FILTER_MODULE))) return FILTER_MATCH; - comm = get_comm(buf); - if (!comm) - return FILTER_ERROR; - - if (!match_str_list(comm, fc.comms, fc.comms_size)) { + if (filter & FILTER_COMM) { + comm = get_comm(buf); + if (!comm) + return FILTER_ERROR; + if (!match_str_list(comm, fc.comms, fc.comms_size)) { + free(comm); + return FILTER_SKIP; + } free(comm); - return FILTER_SKIP; } - free(comm); + if (filter & FILTER_MODULE) { + module = get_module(buf); + if (!module) + return FILTER_ERROR; + if (!match_str_list(module, fc.modules, fc.modules_size)) { + free(module); + return FILTER_SKIP; + } + free(module); + } return FILTER_MATCH; } @@ -547,6 +602,12 @@ static bool add_list(char *buf, int len, char *ext_buf) list[list_size].stacktrace++; list[list_size].ts_nsec = get_ts_nsec(buf); list[list_size].allocator = get_allocator(buf, ext_buf); + list[list_size].module = get_module(buf); + if (!list[list_size].module) { + fprintf(stderr, "Out of memory\n"); + free_block_list(&list[list_size]); + return false; + } list_size++; if (list_size % 1000 == 0) { printf("loaded %d\r", list_size); @@ -573,6 +634,8 @@ static bool parse_cull_args(const char *arg_str) cull |= CULL_STACKTRACE; else if (arg_type == ARG_ALLOCATOR) cull |= CULL_ALLOCATOR; + else if (arg_type == ARG_MODULE) + cull |= CULL_MODULE; else { free_explode(args, size); return false; @@ -635,6 +698,8 @@ static bool parse_sort_args(const char *arg_str) sc.cmps[i] = compare_txt; else if (arg_type == ARG_ALLOCATOR) sc.cmps[i] = compare_allocator; + else if (arg_type == ARG_MODULE) + sc.cmps[i] = compare_module; else { free_explode(args, size); sc.size = 0; @@ -691,7 +756,8 @@ static void usage(void) "-p\t\t\tSort by pid.\n" "-P\t\t\tSort by tgid.\n" "-s\t\t\tSort by the stacktrace.\n" - "-t\t\t\tSort by number of times record is seen (default).\n\n" + "-t\t\t\tSort by number of times record is seen (default).\n" + "-M\t\t\tSort by module name.\n\n" "--pid \t\tSelect by pid. This selects the information" " of\n\t\t\tblocks whose process ID numbers appear in .\n" "--tgid \tSelect by tgid. This selects the information" @@ -700,10 +766,11 @@ static void usage(void) "--name \tSelect by command name. This selects the" " information\n\t\t\tof blocks whose command name appears in" " .\n" - "--cull \t\tCull by user-defined rules. is a " - "single\n\t\t\targument in the form of a comma-separated list " - "with some\n\t\t\tcommon fields predefined (pid, tgid, comm, " - "stacktrace, allocator)\n" + "--module \tSelect by module name. This selects the information\n" + "\t\t\tof blocks whose module name appears in .\n" + "--cull \t\tCull by user-defined rules. is a single\n" + "\t\t\targument in the form of a comma-separated list with some\n" + "\t\t\tcommon fields predefined (pid, tgid, comm, stacktrace, allocator, module)\n" "--sort \t\tSpecify sort order as: [+|-]key[,[+|-]key[,...]]\n" ); } @@ -721,13 +788,14 @@ int main(int argc, char **argv) { "name", required_argument, NULL, 3 }, { "cull", required_argument, NULL, 4 }, { "sort", required_argument, NULL, 5 }, + { "module", required_argument, NULL, 6 }, { "help", no_argument, NULL, 'h' }, { 0, 0, 0, 0}, }; compare_flag = COMP_NO_FLAG; - while ((opt = getopt_long(argc, argv, "admnpstPh", longopts, NULL)) != -1) + while ((opt = getopt_long(argc, argv, "admnpstPMh", longopts, NULL)) != -1) switch (opt) { case 'a': compare_flag |= COMP_ALLOC; @@ -753,6 +821,9 @@ int main(int argc, char **argv) case 'n': compare_flag |= COMP_COMM; break; + case 'M': + compare_flag |= COMP_MODULE; + break; case 'h': usage(); exit(0); @@ -792,6 +863,10 @@ int main(int argc, char **argv) exit(1); } break; + case 6: + filter = filter | FILTER_MODULE; + fc.modules = explode(',', optarg, &fc.modules_size); + break; default: usage(); exit(1); @@ -833,6 +908,9 @@ int main(int argc, char **argv) case COMP_COMM: set_single_cmp(compare_comm, SORT_ASC); break; + case COMP_MODULE: + set_single_cmp(compare_module, SORT_ASC); + break; default: usage(); exit(1); @@ -855,6 +933,8 @@ int main(int argc, char **argv) goto out_comm; if (!check_regcomp(&ts_nsec_pattern, "ts\\s*([0-9]*)\\s*ns")) goto out_ts; + if (!check_regcomp(&module_pattern, "\\+0x[0-9a-f]+/0x[0-9a-f]+\\s*\\[([a-zA-Z0-9_-]+)\\]")) + goto out_module; fstat(fileno(fin), &st); max_size = st.st_size / 100; /* hack ... */ @@ -920,6 +1000,8 @@ int main(int argc, char **argv) fprintf(fout, ", TGID %d", list[i].tgid); if (cull & CULL_COMM || filter & FILTER_COMM) fprintf(fout, ", task_comm_name: %s", list[i].comm); + if (cull & CULL_MODULE || filter & FILTER_MODULE) + fprintf(fout, ", module: %s", list[i].module); if (cull & CULL_ALLOCATOR) { fprintf(fout, ", "); print_allocator(fout, list[i].allocator); @@ -940,6 +1022,8 @@ int main(int argc, char **argv) free_block_list(&list[i]); free(list); } +out_module: + regfree(&module_pattern); out_ts: regfree(&ts_nsec_pattern); out_comm: From 3bc3b7ef6b67507745e0ada57056e06baea6ea55 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:16:11 +0800 Subject: [PATCH 607/857] tools/mm/page_owner_sort: show available sort keys in usage text The --sort option accepts abbreviated or complete key names, but the usage text never listed them. Users had to read the source or the documentation to discover valid keys. List all available keys (full form and abbreviation) with a brief description and examples directly in the --sort help section. Link: https://lore.kernel.org/20260819021611.2910835-4-ye.liu@linux.dev Signed-off-by: Ye Liu Acked-by: David Hildenbrand (Arm) Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Yichong Chen Cc: Liu Jing Signed-off-by: Andrew Morton --- tools/mm/page_owner_sort.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index a5b61b4abc2f08..6c2ac5d9dc4b6e 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -772,6 +772,15 @@ static void usage(void) "\t\t\targument in the form of a comma-separated list with some\n" "\t\t\tcommon fields predefined (pid, tgid, comm, stacktrace, allocator, module)\n" "--sort \t\tSpecify sort order as: [+|-]key[,[+|-]key[,...]]\n" + "\t\t\tAvailable keys:\n" + "\t\t\t pid(p), tgid(tg), name(n), stacktrace(st),\n" + "\t\t\t txt(T), alloc_ts(at), allocator(ator), module(mod)\n" + "\t\t\tThe \"+\" is optional since default direction is\n" + "\t\t\tincreasing numerical or lexicographic order.\n" + "\t\t\tMixed use of abbreviated and complete-form is allowed.\n" + "\t\t\tExamples:\n" + "\t\t\t --sort=n,+pid,-tgid\n" + "\t\t\t --sort=mod,at\n" ); } From 14811b1d9ba0a1658adf136a1173a150d82c74d6 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 19 Aug 2026 18:20:10 -0700 Subject: [PATCH 608/857] memcg: trim the per-cpu charge stock instead of draining it Joy reported that an application generating a request/response traffic pattern spends 44.6% to 57.0% of CPU in the memcg charge/uncharge path for a range of message sizes, against 0.27% to 0.71% outside that range. Running from the root memcg, where socket memory accounting is skipped, recovers the performance. Tracing the charge path showed that the application generates a pattern where the write syscall charges one page and the read syscall uncharges two pages on the same CPU. This hits a corner case in the memcg percpu stock code that thrashes the stock continuously. In the memcg percpu stock code, MEMCG_CHARGE_BATCH (64) is both the high watermark and the emptying target, i.e. on a request to charge one page the kernel charges MEMCG_CHARGE_BATCH pages and caches (MEMCG_CHARGE_BATCH - 1) of them in the percpu stock. The following uncharge of 2 pages takes the cached count to (MEMCG_CHARGE_BATCH + 1), and refill_stock() then empties the cache completely. With such a pattern the percpu stock becomes completely ineffective. Instead of a single boundary point for charges, use the technique the page allocator uses for its own percpu caches, which keeps the watermark and the emptying target apart: nr_pcp_free() frees between batch and high - batch pages, leaving at least pcp->batch on the list. Add a high watermark MEMCG_STOCK_HIGH and, once the cached count goes over it, return only the pages above MEMCG_STOCK_LOW. The watermarks are MEMCG_CHARGE_BATCH apart, so a page_counter update still covers a full batch. For now, keep MEMCG_STOCK_HIGH same as MEMCG_CHARGE_BATCH and in future we will reevaluate if it makes sense to increase it. Link: https://lore.kernel.org/20260820012010.2016086-1-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Reported-by: Joy Chaoyue Xiong Acked-by: Michal Hocko Cc: Jakub Kacinski Cc: Johannes Weiner Cc: Joshua Hahn Cc: Muchun Song Cc: Roman Gushchin Signed-off-by: Andrew Morton --- mm/memcontrol.c | 25 +++++++++++++++++++------ 1 file changed, 19 insertions(+), 6 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 05f5338765a995..bfd0a74fac9239 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2048,6 +2048,15 @@ void mem_cgroup_print_oom_group(struct mem_cgroup *memcg) * nr_pages in a single cacheline. This may change in future. */ #define NR_MEMCG_STOCK 7 + +/* + * Watermarks for a charge stock slot, in the spirit of pcp->high and + * pcp->batch: MEMCG_STOCK_HIGH is the high watermark at which a slot is + * trimmed, and it is trimmed down to MEMCG_STOCK_LOW rather than emptied. + */ +#define MEMCG_STOCK_LOW (MEMCG_CHARGE_BATCH / 2) +#define MEMCG_STOCK_HIGH (MEMCG_CHARGE_BATCH) + #define FLUSHING_CACHED_CHARGE 0 struct memcg_stock_pcp { local_trylock_t lock; @@ -2228,17 +2237,18 @@ static void refill_stock(struct mem_cgroup *memcg, unsigned int nr_pages) { struct memcg_stock_pcp *stock; struct mem_cgroup *cached; - uint8_t stock_pages; + unsigned int stock_pages; bool success = false; int empty_slot = -1; int i; /* - * For now limit MEMCG_CHARGE_BATCH to 127 and less. In future if we - * decide to increase it more than 127 then we will need more careful - * handling of nr_pages[] in struct memcg_stock_pcp. + * nr_pages[] is a uint8_t and a slot's count is capped at + * MEMCG_STOCK_HIGH. Raising MEMCG_CHARGE_BATCH beyond 127 would need + * more careful handling of nr_pages[] in struct memcg_stock_pcp. */ BUILD_BUG_ON(MEMCG_CHARGE_BATCH > S8_MAX); + BUILD_BUG_ON(MEMCG_STOCK_HIGH > U8_MAX); VM_WARN_ON_ONCE(mem_cgroup_is_root(memcg)); @@ -2259,9 +2269,12 @@ static void refill_stock(struct mem_cgroup *memcg, unsigned int nr_pages) empty_slot = i; if (memcg == READ_ONCE(stock->cached[i])) { stock_pages = READ_ONCE(stock->nr_pages[i]) + nr_pages; + if (stock_pages > MEMCG_STOCK_HIGH) { + memcg_uncharge(memcg, + stock_pages - MEMCG_STOCK_LOW); + stock_pages = MEMCG_STOCK_LOW; + } WRITE_ONCE(stock->nr_pages[i], stock_pages); - if (stock_pages > MEMCG_CHARGE_BATCH) - drain_stock(stock, i); success = true; break; } From 43d79f7c7631a354f9fc8ef9f4cb88c873099269 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:04 -0700 Subject: [PATCH 609/857] memcg: remove v1 soft limit reclaim Patch series "memcg: remove the v1 soft limit", v2. The v1 soft limit was deprecated in v6.12 by commit 569c4f62d84a ("memcg: initiate deprecation of v1 soft limit") and nobody has reported depending on it in the ~21 months since. memory.low and memory.min in v2 have covered the same ground for far longer. The knob has since been made inert by "memcg: make the v1 soft limit knob inert", already queued in mm-hotfixes as a backportable fix for a syzbot report [1]. Nothing can enter the soft limit rbtree anymore, so this series just deletes the machinery that is now dead: the reclaim pass in kswapd and direct reclaim, mem_cgroup_shrink_node() and its tracepoints, the per-node rbtree, lru_gen_soft_reclaim() and the MEMCG_LRU_HEAD op, the per-node tree fields, mem_cgroup->soft_limit, and finally the v1 event ratelimiting which is now down to a single target. memory.soft_limit_in_bytes itself is untouched: writes stay ignored and reads keep returning the maximum value. This patch (of 8): Nothing can put a cgroup on the soft limit rbtree anymore, so the tree is always empty and both callers of memcg1_soft_limit_reclaim() are guaranteed no-ops. Remove the reclaim pass from direct reclaim and from kswapd, along with its implementation. In shrink_zones() this leaves the global reclaim branch with a last_pgdat check that is now redundant with the identical check right below it, so drop it and move the explaining comment down to the check that remains. That check could only ever fire once last_pgdat was set, which implies first_pgdat had already been assigned, so skipping it does not change which node consider_reclaim_throttle() gets. Link: https://lore.kernel.org/20260902174311.1772372-1-shakeel.butt@linux.dev Link: https://lore.kernel.org/20260902174311.1772372-2-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier Signed-off-by: Andrew Morton --- include/linux/memcontrol.h | 12 --- mm/memcontrol-v1.c | 175 ------------------------------------- mm/vmscan.c | 39 ++------- 3 files changed, 6 insertions(+), 220 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index da625d2edb3bab..11c1fa88d6fd0c 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -1924,10 +1924,6 @@ static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) /* Cgroup v1-related declarations */ #ifdef CONFIG_MEMCG_V1 -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned); - bool mem_cgroup_oom_synchronize(bool wait); static inline bool task_in_memcg_oom(struct task_struct *p) @@ -1948,14 +1944,6 @@ static inline void mem_cgroup_exit_user_fault(void) } #else /* CONFIG_MEMCG_V1 */ -static inline -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - return 0; -} - static inline bool task_in_memcg_oom(struct task_struct *p) { return false; diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index 05ef55cae4dc61..b38b8d0f7f51c5 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -34,13 +34,6 @@ struct mem_cgroup_tree { static struct mem_cgroup_tree soft_limit_tree __read_mostly; -/* - * Maximum loops in mem_cgroup_soft_reclaim(), used for soft - * limit reclaim to prevent infinite loops, if they ever occur. - */ -#define MEM_CGROUP_MAX_RECLAIM_LOOPS 100 -#define MEM_CGROUP_MAX_SOFT_LIMIT_RECLAIM_LOOPS 2 - /* for OOM */ struct mem_cgroup_eventfd_list { struct list_head list; @@ -233,174 +226,6 @@ void memcg1_remove_from_trees(struct mem_cgroup *memcg) } } -static struct mem_cgroup_per_node * -__mem_cgroup_largest_soft_limit_node(struct mem_cgroup_tree_per_node *mctz) -{ - struct mem_cgroup_per_node *mz; - -retry: - mz = NULL; - if (!mctz->rb_rightmost) - goto done; /* Nothing to reclaim from */ - - mz = rb_entry(mctz->rb_rightmost, - struct mem_cgroup_per_node, tree_node); - /* - * Remove the node now but someone else can add it back, - * we will to add it back at the end of reclaim to its correct - * position in the tree. - */ - __mem_cgroup_remove_exceeded(mz, mctz); - if (!soft_limit_excess(mz->memcg) || - !css_tryget(&mz->memcg->css)) - goto retry; -done: - return mz; -} - -static struct mem_cgroup_per_node * -mem_cgroup_largest_soft_limit_node(struct mem_cgroup_tree_per_node *mctz) -{ - struct mem_cgroup_per_node *mz; - - spin_lock_irq(&mctz->lock); - mz = __mem_cgroup_largest_soft_limit_node(mctz); - spin_unlock_irq(&mctz->lock); - return mz; -} - -static int mem_cgroup_soft_reclaim(struct mem_cgroup *root_memcg, - pg_data_t *pgdat, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - struct mem_cgroup *victim = NULL; - int total = 0; - int loop = 0; - unsigned long excess; - unsigned long nr_scanned; - struct mem_cgroup_reclaim_cookie reclaim = { - .pgdat = pgdat, - }; - - excess = soft_limit_excess(root_memcg); - - while (1) { - victim = mem_cgroup_iter(root_memcg, victim, &reclaim); - if (!victim) { - loop++; - if (loop >= 2) { - /* - * If we have not been able to reclaim - * anything, it might because there are - * no reclaimable pages under this hierarchy - */ - if (!total) - break; - /* - * We want to do more targeted reclaim. - * excess >> 2 is not to excessive so as to - * reclaim too much, nor too less that we keep - * coming back to reclaim from this cgroup - */ - if (total >= (excess >> 2) || - (loop > MEM_CGROUP_MAX_RECLAIM_LOOPS)) - break; - } - continue; - } - total += mem_cgroup_shrink_node(victim, gfp_mask, false, - pgdat, &nr_scanned); - *total_scanned += nr_scanned; - if (!soft_limit_excess(root_memcg)) - break; - } - mem_cgroup_iter_break(root_memcg, victim); - return total; -} - -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - unsigned long nr_reclaimed = 0; - struct mem_cgroup_per_node *mz, *next_mz = NULL; - unsigned long reclaimed; - int loop = 0; - struct mem_cgroup_tree_per_node *mctz; - unsigned long excess; - - if (lru_gen_enabled()) - return 0; - - if (order > 0) - return 0; - - mctz = soft_limit_tree.rb_tree_per_node[pgdat->node_id]; - - /* - * Do not even bother to check the largest node if the root - * is empty. Do it lockless to prevent lock bouncing. Races - * are acceptable as soft limit is best effort anyway. - */ - if (!mctz || RB_EMPTY_ROOT(&mctz->rb_root)) - return 0; - - /* - * This loop can run a while, specially if mem_cgroup's continuously - * keep exceeding their soft limit and putting the system under - * pressure - */ - do { - if (next_mz) - mz = next_mz; - else - mz = mem_cgroup_largest_soft_limit_node(mctz); - if (!mz) - break; - - reclaimed = mem_cgroup_soft_reclaim(mz->memcg, pgdat, - gfp_mask, total_scanned); - nr_reclaimed += reclaimed; - spin_lock_irq(&mctz->lock); - - /* - * If we failed to reclaim anything from this memory cgroup - * it is time to move on to the next cgroup - */ - next_mz = NULL; - if (!reclaimed) - next_mz = __mem_cgroup_largest_soft_limit_node(mctz); - - excess = soft_limit_excess(mz->memcg); - /* - * One school of thought says that we should not add - * back the node to the tree if reclaim returns 0. - * But our reclaim could return 0, simply because due - * to priority we are exposing a smaller subset of - * memory to reclaim from. Consider this as a longer - * term TODO. - */ - /* If excess == 0, no tree ops */ - __mem_cgroup_insert_exceeded(mz, mctz, excess); - spin_unlock_irq(&mctz->lock); - css_put(&mz->memcg->css); - loop++; - /* - * Could not reclaim anything and there are no more - * mem cgroups to try or we seem to be looping without - * reclaiming anything. - */ - if (!nr_reclaimed && - (next_mz == NULL || - loop > MEM_CGROUP_MAX_SOFT_LIMIT_RECLAIM_LOOPS)) - break; - } while (!nr_reclaimed); - if (next_mz) - css_put(&next_mz->memcg->css); - return nr_reclaimed; -} - static u64 mem_cgroup_move_charge_read(struct cgroup_subsys_state *css, struct cftype *cft) { diff --git a/mm/vmscan.c b/mm/vmscan.c index fdd13299a04a93..0e04eaf64af3ae 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -6439,8 +6439,6 @@ static void shrink_zones(struct zonelist *zonelist, struct scan_control *sc) { struct zoneref *z; struct zone *zone; - unsigned long nr_soft_reclaimed; - unsigned long nr_soft_scanned; gfp_t orig_mask; pg_data_t *last_pgdat = NULL; pg_data_t *first_pgdat = NULL; @@ -6482,35 +6480,17 @@ static void shrink_zones(struct zonelist *zonelist, struct scan_control *sc) sc->compaction_ready = true; continue; } - - /* - * Shrink each node in the zonelist once. If the - * zonelist is ordered by zone (not the default) then a - * node may be shrunk multiple times but in that case - * the user prefers lower zones being preserved. - */ - if (zone->zone_pgdat == last_pgdat) - continue; - - /* - * This steals pages from memory cgroups over softlimit - * and returns the number of reclaimed pages and - * scanned pages. This works for global memory pressure - * and balancing, not for a memcg's limit. - */ - nr_soft_scanned = 0; - nr_soft_reclaimed = memcg1_soft_limit_reclaim(zone->zone_pgdat, - sc->order, sc->gfp_mask, - &nr_soft_scanned); - sc->nr_reclaimed += nr_soft_reclaimed; - sc->nr_scanned += nr_soft_scanned; - /* need some check for avoid more shrink_zone() */ } if (!first_pgdat) first_pgdat = zone->zone_pgdat; - /* See comment about same check for global reclaim above */ + /* + * Shrink each node in the zonelist once. If the zonelist is + * ordered by zone (not the default) then a node may be shrunk + * multiple times but in that case the user prefers lower zones + * being preserved. + */ if (zone->zone_pgdat == last_pgdat) continue; last_pgdat = zone->zone_pgdat; @@ -7171,8 +7151,6 @@ clear_reclaim_active(pg_data_t *pgdat, int highest_zoneidx) static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx) { int i; - unsigned long nr_soft_reclaimed; - unsigned long nr_soft_scanned; unsigned long pflags; unsigned long nr_boost_reclaim; unsigned long zone_boosts[MAX_NR_ZONES] = { 0, }; @@ -7278,12 +7256,7 @@ static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx) */ kswapd_age_node(pgdat, &sc); - /* Call soft limit reclaim before calling shrink_node. */ sc.nr_scanned = 0; - nr_soft_scanned = 0; - nr_soft_reclaimed = memcg1_soft_limit_reclaim(pgdat, sc.order, - sc.gfp_mask, &nr_soft_scanned); - sc.nr_reclaimed += nr_soft_reclaimed; /* * There should be no need to raise the scanning priority if From b873274d8d7837f82d62631437bbdd4b41c333e5 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:05 -0700 Subject: [PATCH 610/857] memcg: remove mem_cgroup_shrink_node() Its only caller was soft limit reclaim, which is gone. Link: https://lore.kernel.org/20260902174311.1772372-3-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier Signed-off-by: Andrew Morton --- mm/internal.h | 4 ---- mm/vmscan.c | 41 ----------------------------------------- 2 files changed, 45 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 5cc220db907668..e16f1250b25c80 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -78,10 +78,6 @@ unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg, gfp_t gfp_mask, unsigned int reclaim_options, int *swappiness); -unsigned long mem_cgroup_shrink_node(struct mem_cgroup *memcg, - gfp_t gfp_mask, bool noswap, - pg_data_t *pgdat, - unsigned long *nr_scanned); #ifdef CONFIG_NUMA extern int sysctl_min_unmapped_ratio; diff --git a/mm/vmscan.c b/mm/vmscan.c index 0e04eaf64af3ae..d66b5cd167d6f2 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -6805,47 +6805,6 @@ unsigned long try_to_free_pages(struct zonelist *zonelist, int order, #ifdef CONFIG_MEMCG -/* Only used by soft limit reclaim. Do not reuse for anything else. */ -unsigned long mem_cgroup_shrink_node(struct mem_cgroup *memcg, - gfp_t gfp_mask, bool noswap, - pg_data_t *pgdat, - unsigned long *nr_scanned) -{ - struct lruvec *lruvec = mem_cgroup_lruvec(memcg, pgdat); - struct scan_control sc = { - .nr_to_reclaim = SWAP_CLUSTER_MAX, - .target_mem_cgroup = memcg, - .may_writepage = 1, - .may_unmap = 1, - .reclaim_idx = MAX_NR_ZONES - 1, - .may_swap = !noswap, - }; - - WARN_ON_ONCE(!current->reclaim_state); - - sc.gfp_mask = (gfp_mask & GFP_RECLAIM_MASK) | - (GFP_HIGHUSER_MOVABLE & ~GFP_RECLAIM_MASK); - - trace_mm_vmscan_memcg_softlimit_reclaim_begin(sc.gfp_mask, - sc.order, - memcg); - - /* - * NOTE: Although we can get the priority field, using it - * here is not a good idea, since it limits the pages we can scan. - * if we don't reclaim here, the shrink_node from balance_pgdat - * will pick up pages from other mem cgroup's as well. We hack - * the priority and make it zero. - */ - shrink_lruvec(lruvec, &sc); - - trace_mm_vmscan_memcg_softlimit_reclaim_end(sc.nr_reclaimed, memcg); - - *nr_scanned = sc.nr_scanned; - - return sc.nr_reclaimed; -} - unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg, unsigned long nr_pages, gfp_t gfp_mask, From b39b9cb367bdb4e8808b142b30bf03c6162d005f Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:06 -0700 Subject: [PATCH 611/857] memcg: remove the soft limit reclaim tracepoints mm_vmscan_memcg_softlimit_reclaim_begin and mm_vmscan_memcg_softlimit_reclaim_end were only emitted by mem_cgroup_shrink_node(), which is gone, so they can never fire again. Link: https://lore.kernel.org/20260902174311.1772372-4-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier Signed-off-by: Andrew Morton --- include/trace/events/vmscan.h | 14 -------------- 1 file changed, 14 deletions(-) diff --git a/include/trace/events/vmscan.h b/include/trace/events/vmscan.h index b4bf7b8def1f5f..8a872990b4bee6 100644 --- a/include/trace/events/vmscan.h +++ b/include/trace/events/vmscan.h @@ -214,13 +214,6 @@ DEFINE_EVENT(mm_vmscan_direct_reclaim_begin_template, mm_vmscan_memcg_reclaim_be TP_ARGS(gfp_flags, order, memcg) ); - -DEFINE_EVENT(mm_vmscan_direct_reclaim_begin_template, mm_vmscan_memcg_softlimit_reclaim_begin, - - TP_PROTO(gfp_t gfp_flags, int order, struct mem_cgroup *memcg), - - TP_ARGS(gfp_flags, order, memcg) -); #endif /* CONFIG_MEMCG */ DECLARE_EVENT_CLASS(mm_vmscan_direct_reclaim_end_template, @@ -260,13 +253,6 @@ DEFINE_EVENT(mm_vmscan_direct_reclaim_end_template, mm_vmscan_memcg_reclaim_end, TP_ARGS(nr_reclaimed, memcg) ); - -DEFINE_EVENT(mm_vmscan_direct_reclaim_end_template, mm_vmscan_memcg_softlimit_reclaim_end, - - TP_PROTO(unsigned long nr_reclaimed, struct mem_cgroup *memcg), - - TP_ARGS(nr_reclaimed, memcg) -); #endif /* CONFIG_MEMCG */ TRACE_EVENT(mm_shrink_slab_start, From 02da79b484464d7421d8a9576f72bb1494cd2c35 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:07 -0700 Subject: [PATCH 612/857] memcg: remove the soft limit rbtree With soft limit reclaim gone, the per-node rbtree of cgroups in excess has no readers left. Remove the tree, the helpers maintaining it, and the subsys_initcall that existed only to allocate it. memcg1_check_events() no longer needs to feed it, which also drops the last caller of lru_gen_soft_reclaim(). Link: https://lore.kernel.org/20260902174311.1772372-5-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier Signed-off-by: Andrew Morton --- mm/memcontrol-v1.c | 176 +-------------------------------------------- mm/memcontrol-v1.h | 2 - mm/memcontrol.c | 1 - 3 files changed, 2 insertions(+), 177 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index b38b8d0f7f51c5..475f998b764318 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -17,23 +17,6 @@ #include "swap_table.h" #include "memcontrol-v1.h" -/* - * Cgroups above their limits are maintained in a RB-Tree, independent of - * their hierarchy representation - */ - -struct mem_cgroup_tree_per_node { - struct rb_root rb_root; - struct rb_node *rb_rightmost; - spinlock_t lock; -}; - -struct mem_cgroup_tree { - struct mem_cgroup_tree_per_node *rb_tree_per_node[MAX_NUMNODES]; -}; - -static struct mem_cgroup_tree soft_limit_tree __read_mostly; - /* for OOM */ struct mem_cgroup_eventfd_list { struct list_head list; @@ -99,133 +82,6 @@ static struct lockdep_map memcg_oom_lock_dep_map = { DEFINE_SPINLOCK(memcg_oom_lock); -static void __mem_cgroup_insert_exceeded(struct mem_cgroup_per_node *mz, - struct mem_cgroup_tree_per_node *mctz, - unsigned long new_usage_in_excess) -{ - struct rb_node **p = &mctz->rb_root.rb_node; - struct rb_node *parent = NULL; - struct mem_cgroup_per_node *mz_node; - bool rightmost = true; - - if (mz->on_tree) - return; - - mz->usage_in_excess = new_usage_in_excess; - if (!mz->usage_in_excess) - return; - while (*p) { - parent = *p; - mz_node = rb_entry(parent, struct mem_cgroup_per_node, - tree_node); - if (mz->usage_in_excess < mz_node->usage_in_excess) { - p = &(*p)->rb_left; - rightmost = false; - } else { - p = &(*p)->rb_right; - } - } - - if (rightmost) - mctz->rb_rightmost = &mz->tree_node; - - rb_link_node(&mz->tree_node, parent, p); - rb_insert_color(&mz->tree_node, &mctz->rb_root); - mz->on_tree = true; -} - -static void __mem_cgroup_remove_exceeded(struct mem_cgroup_per_node *mz, - struct mem_cgroup_tree_per_node *mctz) -{ - if (!mz->on_tree) - return; - - if (&mz->tree_node == mctz->rb_rightmost) - mctz->rb_rightmost = rb_prev(&mz->tree_node); - - rb_erase(&mz->tree_node, &mctz->rb_root); - mz->on_tree = false; -} - -static void mem_cgroup_remove_exceeded(struct mem_cgroup_per_node *mz, - struct mem_cgroup_tree_per_node *mctz) -{ - unsigned long flags; - - spin_lock_irqsave(&mctz->lock, flags); - __mem_cgroup_remove_exceeded(mz, mctz); - spin_unlock_irqrestore(&mctz->lock, flags); -} - -static unsigned long soft_limit_excess(struct mem_cgroup *memcg) -{ - unsigned long nr_pages = page_counter_read(&memcg->memory); - unsigned long soft_limit = READ_ONCE(memcg->soft_limit); - unsigned long excess = 0; - - if (nr_pages > soft_limit) - excess = nr_pages - soft_limit; - - return excess; -} - -static void memcg1_update_tree(struct mem_cgroup *memcg, int nid) -{ - unsigned long excess; - struct mem_cgroup_per_node *mz; - struct mem_cgroup_tree_per_node *mctz; - - if (lru_gen_enabled()) { - if (soft_limit_excess(memcg)) - lru_gen_soft_reclaim(memcg, nid); - return; - } - - mctz = soft_limit_tree.rb_tree_per_node[nid]; - if (!mctz) - return; - /* - * Necessary to update all ancestors when hierarchy is used. - * because their event counter is not touched. - */ - for (; memcg; memcg = parent_mem_cgroup(memcg)) { - mz = memcg->nodeinfo[nid]; - excess = soft_limit_excess(memcg); - /* - * We have to update the tree if mz is on RB-tree or - * mem is over its softlimit. - */ - if (excess || mz->on_tree) { - unsigned long flags; - - spin_lock_irqsave(&mctz->lock, flags); - /* if on-tree, remove it */ - if (mz->on_tree) - __mem_cgroup_remove_exceeded(mz, mctz); - /* - * Insert again. mz->usage_in_excess will be updated. - * If excess is 0, no tree ops. - */ - __mem_cgroup_insert_exceeded(mz, mctz, excess); - spin_unlock_irqrestore(&mctz->lock, flags); - } - } -} - -void memcg1_remove_from_trees(struct mem_cgroup *memcg) -{ - struct mem_cgroup_tree_per_node *mctz; - struct mem_cgroup_per_node *mz; - int nid; - - for_each_node(nid) { - mz = memcg->nodeinfo[nid]; - mctz = soft_limit_tree.rb_tree_per_node[nid]; - if (mctz) - mem_cgroup_remove_exceeded(mz, mctz); - } -} - static u64 mem_cgroup_move_charge_read(struct cgroup_subsys_state *css, struct cftype *cft) { @@ -336,7 +192,7 @@ static void mem_cgroup_threshold(struct mem_cgroup *memcg) } } -/* Cgroup1: threshold notifications & softlimit tree updates */ +/* Cgroup1: threshold notifications */ /* * Per memcg event counter is incremented at every pagein/pageout. With THP, @@ -405,17 +261,8 @@ static void memcg1_check_events(struct mem_cgroup *memcg, int nid) if (IS_ENABLED(CONFIG_PREEMPT_RT)) return; - /* threshold event is triggered in finer grain than soft limit */ - if (unlikely(memcg1_event_ratelimit(memcg, - MEM_CGROUP_TARGET_THRESH))) { - bool do_softlimit; - - do_softlimit = memcg1_event_ratelimit(memcg, - MEM_CGROUP_TARGET_SOFTLIMIT); + if (unlikely(memcg1_event_ratelimit(memcg, MEM_CGROUP_TARGET_THRESH))) mem_cgroup_threshold(memcg); - if (unlikely(do_softlimit)) - memcg1_update_tree(memcg, nid); - } } void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg) @@ -2391,22 +2238,3 @@ void memcg1_free_events(struct mem_cgroup *memcg) { free_percpu(memcg->events_percpu); } - -static int __init memcg1_init(void) -{ - int node; - - for_each_node(node) { - struct mem_cgroup_tree_per_node *rtpn; - - rtpn = kzalloc_node(sizeof(*rtpn), GFP_KERNEL, node); - - rtpn->rb_root = RB_ROOT; - rtpn->rb_rightmost = NULL; - spin_lock_init(&rtpn->lock); - soft_limit_tree.rb_tree_per_node[node] = rtpn; - } - - return 0; -} -subsys_initcall(memcg1_init); diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index 1e394269c613d8..fd611e66859a32 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -41,7 +41,6 @@ bool memcg1_alloc_events(struct mem_cgroup *memcg); void memcg1_free_events(struct mem_cgroup *memcg); void memcg1_memcg_init(struct mem_cgroup *memcg); -void memcg1_remove_from_trees(struct mem_cgroup *memcg); static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) { @@ -98,7 +97,6 @@ static inline bool memcg1_alloc_events(struct mem_cgroup *memcg) { return true; static inline void memcg1_free_events(struct mem_cgroup *memcg) {} static inline void memcg1_memcg_init(struct mem_cgroup *memcg) {} -static inline void memcg1_remove_from_trees(struct mem_cgroup *memcg) {} static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) {} static inline void memcg1_css_offline(struct mem_cgroup *memcg) {} diff --git a/mm/memcontrol.c b/mm/memcontrol.c index bfd0a74fac9239..29def037681943 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4429,7 +4429,6 @@ static void mem_cgroup_css_free(struct cgroup_subsys_state *css) vmpressure_cleanup(&memcg->vmpressure); cancel_work_sync(&memcg->high_work); - memcg1_remove_from_trees(memcg); free_shrinker_info(memcg); mem_cgroup_free(memcg); } From 0e1c768bbfd3101a899ee4e9a4780288923e9a56 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:08 -0700 Subject: [PATCH 613/857] memcg: remove lru_gen_soft_reclaim() The soft limit rbtree was the only caller. Dropping it leaves MEMCG_LRU_HEAD unreachable, since nothing else ever rotates a memcg with that op, so remove the op too and update the memcg LRU comment. Link: https://lore.kernel.org/20260902174311.1772372-6-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Reviewed-by: T.J. Mercier Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Signed-off-by: Andrew Morton --- include/linux/mmzone.h | 30 +++++++++++------------------- mm/vmscan.c | 16 ++-------------- 2 files changed, 13 insertions(+), 33 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 67c84a8a72583e..84e237f2c17dbc 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -638,35 +638,32 @@ struct lru_gen_mm_walk { * For each node, memcgs are divided into two generations: the old and the * young. For each generation, memcgs are randomly sharded into multiple bins * to improve scalability. For each bin, the hlist_nulls is virtually divided - * into three segments: the head, the tail and the default. + * into two segments: the tail and the default. * * An onlining memcg is added to the tail of a random bin in the old generation. * The eviction starts at the head of a random bin in the old generation. The * per-node memcg generation counter, whose reminder (mod MEMCG_NR_GENS) indexes * the old generation, is incremented when all its bins become empty. * - * There are four operations: - * 1. MEMCG_LRU_HEAD, which moves a memcg to the head of a random bin in its - * current generation (old or young) and updates its "seg" to "head"; - * 2. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its + * There are three operations: + * 1. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its * current generation (old or young) and updates its "seg" to "tail"; - * 3. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old + * 2. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old * generation, updates its "gen" to "old" and resets its "seg" to "default"; - * 4. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the + * 3. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the * young generation, updates its "gen" to "young" and resets its "seg" to * "default". * * The events that trigger the above operations are: - * 1. Exceeding the soft limit, which triggers MEMCG_LRU_HEAD; - * 2. The first attempt to reclaim a memcg below low, which triggers + * 1. The first attempt to reclaim a memcg below low, which triggers * MEMCG_LRU_TAIL; - * 3. The first attempt to reclaim a memcg offlined or below reclaimable size + * 2. The first attempt to reclaim a memcg offlined or below reclaimable size * threshold, which triggers MEMCG_LRU_TAIL; - * 4. The second attempt to reclaim a memcg offlined or below reclaimable size + * 3. The second attempt to reclaim a memcg offlined or below reclaimable size * threshold, which triggers MEMCG_LRU_YOUNG; - * 5. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG; - * 6. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG; - * 7. Offlining a memcg, which triggers MEMCG_LRU_OLD. + * 4. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG; + * 5. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG; + * 6. Offlining a memcg, which triggers MEMCG_LRU_OLD. * * Notes: * 1. Memcg LRU only applies to global reclaim, and the round-robin incrementing @@ -699,7 +696,6 @@ void lru_gen_exit_memcg(struct mem_cgroup *memcg); void lru_gen_online_memcg(struct mem_cgroup *memcg); void lru_gen_offline_memcg(struct mem_cgroup *memcg); void lru_gen_release_memcg(struct mem_cgroup *memcg); -void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid); void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid); bool recheck_lru_gen_max_memcg(struct mem_cgroup *memcg, int nid); void lru_gen_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent, int nid); @@ -740,10 +736,6 @@ static inline void lru_gen_release_memcg(struct mem_cgroup *memcg) { } -static inline void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid) -{ -} - static inline void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid) { } diff --git a/mm/vmscan.c b/mm/vmscan.c index d66b5cd167d6f2..deb087c57007db 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4373,7 +4373,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) /* see the comment on MEMCG_NR_GENS */ enum { MEMCG_LRU_NOP, - MEMCG_LRU_HEAD, MEMCG_LRU_TAIL, MEMCG_LRU_OLD, MEMCG_LRU_YOUNG, @@ -4395,9 +4394,7 @@ static void lru_gen_rotate_memcg(struct lruvec *lruvec, int op) new = old = lruvec->lrugen.gen; /* see the comment on MEMCG_NR_GENS */ - if (op == MEMCG_LRU_HEAD) - seg = MEMCG_LRU_HEAD; - else if (op == MEMCG_LRU_TAIL) + if (op == MEMCG_LRU_TAIL) seg = MEMCG_LRU_TAIL; else if (op == MEMCG_LRU_OLD) new = get_memcg_gen(pgdat->memcg_lru.seq); @@ -4411,7 +4408,7 @@ static void lru_gen_rotate_memcg(struct lruvec *lruvec, int op) hlist_nulls_del_rcu(&lruvec->lrugen.list); - if (op == MEMCG_LRU_HEAD || op == MEMCG_LRU_OLD) + if (op == MEMCG_LRU_OLD) hlist_nulls_add_head_rcu(&lruvec->lrugen.list, &pgdat->memcg_lru.fifo[new][bin]); else hlist_nulls_add_tail_rcu(&lruvec->lrugen.list, &pgdat->memcg_lru.fifo[new][bin]); @@ -4489,15 +4486,6 @@ void lru_gen_release_memcg(struct mem_cgroup *memcg) } } -void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid) -{ - struct lruvec *lruvec = get_lruvec(memcg, nid); - - /* see the comment on MEMCG_NR_GENS */ - if (READ_ONCE(lruvec->lrugen.seg) != MEMCG_LRU_HEAD) - lru_gen_rotate_memcg(lruvec, MEMCG_LRU_HEAD); -} - bool recheck_lru_gen_max_memcg(struct mem_cgroup *memcg, int nid) { struct lruvec *lruvec = get_lruvec(memcg, nid); From b2d728b0df3bef160fcac4026638c1f4112047c8 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:09 -0700 Subject: [PATCH 614/857] memcg: remove the per-node soft limit tree fields tree_node, usage_in_excess and on_tree only existed for the soft limit rbtree. They also doubled as the buffer between the read-mostly head of struct mem_cgroup_per_node and its update-often tail, so replace them with the explicit padding that CONFIG_MEMCG_V1=n already used. Link: https://lore.kernel.org/20260902174311.1772372-7-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier Signed-off-by: Andrew Morton --- include/linux/memcontrol.h | 13 ------------- 1 file changed, 13 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 11c1fa88d6fd0c..1eababed16f532 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -95,20 +95,7 @@ struct mem_cgroup_per_node { struct lruvec_stats *lruvec_stats; struct shrinker_info __rcu *shrinker_info; -#ifdef CONFIG_MEMCG_V1 - /* - * Memcg-v1 only stuff in middle as buffer between read mostly fields - * and update often fields to avoid false sharing. If v1 stuff is - * not present, an explicit padding is needed. - */ - - struct rb_node tree_node; /* RB tree node */ - unsigned long usage_in_excess;/* Set to the value by which */ - /* the soft limit is exceeded*/ - bool on_tree; -#else CACHELINE_PADDING(_pad1_); -#endif /* Fields which get updated often at the end. */ struct lruvec lruvec; From aea4244a9832d2c22289edc41082be8531050b18 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:10 -0700 Subject: [PATCH 615/857] memcg: remove mem_cgroup->soft_limit Nothing reads it anymore, so the field and the helper that reset it on css alloc and css reset can go. Link: https://lore.kernel.org/20260902174311.1772372-8-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier Signed-off-by: Andrew Morton --- include/linux/memcontrol.h | 2 -- mm/memcontrol-v1.h | 6 ------ mm/memcontrol.c | 2 -- 3 files changed, 10 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 1eababed16f532..c799926435560f 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -280,8 +280,6 @@ struct mem_cgroup { struct memcg1_events_percpu __percpu *events_percpu; - unsigned long soft_limit; - /* protected by memcg_oom_lock */ bool oom_lock; int under_oom; diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index fd611e66859a32..f48d0e22e615b6 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -42,11 +42,6 @@ void memcg1_free_events(struct mem_cgroup *memcg); void memcg1_memcg_init(struct mem_cgroup *memcg); -static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) -{ - WRITE_ONCE(memcg->soft_limit, PAGE_COUNTER_MAX); -} - struct cgroup_taskset; void memcg1_css_offline(struct mem_cgroup *memcg); @@ -97,7 +92,6 @@ static inline bool memcg1_alloc_events(struct mem_cgroup *memcg) { return true; static inline void memcg1_free_events(struct mem_cgroup *memcg) {} static inline void memcg1_memcg_init(struct mem_cgroup *memcg) {} -static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) {} static inline void memcg1_css_offline(struct mem_cgroup *memcg) {} static inline bool memcg1_oom_prepare(struct mem_cgroup *memcg, bool *locked) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 29def037681943..bce3962dba5752 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4257,7 +4257,6 @@ mem_cgroup_css_alloc(struct cgroup_subsys_state *parent_css) return ERR_CAST(memcg); page_counter_set_high(&memcg->memory, PAGE_COUNTER_MAX); - memcg1_soft_limit_reset(memcg); #ifdef CONFIG_ZSWAP memcg->zswap_max = PAGE_COUNTER_MAX; WRITE_ONCE(memcg->zswap_writeback, true); @@ -4464,7 +4463,6 @@ static void mem_cgroup_css_reset(struct cgroup_subsys_state *css) page_counter_set_min(&memcg->memory, 0); page_counter_set_low(&memcg->memory, 0); page_counter_set_high(&memcg->memory, PAGE_COUNTER_MAX); - memcg1_soft_limit_reset(memcg); page_counter_set_high(&memcg->swap, PAGE_COUNTER_MAX); memcg_wb_domain_size_changed(memcg); } From 488f03e22b5b6eda211b2284f66c52b522f9d2ea Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:11 -0700 Subject: [PATCH 616/857] memcg: simplify v1 event ratelimiting Thresholds are the only periodic v1 event left, so the target enum, the per-cpu target array and the switch in memcg1_event_ratelimit() all collapse to a single counter. memcg1_check_events() no longer needs a node id either, which lets memcg1_uncharge_batch() drop its nid argument and struct uncharge_gather drop the field feeding it. Link: https://lore.kernel.org/20260902174311.1772372-9-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier Signed-off-by: Andrew Morton --- mm/memcontrol-v1.c | 43 +++++++++++-------------------------------- mm/memcontrol-v1.h | 4 ++-- mm/memcontrol.c | 4 +--- 3 files changed, 14 insertions(+), 37 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index 475f998b764318..bf2c7d53b01b1c 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -200,15 +200,9 @@ static void mem_cgroup_threshold(struct mem_cgroup *memcg) * to trigger some periodic events. This is straightforward and better * than using jiffies etc. to handle periodic memcg event. */ -enum mem_cgroup_events_target { - MEM_CGROUP_TARGET_THRESH, - MEM_CGROUP_TARGET_SOFTLIMIT, - MEM_CGROUP_NTARGETS, -}; - struct memcg1_events_percpu { unsigned long nr_page_events; - unsigned long targets[MEM_CGROUP_NTARGETS]; + unsigned long threshold_target; }; static void memcg1_charge_statistics(struct mem_cgroup *memcg, int nr_pages) @@ -225,43 +219,28 @@ static void memcg1_charge_statistics(struct mem_cgroup *memcg, int nr_pages) } #define THRESHOLDS_EVENTS_TARGET 128 -#define SOFTLIMIT_EVENTS_TARGET 1024 -static bool memcg1_event_ratelimit(struct mem_cgroup *memcg, - enum mem_cgroup_events_target target) +static bool memcg1_event_ratelimit(struct mem_cgroup *memcg) { unsigned long val, next; val = __this_cpu_read(memcg->events_percpu->nr_page_events); - next = __this_cpu_read(memcg->events_percpu->targets[target]); + next = __this_cpu_read(memcg->events_percpu->threshold_target); /* from time_after() in jiffies.h */ if ((long)(next - val) < 0) { - switch (target) { - case MEM_CGROUP_TARGET_THRESH: - next = val + THRESHOLDS_EVENTS_TARGET; - break; - case MEM_CGROUP_TARGET_SOFTLIMIT: - next = val + SOFTLIMIT_EVENTS_TARGET; - break; - default: - break; - } - __this_cpu_write(memcg->events_percpu->targets[target], next); + __this_cpu_write(memcg->events_percpu->threshold_target, + val + THRESHOLDS_EVENTS_TARGET); return true; } return false; } -/* - * Check events in order. - * - */ -static void memcg1_check_events(struct mem_cgroup *memcg, int nid) +static void memcg1_check_events(struct mem_cgroup *memcg) { if (IS_ENABLED(CONFIG_PREEMPT_RT)) return; - if (unlikely(memcg1_event_ratelimit(memcg, MEM_CGROUP_TARGET_THRESH))) + if (unlikely(memcg1_event_ratelimit(memcg))) mem_cgroup_threshold(memcg); } @@ -271,7 +250,7 @@ void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg) local_irq_save(flags); memcg1_charge_statistics(memcg, folio_nr_pages(folio)); - memcg1_check_events(memcg, folio_nid(folio)); + memcg1_check_events(memcg); local_irq_restore(flags); } @@ -344,7 +323,7 @@ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) VM_WARN_ON_IRQS_ENABLED(); memcg1_charge_statistics(memcg, -folio_nr_pages(folio)); preempt_enable_nested(); - memcg1_check_events(memcg, folio_nid(folio)); + memcg1_check_events(memcg); rcu_read_unlock(); obj_cgroup_put(objcg); @@ -398,14 +377,14 @@ void memcg1_swapin(struct folio *folio) #endif void memcg1_uncharge_batch(struct mem_cgroup *memcg, unsigned long pgpgout, - unsigned long nr_memory, int nid) + unsigned long nr_memory) { unsigned long flags; local_irq_save(flags); count_memcg_events(memcg, PGPGOUT, pgpgout); __this_cpu_add(memcg->events_percpu->nr_page_events, nr_memory); - memcg1_check_events(memcg, nid); + memcg1_check_events(memcg); local_irq_restore(flags); } diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index f48d0e22e615b6..b9a21f0fd2c3ac 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -59,7 +59,7 @@ void memcg1_oom_recover(struct mem_cgroup *memcg); void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg); void memcg1_uncharge_batch(struct mem_cgroup *memcg, unsigned long pgpgout, - unsigned long nr_memory, int nid); + unsigned long nr_memory); void memcg1_stat_format(struct mem_cgroup *memcg, struct seq_buf *s); void reparent_memcg1_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent); @@ -107,7 +107,7 @@ static inline void memcg1_commit_charge(struct folio *folio, static inline void memcg1_uncharge_batch(struct mem_cgroup *memcg, unsigned long pgpgout, - unsigned long nr_memory, int nid) {} + unsigned long nr_memory) {} static inline void memcg1_stat_format(struct mem_cgroup *memcg, struct seq_buf *s) {} diff --git a/mm/memcontrol.c b/mm/memcontrol.c index bce3962dba5752..30636b9d96739e 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5328,7 +5328,6 @@ struct uncharge_gather { unsigned long nr_memory; unsigned long pgpgout; unsigned long nr_kmem; - int nid; }; static inline void uncharge_gather_clear(struct uncharge_gather *ug) @@ -5351,7 +5350,7 @@ static void uncharge_batch(const struct uncharge_gather *ug) memcg1_oom_recover(memcg); } - memcg1_uncharge_batch(memcg, ug->pgpgout, ug->nr_memory, ug->nid); + memcg1_uncharge_batch(memcg, ug->pgpgout, ug->nr_memory); rcu_read_unlock(); /* drop reference from uncharge_folio */ @@ -5380,7 +5379,6 @@ static void uncharge_folio(struct folio *folio, struct uncharge_gather *ug) uncharge_gather_clear(ug); } ug->objcg = objcg; - ug->nid = folio_nid(folio); /* pairs with obj_cgroup_put in uncharge_batch */ obj_cgroup_get(objcg); From 2ebcb6190829bd80dfd5fa7bf8c2a8902339a5e5 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:00 -0400 Subject: [PATCH 617/857] mm/page_io: convert write completion handlers to folios Patch series "mm/page_io: folio conversion cleanups", v2. Convert the remaining struct page usage in mm/page_io.c to folios. This removes one of the last callers of end_page_writeback(), along with the last caller of ClearPageReclaim(). This allows us to remove the PG_reclaim page accessors entirely. Clean up a few other stale references to pages throughout while at it. The rename of mm/page_io.c to mm/swap_io.c is deferred to a separate series. This patch (of 6): Convert swap_write_end() and swap_fs_write_complete() to operate on folios directly instead of going through the folio-compat page APIs. This removes calls to end_page_writeback() and set_page_dirty(), and the last caller of ClearPageReclaim(), saving two calls to compound_head() per folio on the write error path. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-0-649728091117@columbia.edu Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-1-649728091117@columbia.edu Signed-off-by: Tal Zussman Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Baoquan He Cc: Barry Song Cc: Chengming Zhou Cc: Chris Li Cc: Christoph Hellwig Cc: David Hildenbrand Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/page_io.c | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 88962571cb931e..fbcf58ff292d8b 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -497,13 +497,13 @@ static void swap_write_end(struct swap_iocb *sio, bool failed) int p; for (p = 0; p < sio->nr_bvecs; p++) { - struct page *page = sio->bvecs[p].bv_page; + struct folio *folio = bvec_folio(&sio->bvecs[p]); if (failed) { - set_page_dirty(page); - ClearPageReclaim(page); + folio_mark_dirty(folio); + folio_clear_reclaim(folio); } - end_page_writeback(page); + folio_end_writeback(folio); } mempool_free(sio, sio_pool); } @@ -514,16 +514,16 @@ static void swap_fs_write_complete(struct kiocb *iocb, long ret) bool failed = ret != sio->len; if (failed) { - struct page *page = sio->bvecs[0].bv_page; + struct folio *folio = bvec_folio(&sio->bvecs[0]); /* * In the case of swap-over-nfs, this can be a temporary failure * if the system has limited memory for allocating transmit - * buffers. Mark the page dirty and avoid + * buffers. Mark the folio dirty and avoid * folio_rotate_reclaimable but rate-limit the messages. */ pr_err_ratelimited("Write error %ld on dio swapfile (%llu)\n", - ret, swap_dev_pos(page_swap_entry(page))); + ret, swap_dev_pos(folio->swap)); } swap_write_end(sio, failed); From e2c379e0128184039d09d08760cdad0b776de563 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:01 -0400 Subject: [PATCH 618/857] mm: remove PageReclaim This flag is now only used on folios, so we can remove all the page accessors. folio_test_clear_reclaim() is not used, so don't add FOLIO_TEST_CLEAR_FLAG() for it. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-2-649728091117@columbia.edu Signed-off-by: Tal Zussman Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton --- include/linux/page-flags.h | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 7a863572adce79..ae2ebaed6d4d96 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -593,8 +593,7 @@ TESTPAGEFLAG(Writeback, writeback, PF_NO_TAIL) FOLIO_FLAG(mappedtodisk, FOLIO_HEAD_PAGE) /* PG_readahead is only used for reads; PG_reclaim is only for writes */ -PAGEFLAG(Reclaim, reclaim, PF_NO_TAIL) - TESTCLEARFLAG(Reclaim, reclaim, PF_NO_TAIL) +FOLIO_FLAG(reclaim, FOLIO_HEAD_PAGE) FOLIO_FLAG(readahead, FOLIO_HEAD_PAGE) FOLIO_TEST_CLEAR_FLAG(readahead, FOLIO_HEAD_PAGE) From ff793fd6a1c662b35b45bd1515bbbc1f1a4cabdc Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:02 -0400 Subject: [PATCH 619/857] mm/page_io: use swap entries directly in zeromap helpers Increment swp_entry_t::val directly instead of recomputing each entry with page_swap_entry(). This removes the last struct page usage in page_io.c and saves one call to compound_head() per page. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-3-649728091117@columbia.edu Signed-off-by: Tal Zussman Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton --- mm/page_io.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index fbcf58ff292d8b..dd95cdb0cd7a18 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -160,7 +160,7 @@ static void swap_zeromap_folio_set(struct folio *folio) struct obj_cgroup *objcg = get_obj_cgroup_from_folio(folio); int nr_pages = folio_nr_pages(folio); struct swap_cluster_info *ci; - swp_entry_t entry; + swp_entry_t entry = folio->swap; unsigned int i; VM_WARN_ON_ONCE_FOLIO(!folio_test_swapcache(folio), folio); @@ -168,8 +168,8 @@ static void swap_zeromap_folio_set(struct folio *folio) ci = swap_cluster_get_and_lock(folio); for (i = 0; i < folio_nr_pages(folio); i++) { - entry = page_swap_entry(folio_page(folio, i)); __swap_table_set_zero(ci, swp_cluster_offset(entry)); + entry.val++; } swap_cluster_unlock(ci); @@ -183,7 +183,7 @@ static void swap_zeromap_folio_set(struct folio *folio) static void swap_zeromap_folio_clear(struct folio *folio) { struct swap_cluster_info *ci; - swp_entry_t entry; + swp_entry_t entry = folio->swap; unsigned int i; VM_WARN_ON_ONCE_FOLIO(!folio_test_swapcache(folio), folio); @@ -191,8 +191,8 @@ static void swap_zeromap_folio_clear(struct folio *folio) ci = swap_cluster_get_and_lock(folio); for (i = 0; i < folio_nr_pages(folio); i++) { - entry = page_swap_entry(folio_page(folio, i)); __swap_table_clear_zero(ci, swp_cluster_offset(entry)); + entry.val++; } swap_cluster_unlock(ci); } From 64860fc9229607b4a35d1407d3dcf1fbc0527d9d Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:03 -0400 Subject: [PATCH 620/857] mm/page_io: rename bio_associate_blkg_from_page() This function takes a folio. Rename it to bio_associate_blkg_from_folio() accordingly. While at it, convert the macro in the !CONFIG_MEMCG || !CONFIG_BLK_CGROUP case to a function. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-4-649728091117@columbia.edu Signed-off-by: Tal Zussman Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton --- mm/page_io.c | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index dd95cdb0cd7a18..295cc6ac6244af 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -277,7 +277,7 @@ static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) return can_merge; } -static void bio_associate_blkg_from_page(struct bio *bio, struct folio *folio) +static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) { struct cgroup_subsys_state *css; @@ -298,7 +298,9 @@ static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) { return true; } -#define bio_associate_blkg_from_page(bio, folio) do { } while (0) +static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) +{ +} #endif /* CONFIG_MEMCG && CONFIG_BLK_CGROUP */ static mempool_t *sio_pool; @@ -596,7 +598,7 @@ static void swap_bdev_submit_write(struct swap_io_ctx *ctx) REQ_OP_WRITE | REQ_SWAP); bio->bi_iter.bi_size = sio->len; bio->bi_iter.bi_sector = swap_folio_sector(bio_first_folio_all(bio)); - bio_associate_blkg_from_page(bio, bio_first_folio_all(bio)); + bio_associate_blkg_from_folio(bio, bio_first_folio_all(bio)); if (ctx->sis->flags & SWP_SYNCHRONOUS_IO) { submit_bio_wait(bio); From ab9126b6e6a4b7902506ac4d57caff6e5a69f98f Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:04 -0400 Subject: [PATCH 621/857] mm/page_io: refer to folios in swap_writeout() comments swap_writeout() operates on folios, not pages. Update its comments accordingly. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-5-649728091117@columbia.edu Signed-off-by: Tal Zussman Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton --- mm/page_io.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 295cc6ac6244af..36466159cf191e 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -209,7 +209,7 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) goto out_unlock; /* - * Arch code may have to preserve more data than just the page + * Arch code may have to preserve more data than just the folio * contents, e.g. memory tags. */ ret = arch_prepare_to_swap(folio); @@ -220,7 +220,7 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) /* * Use the swap table zero mark to avoid doing IO for zero-filled - * pages. The zero mark is protected by the cluster lock, which is + * folios. The zero mark is protected by the cluster lock, which is * acquired internally by swap_zeromap_folio_set/clear. */ if (is_folio_zero_filled(folio)) { From 021668d47960d53a21749720e5892aba35f1124e Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:05 -0400 Subject: [PATCH 622/857] mm/swap: rename __swap_writepage() to __swap_writeout() Commit 84798514db50 ("mm: Remove swap_writepage() and shmem_writepage()") renamed swap_writepage() to swap_writeout(). Rename __swap_writepage(), which operates on a folio, to match its caller. Update a stale reference to swap_writepage() in swapfile.c as well. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-6-649728091117@columbia.edu Signed-off-by: Tal Zussman Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton --- mm/page_io.c | 4 ++-- mm/swap.h | 2 +- mm/swapfile.c | 2 +- mm/zswap.c | 2 +- 4 files changed, 5 insertions(+), 5 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 36466159cf191e..1da4ff484f0971 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -248,7 +248,7 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) } rcu_read_unlock(); - __swap_writepage(ctx, folio); + __swap_writeout(ctx, folio); return 0; out_unlock: folio_unlock(folio); @@ -369,7 +369,7 @@ static void swap_add_folio(struct swap_io_ctx *ctx, struct folio *folio, int rw) } } -void __swap_writepage(struct swap_io_ctx *ctx, struct folio *folio) +void __swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) { VM_BUG_ON_FOLIO(!folio_test_swapcache(folio), folio); diff --git a/mm/swap.h b/mm/swap.h index fddba7a87500a4..0b5d507739bcb6 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -258,7 +258,7 @@ void swap_read_folio(struct swap_io_ctx *ctx, struct folio *folio); void swap_read_submit(struct swap_io_ctx *ctx); void swap_write_submit(struct swap_io_ctx *ctx); int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio); -void __swap_writepage(struct swap_io_ctx *ctx, struct folio *folio); +void __swap_writeout(struct swap_io_ctx *ctx, struct folio *folio); /* linux/mm/swap_state.c */ extern struct address_space swap_space __read_mostly; diff --git a/mm/swapfile.c b/mm/swapfile.c index 601979b97f95b2..3b2279a16d1cf8 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -2928,7 +2928,7 @@ EXPORT_SYMBOL_GPL(add_swap_extent); /* * A `swap extent' is a simple thing which maps a contiguous range of pages * onto a contiguous range of disk blocks. A rbtree of swap extents is - * built at swapon time and is then used at swap_writepage/swap_read_folio + * built at swapon time and is then used at swap_writeout/swap_read_folio * time for locating where on disk a page belongs. * * If the swapfile is an S_ISBLK block device, a single extent is installed. diff --git a/mm/zswap.c b/mm/zswap.c index c1dc60926bad99..fc869d60ef1b48 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1037,7 +1037,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, folio_set_reclaim(folio); /* start writeback */ - __swap_writepage(&ctx, folio); + __swap_writeout(&ctx, folio); swap_write_submit(&ctx); out: From 896798bd849d7762729cb43385d0b87950b50ba4 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sat, 29 Aug 2026 15:42:03 +0800 Subject: [PATCH 623/857] mm/mglru: make type fallback logic explicit in isolate_folios() Patch series "mm/mglru: clean up isolate_folios for readability and clarity", v2. Right now, isolate_folios() is quite difficult to follow: 1. It uses for_each_evictable_type(i, swappiness) to iterate over the types, but 'i' is not actually used as the type within the loop body. 2. It retries the same type when folios were scanned but none could be isolated, but the retry is implemented in a rather subtle way that is difficult to understand. This patchset makes both behaviors explicit and much easier to follow. There are no functional changes for swappiness values from 1 to 200. There is a slight functional change for 0 and 201: with the existing code, there is no chance to retry for these values because for_each_evictable_type() only iterates once. After this patch, 0 and 201 have behavior that is more consistent with the 1-200 range. This patch (of 2): The for_each_evictable_type() loop in isolate_folios() is misleading: it does not actually iterate over each evictable type. Instead, get_type_to_scan() selects the type to scan, while the iterator `i` merely bounds the number of attempts. Make the fallback behavior explicit in the code and remove the opaque for_each_evictable_type(i, swappiness). Link: https://lore.kernel.org/20260829074204.45304-1-baohua@kernel.org Link: https://lore.kernel.org/20260829074204.45304-2-baohua@kernel.org Signed-off-by: Ridong Chen Co-developed-by: Barry Song (Xiaomi) Signed-off-by: Barry Song (Xiaomi) Reviewed-by: Lian Wang Reviewed-by: Baoquan He Cc: Axel Rasmussen Cc: Baolin Wang Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Signed-off-by: Andrew Morton --- mm/vmscan.c | 46 ++++++++++++++++++++++++++-------------------- 1 file changed, 26 insertions(+), 20 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index deb087c57007db..1b40c63706f2ed 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4826,35 +4826,41 @@ static int get_type_to_scan(struct lruvec *lruvec, int swappiness) return positive_ctrl_err(&sp, &pv); } +static inline bool is_single_type_reclaim(int swappiness) +{ + return swappiness == MIN_SWAPPINESS || + swappiness == SWAPPINESS_ANON_ONLY; +} + static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, struct scan_control *sc, int swappiness, struct list_head *list, int *isolated, int *isolate_type, int *isolate_scanned) { - int i; - int total_scanned = 0; + bool type_fallback_allowed = !is_single_type_reclaim(swappiness); int type = get_type_to_scan(lruvec, swappiness); + int total_scanned = 0, scanned, tier; - for_each_evictable_type(i, swappiness) { - int scanned; - int tier = get_tier_idx(lruvec, type); +retry: + tier = get_tier_idx(lruvec, type); + scanned = scan_folios(nr_to_scan, lruvec, sc, + type, tier, list, isolated); - scanned = scan_folios(nr_to_scan, lruvec, sc, - type, tier, list, isolated); + total_scanned += scanned; + if (*isolated) { + *isolate_type = type; + *isolate_scanned = scanned; + return total_scanned; + } - total_scanned += scanned; - if (*isolated) { - *isolate_type = type; - *isolate_scanned = scanned; - break; - } - /* - * If scanned > 0 and isolated == 0, avoid falling back to the - * other type, as this type remains sufficient. Falling back - * too readily can disrupt the positive_ctrl_err() bias. - */ - if (!scanned) - type = !type; + /* + * We are running out of the current reclaim type. Fall back to + * the other type if allowed. + */ + if (!scanned && type_fallback_allowed) { + type = !type; + type_fallback_allowed = false; + goto retry; } return total_scanned; From 40201c38c3e41737ce9d04eb64299c572c9a687e Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Sat, 29 Aug 2026 15:42:04 +0800 Subject: [PATCH 624/857] mm/mglru: make retry logic explicit in isolate_folios() The existing mainline code retries the same type once in a rather subtle way. `for_each_evictable_type()` may provide one more iteration, allowing the same type to be retried if we scanned some folios but failed to isolate any due to protections, promotions, or races. This patch makes the retry behavior explicit. Link: https://lore.kernel.org/20260829074204.45304-3-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Signed-off-by: Andrew Morton --- mm/vmscan.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/mm/vmscan.c b/mm/vmscan.c index 1b40c63706f2ed..413efe44d1f695 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4840,6 +4840,7 @@ static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, bool type_fallback_allowed = !is_single_type_reclaim(swappiness); int type = get_type_to_scan(lruvec, swappiness); int total_scanned = 0, scanned, tier; + bool tried = false; retry: tier = get_tier_idx(lruvec, type); @@ -4859,9 +4860,18 @@ static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, */ if (!scanned && type_fallback_allowed) { type = !type; + tried = true; type_fallback_allowed = false; goto retry; } + /* + * We scanned some folios but failed to isolate any due to promotions, + * protections, or races. Retry once to avoid a larger loop. + */ + if (scanned && !tried) { + tried = true; + goto retry; + } return total_scanned; } From 0cce81d913c24d69bb2bc2f4b909322532d74bb0 Mon Sep 17 00:00:00 2001 From: Avi Weiss Date: Sat, 29 Aug 2026 20:11:12 +0300 Subject: [PATCH 625/857] mm/memory: simplify error handling in insert_pages() Patch series "mm/memory: improve insert_pages() error handling", v3. Improve insert_pages() error handling. The first patch simplifies error handling by initializing the error status to zero and assigning error codes at their respective failure sites. The second patch returns -ENOMEM when walk_to_pmd() fails. A NULL return from walk_to_pmd() indicates failure to allocate an upper page-table level, so -ENOMEM is more appropriate than -EFAULT and is consistent with the subsequent pte_alloc() failure. This patch (of 2): Initialize error return status to zero and then set it as needed at each point of failure. Assign -ENOMEM explicitly when pte_alloc() fails as the pte_alloc() macro returns a boolean. Link: https://lore.kernel.org/cover.1788022178.git.thnkslprpt@gmail.com Link: https://lore.kernel.org/dd3a672c858b38c7525541b19a919e120c4e5a0e.1788022178.git.thnkslprpt@gmail.com Signed-off-by: Avi Weiss Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/memory.c | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 9cbce5c90bffde..2561dc6bdde635 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2565,20 +2565,22 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, unsigned long curr_page_idx = 0; unsigned long remaining_pages_total = *num; unsigned long pages_to_write_in_pmd; - int ret; + int err = 0; more: - ret = -EFAULT; pmd = walk_to_pmd(mm, addr); - if (!pmd) + if (!pmd) { + err = -EFAULT; goto out; + } pages_to_write_in_pmd = min_t(unsigned long, remaining_pages_total, PTRS_PER_PTE - pte_index(addr)); /* Allocate the PTE if necessary; takes PMD lock once only. */ - ret = -ENOMEM; - if (pte_alloc(mm, pmd)) + if (pte_alloc(mm, pmd)) { + err = -ENOMEM; goto out; + } while (pages_to_write_in_pmd) { int pte_idx = 0; @@ -2586,15 +2588,14 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, start_pte = pte_offset_map_lock(mm, pmd, addr, &pte_lock); if (!start_pte) { - ret = -EFAULT; + err = -EFAULT; goto out; } for (pte = start_pte; pte_idx < batch_size; ++pte, ++pte_idx) { - int err = insert_page_in_batch_locked(vma, pte, - addr, pages[curr_page_idx], prot); + err = insert_page_in_batch_locked(vma, pte, addr, + pages[curr_page_idx], prot); if (unlikely(err)) { pte_unmap_unlock(start_pte, pte_lock); - ret = err; remaining_pages_total -= pte_idx; goto out; } @@ -2607,10 +2608,9 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, } if (remaining_pages_total) goto more; - ret = 0; out: *num = remaining_pages_total; - return ret; + return err; } /** From 7ad46d089da75bb032cd28a374b5d7962deb0cea Mon Sep 17 00:00:00 2001 From: Avi Weiss Date: Sat, 29 Aug 2026 20:11:13 +0300 Subject: [PATCH 626/857] mm/memory: return -ENOMEM for page-table allocation failure in insert_pages() walk_to_pmd() returns NULL only when p4d_alloc(), pud_alloc(), or pmd_alloc() fails. These are page-table allocation failures, but insert_pages() currently reports them as -EFAULT. Return -ENOMEM instead, consistent with the subsequent pte_alloc() failure and with the single-page insert_page() path, which reports failure of the same page-table allocation chain as -ENOMEM. Address and range validation failures in vm_insert_pages() continue to return -EFAULT. Keep the later -EFAULT return for pte_offset_map_lock(), which is not an allocation failure. Link: https://lore.kernel.org/9d990c3ed43608e674d4b12a8c221a09fd200f49.1788022178.git.thnkslprpt@gmail.com Signed-off-by: Avi Weiss Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/memory.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/memory.c b/mm/memory.c index 2561dc6bdde635..27ffe1a99a08b0 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2569,7 +2569,7 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, more: pmd = walk_to_pmd(mm, addr); if (!pmd) { - err = -EFAULT; + err = -ENOMEM; goto out; } From 7ed33afbb4ec9f230decce1c12c62ddcd33f81cf Mon Sep 17 00:00:00 2001 From: Wei Yang Date: Sat, 29 Aug 2026 02:58:47 +0000 Subject: [PATCH 627/857] mm: adjust out-dated document of __GFP_NOFAIL Commit ee040cbd6e48 ("mm/page_alloc: don't warn about large allocations with __GFP_NOFAIL") remove a warning on allocating large folio with __GFP_NOFAIL, which was placed there by commit 903edea6c53f ("mm: warn about illegal __GFP_NOFAIL usage in a more appropriate location and manner"). While in that commit, it also documented this behavior which is out-dated now. Adjust the document to align to current code, and adjust the comment while at it. Link: https://lore.kernel.org/20260829025847.26779-1-richard.weiyang@gmail.com Signed-off-by: Wei Yang Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Zi Yan Reviewed-by: Vlastimil Babka (SUSE) Cc: Brendan Jackman Cc: David Hildenbrand Cc: Johannes Weiner Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Signed-off-by: Andrew Morton --- include/linux/gfp_types.h | 2 +- mm/page_alloc.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/include/linux/gfp_types.h b/include/linux/gfp_types.h index 190191411009f2..bfd4c43ed77794 100644 --- a/include/linux/gfp_types.h +++ b/include/linux/gfp_types.h @@ -244,7 +244,7 @@ enum { * definitely preferable to use the flag rather than opencode endless * loop around allocator. * Allocating pages from the buddy with __GFP_NOFAIL and order > 1 is - * not supported. Please consider using kvmalloc() instead. + * discouraged. Please consider using kvmalloc() instead if possible. */ #define __GFP_IO ((__force gfp_t)___GFP_IO) #define __GFP_FS ((__force gfp_t)___GFP_FS) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index ab385bc252ccc0..146f7e0a9462a2 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -4804,7 +4804,7 @@ __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, if (unlikely(nofail)) { /* - * Also we don't support __GFP_NOFAIL without __GFP_DIRECT_RECLAIM, + * We don't support __GFP_NOFAIL without __GFP_DIRECT_RECLAIM, * otherwise, we may result in lockup. */ WARN_ON_ONCE(!can_direct_reclaim); From 891a168c2075a6e16f9d20ffc331139fa95b6773 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Fri, 28 Aug 2026 21:59:42 -0400 Subject: [PATCH 628/857] mm/mempolicy: use SRCU for the weighted interleave state Patch series "mm/mempolicy: stop copying state in the interleave paths". The interleave node selectors and bulk allocators take copies of nodemasks and node weights (for weighted interleave) in the fault path. Both of these copies can be entirely eliminated. For node weights, use SRCU to pin the weights in place. This eliminates a copy and a kmalloc from the bulk allocator path. For nodemasks, we can operate directly on pol->nodes as long as we bounds check the walk. A concurrent rebind can shrink the mask, or tear the read of it so the mask appears empty. - The interleave node selectors fall back to numa_node_id() when that happens, which is what they already did when a copy came back empty. - The bulk allocator simply returns what it managed to allocate. The node count and weight totals are read separately from the nodemask walk that consumes them - creating a time-of-check / time-of-use race. Just clamp the walk to a single pass (number of nodes), and clamp each bulk allocation chunk to the space left in the request. The cost is distribution accuracy during a rebind. The copies never corrected for that either - they only kept the code from dividing by zero and overrunning the allocation request. This patch (of 2): alloc_pages_bulk_weighted_interleave() copies iw_table into a scratch array on every call so it can walk the weights outside of RCU. The copy exists only because the loop may sleep in the page allocator and so cannot hold rcu_read_lock(). Use SRCU to pin the global iw_table object and use it in-place instead. Retire through both flavors - call_srcu() for the sleeping readers, then kfree_rcu() for the reference-less ones - so writers no longer block on synchronize_rcu() either. Tested in a VM with KASAN, PROVE_LOCKING and DEBUG_OBJECTS_RCU_HEAD, with a udelay() injected into the read section to widen the race against concurrent sysfs weight writers, and placement checked against the configured weights. Every retired state reached its callback. Swapping the deferred free for a bare kfree() in the same test reports a use-after-free immediately. Link: https://lore.kernel.org/20260829015943.1258774-1-gourry@gourry.net Link: https://lore.kernel.org/20260829015943.1258774-2-gourry@gourry.net Signed-off-by: Gregory Price (Meta) Suggested-by: Andrew Morton Suggested-by: Matthew Wilcox Assisted-by: Claude:claude-opus-5 Cc: Alistair Popple Cc: Byungchul Park Cc: Chenwandun Cc: David Hildenbrand Cc: "Huang, Ying" Cc: Joshua Hahn Cc: Rakie Kim Cc: "Uladzislau Rezki (Sony)" Cc: Zi Yan Signed-off-by: Andrew Morton --- mm/mempolicy.c | 70 ++++++++++++++++++++++++-------------------------- 1 file changed, 34 insertions(+), 36 deletions(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 060a0eb2691709..2643915dc96699 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -112,6 +112,7 @@ #include #include #include +#include #include #include @@ -157,6 +158,7 @@ static const int weightiness = 32; */ struct weighted_interleave_state { bool mode_auto; + struct rcu_head rcu; u8 iw_table[]; }; static struct weighted_interleave_state __rcu *wi_state; @@ -168,6 +170,24 @@ static unsigned int *node_bw_table; */ static DEFINE_MUTEX(wi_state_lock); +/* Readers that sleep while walking iw_table hold this instead */ +DEFINE_STATIC_SRCU_FAST(wi_srcu); + +static void wi_state_free_rcu(struct rcu_head *head) +{ + struct weighted_interleave_state *state = + container_of(head, struct weighted_interleave_state, rcu); + + kfree_rcu(state, rcu); +} + +/* Retire through both flavors: sleeping readers use SRCU, the rest RCU */ +static void wi_state_retire(struct weighted_interleave_state *state) +{ + if (state) + call_srcu(&wi_srcu, &state->rcu, wi_state_free_rcu); +} + static u8 get_il_weight(int node) { struct weighted_interleave_state *state; @@ -266,10 +286,7 @@ int mempolicy_set_node_perf(unsigned int node, struct access_coordinate *coords) rcu_assign_pointer(wi_state, new_wi_state); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); out: kfree(old_bw); return 0; @@ -2644,7 +2661,8 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, unsigned long nr_allocated = 0; unsigned long rounds; unsigned long node_pages, delta; - u8 *weights, weight; + struct srcu_ctr __percpu *scp; + u8 *table, weight; unsigned int weight_total = 0; unsigned long rem_pages = nr_pages; nodemask_t nodes; @@ -2688,25 +2706,14 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, me->il_weight = 0; prev_node = node; - /* create a local copy of node weights to operate on outside rcu */ - weights = kmalloc(nr_node_ids, gfp & GFP_RECLAIM_MASK); - if (!weights) - return total_allocated; - - rcu_read_lock(); - state = rcu_dereference(wi_state); - if (state) { - memcpy(weights, state->iw_table, nr_node_ids * sizeof(u8)); - rcu_read_unlock(); - } else { - rcu_read_unlock(); - for (i = 0; i < nr_node_ids; i++) - weights[i] = 1; - } + /* The page allocator may sleep, pin the weight table with SRCU */ + scp = srcu_read_lock_fast(&wi_srcu); + state = srcu_dereference(wi_state, &wi_srcu); + table = state ? state->iw_table : NULL; /* calculate total, detect system default usage */ for_each_node_mask(node, nodes) - weight_total += weights[node]; + weight_total += table ? table[node] : 1; /* * Calculate rounds/partial rounds to minimize __alloc_pages_bulk calls. @@ -2718,10 +2725,10 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, rounds = rem_pages / weight_total; delta = rem_pages % weight_total; resume_node = next_node_in(prev_node, nodes); - resume_weight = weights[resume_node]; + resume_weight = table ? table[resume_node] : 1; for (i = 0; i < nnodes; i++) { node = next_node_in(prev_node, nodes); - weight = weights[node]; + weight = table ? table[node] : 1; node_pages = weight * rounds; /* If a delta exists, add this node's portion of the delta */ if (delta > weight) { @@ -2747,7 +2754,7 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, } me->il_prev = resume_node; me->il_weight = resume_weight; - kfree(weights); + srcu_read_unlock_fast(&wi_srcu, scp); return total_allocated; } @@ -3673,10 +3680,7 @@ static ssize_t node_store(struct kobject *kobj, struct kobj_attribute *attr, rcu_assign_pointer(wi_state, new_wi_state); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); return count; } @@ -3742,10 +3746,7 @@ static ssize_t weighted_interleave_auto_store(struct kobject *kobj, update_wi_state: rcu_assign_pointer(wi_state, new_wi_state); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); return count; } @@ -3789,10 +3790,7 @@ static void wi_state_free(void) rcu_assign_pointer(wi_state, NULL); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); } static struct kobj_attribute wi_auto_attr = { From a77adf1514c138dcfa3ee572053fdf65047e7465 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Fri, 28 Aug 2026 21:59:43 -0400 Subject: [PATCH 629/857] mm/mempolicy: stop copying the nodemask in the interleave paths The interleave node selectors copy pol->nodes onto the stack so the mask cannot change while they walk it. nodemask_t is 128 bytes at MAX_NUMNODES=1024, and two of the three run per folio fault. The copy only buys consistency between the node count and the walk. Drop the consistency and just bounds check the walk instead. If an empty nodelist or weight is perceived, fall back to numa_node_id(), which is what the functions already did when the copy came back empty. weighted_interleave_nid() counts the nodes as we sum the weights. We use that node count to limit the maximum skew a single node can host. interleave_nid() walks with next_node_in() rather than next_node(), so a mask that shrank mid-walk wraps to a node still in the policy. alloc_pages_bulk_weighted_interleave() derives per-node counts from a weight total summed over the mask, so a changing mask can make them exceed the request. Clamp each chunk to the space left in page_array. A cpuset cookie will not work here: two of these take VMA policies, which mpol_rebind_mm() rebinds under mmap_write_lock(), not mems_allowed_seq. Cost is distribution accuracy during a rebind - but the copy never corrected this anyway, it was just a safety mechanism to prevent div/0 and overrunning the alloc request buffer. Remove read_once_policy_nodemask(), now unused. -fstack-usage at MAX_NUMNODES=1024: weighted_interleave_nid 184 -> 56 interleave_nid 168 -> 32 alloc_pages_bulk_mempolicy_noprof 360 -> 136 Link: https://lore.kernel.org/20260829015943.1258774-3-gourry@gourry.net Signed-off-by: Gregory Price (Meta) Assisted-by: Claude:claude-opus-5 Cc: Alistair Popple Cc: Byungchul Park Cc: Chenwandun Cc: David Hildenbrand Cc: "Huang, Ying" Cc: Joshua Hahn Cc: Matthew Wilcox Cc: Rakie Kim Cc: "Uladzislau Rezki (Sony)" Cc: Zi Yan Signed-off-by: Andrew Morton --- mm/mempolicy.c | 86 ++++++++++++++++++++++++++++---------------------- 1 file changed, 49 insertions(+), 37 deletions(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 2643915dc96699..2961291261090a 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2197,34 +2197,15 @@ unsigned int mempolicy_slab_node(void) } } -static unsigned int read_once_policy_nodemask(struct mempolicy *pol, - nodemask_t *mask) -{ - /* - * barrier stabilizes the nodemask locally so that it can be iterated - * over safely without concern for changes. Allocators validate node - * selection does not violate mems_allowed, so this is safe. - */ - barrier(); - memcpy(mask, &pol->nodes, sizeof(nodemask_t)); - barrier(); - return nodes_weight(*mask); -} - static unsigned int weighted_interleave_nid(struct mempolicy *pol, pgoff_t ilx) { struct weighted_interleave_state *state; - nodemask_t nodemask; - unsigned int target, nr_nodes; + unsigned int target, nnodes = 0; u8 *table = NULL; unsigned int weight_total = 0; u8 weight; int nid = 0; - nr_nodes = read_once_policy_nodemask(pol, &nodemask); - if (!nr_nodes) - return numa_node_id(); - rcu_read_lock(); state = rcu_dereference(wi_state); @@ -2232,22 +2213,40 @@ static unsigned int weighted_interleave_nid(struct mempolicy *pol, pgoff_t ilx) if (state) table = state->iw_table; - /* calculate the total weight */ - for_each_node_mask(nid, nodemask) + /* calculate the total weight and the node count */ + for_each_node_mask(nid, pol->nodes) { weight_total += table ? table[nid] : 1; + nnodes++; + } + + /* the mask is empty */ + if (!weight_total) { + rcu_read_unlock(); + return numa_node_id(); + } /* Calculate the node offset based on totals */ target = ilx % weight_total; - nid = first_node(nodemask); - while (target) { + nid = first_node(pol->nodes); + + /* + * The target was calculated in a separate loop, and a concurrent + * rebind can change the total number of nodes. Clamp this loop to + * a single pass (nnodes) to keep the walk bounded by node count. + */ + while (target && nnodes-- && nid < MAX_NUMNODES) { /* detect system default usage */ weight = table ? table[nid] : 1; if (target < weight) break; target -= weight; - nid = next_node_in(nid, nodemask); + nid = next_node_in(nid, pol->nodes); } rcu_read_unlock(); + + /* the mask emptied under the walk */ + if (nid >= MAX_NUMNODES) + return numa_node_id(); return nid; } @@ -2258,18 +2257,21 @@ static unsigned int weighted_interleave_nid(struct mempolicy *pol, pgoff_t ilx) */ static unsigned int interleave_nid(struct mempolicy *pol, pgoff_t ilx) { - nodemask_t nodemask; unsigned int target, nnodes; int i; int nid; - nnodes = read_once_policy_nodemask(pol, &nodemask); + nnodes = nodes_weight(pol->nodes); if (!nnodes) return numa_node_id(); target = ilx % nnodes; - nid = first_node(nodemask); - for (i = 0; i < target; i++) - nid = next_node(nid, nodemask); + nid = first_node(pol->nodes); + for (i = 0; i < target && nid < MAX_NUMNODES; i++) + nid = next_node_in(nid, pol->nodes); + + /* the mask emptied under the walk */ + if (nid >= MAX_NUMNODES) + return numa_node_id(); return nid; } @@ -2665,7 +2667,6 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, u8 *table, weight; unsigned int weight_total = 0; unsigned long rem_pages = nr_pages; - nodemask_t nodes; int nnodes, node; int resume_node = MAX_NUMNODES - 1; u8 resume_weight = 0; @@ -2675,10 +2676,10 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, if (!nr_pages) return 0; - /* read the nodes onto the stack, retry if done during rebind */ + /* count the nodes, retry if a rebind happened during the read */ do { cpuset_mems_cookie = read_mems_allowed_begin(); - nnodes = read_once_policy_nodemask(pol, &nodes); + nnodes = nodes_weight(pol->nodes); } while (read_mems_allowed_retry(cpuset_mems_cookie)); /* if the nodemask has become invalid, we cannot do anything */ @@ -2688,7 +2689,7 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, /* Continue allocating from most recent node and adjust the nr_pages */ node = me->il_prev; weight = me->il_weight; - if (weight && node_isset(node, nodes)) { + if (weight && node_isset(node, pol->nodes)) { node_pages = min(rem_pages, weight); nr_allocated = __alloc_pages_bulk(gfp, node, NULL, node_pages, page_array); @@ -2712,9 +2713,13 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, table = state ? state->iw_table : NULL; /* calculate total, detect system default usage */ - for_each_node_mask(node, nodes) + for_each_node_mask(node, pol->nodes) weight_total += table ? table[node] : 1; + /* the mask emptied since it was counted */ + if (!weight_total) + goto out; + /* * Calculate rounds/partial rounds to minimize __alloc_pages_bulk calls. * Track which node weighted interleave should resume from. @@ -2724,10 +2729,14 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, */ rounds = rem_pages / weight_total; delta = rem_pages % weight_total; - resume_node = next_node_in(prev_node, nodes); + resume_node = next_node_in(prev_node, pol->nodes); + if (resume_node >= MAX_NUMNODES) + goto out; resume_weight = table ? table[resume_node] : 1; for (i = 0; i < nnodes; i++) { - node = next_node_in(prev_node, nodes); + node = next_node_in(prev_node, pol->nodes); + if (node >= MAX_NUMNODES) + break; weight = table ? table[node] : 1; node_pages = weight * rounds; /* If a delta exists, add this node's portion of the delta */ @@ -2744,6 +2753,8 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, /* node_pages can be 0 if an allocation fails and rounds == 0 */ if (!node_pages) break; + /* a rebind can invalidate the counts: never overrun page_array */ + node_pages = min(node_pages, nr_pages - total_allocated); nr_allocated = __alloc_pages_bulk(gfp, node, NULL, node_pages, page_array); page_array += nr_allocated; @@ -2754,6 +2765,7 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, } me->il_prev = resume_node; me->il_weight = resume_weight; +out: srcu_read_unlock_fast(&wi_srcu, scp); return total_allocated; } From e55ea6bcaed18b543a2ac1e5caecfa0be44e06e3 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 19 Aug 2026 01:48:30 +0900 Subject: [PATCH 630/857] percpu: fix the comment about which sizes share a slot A percpu allocation is at least PCPU_MIN_ALLOC_SIZE bytes, and __pcpu_size_to_slot() returns 1 for sizes below 16 bytes and 2 for sizes from 16 to 31 bytes. So fix the wrong comment. Link: https://lore.kernel.org/20260818164831.3138490-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Cc: Dennis Zhou Cc: Tejun Heo Cc: Christoph Lameter Signed-off-by: Andrew Morton --- mm/percpu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/percpu.c b/mm/percpu.c index a802d72c116fbb..47a903fe3b5124 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -100,7 +100,7 @@ /* * The slots are sorted by the size of the biggest continuous free area. - * 1-31 bytes share the same slot. + * [PCPU_MIN_ALLOC_SIZE..15] bytes share the same slot. */ #define PCPU_SLOT_BASE_SHIFT 5 /* chunks in slots below this are subject to being sidelined on failed alloc */ From bd61c125c8954a1af09d39a111fd87d97ed3c60f Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Tue, 18 Aug 2026 03:06:23 -0700 Subject: [PATCH 631/857] mm, swap: distinguish a malformed swap entry from a dying device Patch series "mm, swap: don't spin on a bad swap entry", v3. I've seen some machines at Meta fleet that show the following type of problem: 1) It gets some weird warning: BUG: Bad page map in process khugepaged pte:f000eef300000017 pmd:00000067 addr:00007f57c0a01000 vm_flags:20200073 anon_vma:ffff88829af7c340 mapping:0000000000000000 index:7f57c0a01 The corruption is most likely the collapse/PT_RECLAIM race fixed by commit 366a4532d96f ("mm: fix the race between collapse and PT_RECLAIM under per-vma lock"). But this series is not about this one. 2) Then the fault never makes progress. do_swap_page() returns 0 when get_swap_device() fails, so the fault is retried, reads the same entry and faults again. Nothing in the round trip changes the PTE, and the same line comes out on every pass: get_swap_device: Bad swap offset entry 3ffffffc043c5 Patch 1 makes get_swap_device() return ERR_PTR(-EIO) for a malformed entry, keeping NULL for a device swapoff is taking away, and converts the callers. No functional change expected. Patch 2 uses that to return VM_FAULT_SIGBUS instead of retrying. This patch (of 2): get_swap_device() returns NULL for two different things: an entry whose type names no swap device or whose offset is past the end of one, and a device that swapoff is taking away. The first never becomes valid, the second does, and callers cannot tell them apart. Return ERR_PTR(-EIO) for the two malformed cases and keep NULL for swapoff. copy_nonpresent_pte() already reports -EIO for an entry whose type names no device. Callers bail out on failure either way, so switch them to IS_ERR_OR_NULL(), and let the two paths that drop the reference skip an error pointer. No functional change. Link: https://lore.kernel.org/20260818-swap-v3-0-d3fa52598a59@debian.org Link: https://lore.kernel.org/20260818-swap-v3-1-d3fa52598a59@debian.org Signed-off-by: Breno Leitao Reviewed-by: Barry Song Acked-by: Kairui Song Reviewed-by: Nhat Pham Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Chengming Zhou Cc: Chris Li Cc: Hugh Dickins Cc: Jann Horn Cc: Johannes Weiner Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Pedro Falcato Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/memory.c | 6 +++--- mm/mincore.c | 2 +- mm/shmem.c | 2 +- mm/swap_state.c | 4 ++-- mm/swapfile.c | 15 ++++++++++----- mm/userfaultfd.c | 4 ++-- mm/zswap.c | 2 +- 7 files changed, 20 insertions(+), 15 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 27ffe1a99a08b0..e59d6c2a343205 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4956,9 +4956,9 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) goto out; } - /* Prevent swapoff from happening to us. */ + /* Prevent swapoff from happening to us, and reject a bad entry. */ si = get_swap_device(entry); - if (unlikely(!si)) + if (IS_ERR_OR_NULL(si)) goto out; folio = swap_cache_get_folio(entry); @@ -5268,7 +5268,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) if (vmf->pte) pte_unmap_unlock(vmf->pte, vmf->ptl); out: - if (si) + if (!IS_ERR_OR_NULL(si)) put_swap_device(si); return ret; out_nomap: diff --git a/mm/mincore.c b/mm/mincore.c index ff4ac828176837..c086836bc4bcc5 100644 --- a/mm/mincore.c +++ b/mm/mincore.c @@ -71,7 +71,7 @@ static unsigned char mincore_swap(swp_entry_t entry, bool shmem) */ if (shmem) { si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return 0; } folio = swap_cache_get_folio(entry); diff --git a/mm/shmem.c b/mm/shmem.c index 9c76032c396ee6..eeb9a78c125a55 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2478,7 +2478,7 @@ static int shmem_swapin_folio(struct inode *inode, pgoff_t index, si = get_swap_device(index_entry); order = shmem_confirm_swap(mapping, index, index_entry); - if (unlikely(!si)) { + if (IS_ERR_OR_NULL(si)) { if (order < 0) return -EEXIST; else diff --git a/mm/swap_state.c b/mm/swap_state.c index f3961fdd857dc6..305877e1f4d7bf 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -716,7 +716,7 @@ struct folio *read_swap_cache_async(struct swap_io_ctx *ctx, swp_entry_t entry, struct folio *folio; si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return NULL; mpol = get_vma_policy(vma, addr, 0, &ilx); @@ -952,7 +952,7 @@ static struct folio *swap_vma_readahead(swp_entry_t targ_entry, gfp_t gfp_mask, */ if (swp_type(entry) != swp_type(targ_entry)) { si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) continue; } folio = swap_cache_read_folio(&ctx, entry, gfp_mask, mpol, ilx, diff --git a/mm/swapfile.c b/mm/swapfile.c index 3b2279a16d1cf8..408f6c72fb5a69 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1504,7 +1504,7 @@ int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp) unsigned long offset = swp_offset(entry); si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return 0; ci = __swap_offset_to_cluster(si, offset); @@ -1859,7 +1859,10 @@ void folio_put_swap(struct folio *folio, struct page *page) * Check whether swap entry is valid in the swap device. If so, * return pointer to swap_info_struct, and keep the swap entry valid * via preventing the swap device from being swapoff, until - * put_swap_device() is called. Otherwise return NULL. + * put_swap_device() is called. Return NULL for an empty entry or a + * device that is going away, and ERR_PTR(-EIO) if the entry's type + * names no swap device or its offset is past the end of one. These EIOs + * are preceded by pr_err(). * * Notice that swapoff or swapoff+swapon can still happen before the * percpu_ref_tryget_live() in get_swap_device() or after the @@ -1900,12 +1903,14 @@ struct swap_info_struct *get_swap_device(swp_entry_t entry) return si; bad_nofile: pr_err_ratelimited("%s: %s%08lx\n", __func__, Bad_file, entry.val); + return ERR_PTR(-EIO); + out: return NULL; put_out: pr_err_ratelimited("%s: %s%08lx\n", __func__, Bad_offset, entry.val); percpu_ref_put(&si->users); - return NULL; + return ERR_PTR(-EIO); } /* @@ -2001,7 +2006,7 @@ int swp_swapcount(swp_entry_t entry) int count; si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return 0; ci = swap_cluster_lock(si, swp_offset(entry)); @@ -2127,7 +2132,7 @@ void swap_put_entries_direct(swp_entry_t entry, int nr) struct swap_info_struct *si; si = get_swap_device(entry); - if (WARN_ON_ONCE(!si)) + if (WARN_ON_ONCE(IS_ERR_OR_NULL(si))) return; if (WARN_ON_ONCE(end_offset > si->max)) goto out; diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 74f04c323c50fb..95c1ed96df9a7a 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -1700,7 +1700,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd } si = get_swap_device(entry); - if (unlikely(!si)) { + if (IS_ERR_OR_NULL(si)) { ret = -EAGAIN; goto out; } @@ -1757,7 +1757,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd if (dst_pte) pte_unmap(dst_pte); mmu_notifier_invalidate_range_end(&range); - if (si) + if (!IS_ERR_OR_NULL(si)) put_swap_device(si); return ret; diff --git a/mm/zswap.c b/mm/zswap.c index fc869d60ef1b48..f3ae3c81e48eac 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -984,7 +984,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, /* try to allocate swap cache folio */ si = get_swap_device(swpentry); - if (!si) + if (IS_ERR_OR_NULL(si)) return -EEXIST; mpol = get_task_policy(current); From 13cc0c6e11a0509d38483a13bb57a80195fb6e94 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Tue, 18 Aug 2026 03:06:24 -0700 Subject: [PATCH 632/857] mm: fail the fault on a malformed swap entry instead of retrying it do_swap_page() returns 0 when get_swap_device() fails, which the fault handler reads as "handled". For an entry that can never become valid the retry takes the same fault again, so the thread spins forever, retrying on the same fault. Return VM_FAULT_SIGBUS (Bad access) for a malformed entry (pr_err() was called at get_swap_device()). Link: https://lore.kernel.org/20260818-swap-v3-2-d3fa52598a59@debian.org Signed-off-by: Breno Leitao Acked-by: Kairui Song Reviewed-by: Barry Song Reviewed-by: Nhat Pham Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Chengming Zhou Cc: Chris Li Cc: Hugh Dickins Cc: Jann Horn Cc: Johannes Weiner Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Pedro Falcato Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/memory.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/mm/memory.c b/mm/memory.c index e59d6c2a343205..347db2acd0f83b 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4958,8 +4958,11 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) /* Prevent swapoff from happening to us, and reject a bad entry. */ si = get_swap_device(entry); - if (IS_ERR_OR_NULL(si)) + if (IS_ERR_OR_NULL(si)) { + if (IS_ERR(si)) + ret = VM_FAULT_SIGBUS; goto out; + } folio = swap_cache_get_folio(entry); if (folio) From 3bcdc18651f8b207462a8765aeb8f11ff5472f34 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Mon, 17 Aug 2026 18:08:08 -0400 Subject: [PATCH 633/857] mm/huge_memory: skip zone device folios in madvise_free_huge_pmd() Patch series "mm: reject zone device folios in more folio walkers", v2. Several LRU-oriented mm walkers resolve the folio backing a PMD entry (or a physical pfn) and then reclaim, age, migrate, or lazyfree it without ever checking for ZONE_DEVICE memory. This series adds missing folio_is_zone_device() rejections, matching the checks that comparable walkers already perform. - mm/huge_memory, mm/madvise: the !pmd_present branch above these sites only filters device-private entries (which are non-present). A present zone device PMD (e.g. device-coherent) would still reach the folio and be lazyfreed / aged / paged out. Add an explicit check. - mm/mempolicy: queue_folios_pmd() can see a present zone device PMD (e.g. device-coherent) and queue it for migration. No crash reproducer - this is a correctness/hardening cleanup found by inspection. All checks are placed after the folio is resolved and before it is acted upon, on paths that already hold the relevant page-table lock, so no locking or refcount changes are involved. This patch (of 3): madvise_free_huge_pmd() resolves the folio backing a PMD via pmd_folio() and marks it lazyfree without checking for zone device memory. The surrounding guards do not cover every zone device case: - MADV_FREE only operates on anonymous VMAs (DAX mappings are excluded) - !pmd_present() branch rejects device-private and migration entries - present zone device PMD (device coherent THP) is not filtered. Unlike vm_normal_page_pmd(), it performs no special/pfnmap check, and would be marked lazyfree here. Bail out when the folio is a zone device folio. Link: https://lore.kernel.org/20260817220810.1175596-1-gourry@gourry.net Link: https://lore.kernel.org/20260817220810.1175596-2-gourry@gourry.net Fixes: a30b48bf1b24 ("mm/migrate_device: implement THP migration of zone device pages") Signed-off-by: Gregory Price (Meta) Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Tested-by: Lance Yang Cc: Alistair Popple Cc: Balbir Singh Cc: Baolin Wang Cc: Barry Song Cc: Byungchul Park Cc: Dev Jain Cc: Gregory Price Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Liam R. Howlett Cc: Matthew Brost Cc: Rakie Kim Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Zi Yan Cc: Signed-off-by: Andrew Morton --- mm/huge_memory.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 1e5d68acf62a52..54494c3fa9835e 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -2423,6 +2423,10 @@ bool madvise_free_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma, } folio = pmd_folio(orig_pmd); + + if (folio_is_zone_device(folio)) + goto out; + /* * If other processes are mapping this folio, we couldn't discard * the folio unless they all do MADV_FREE so let's skip the folio. From e8acbf6d7348e47f64f022d15ce36d49b1af50ac Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Mon, 17 Aug 2026 18:08:09 -0400 Subject: [PATCH 634/857] mm/madvise: skip zone device folios in cold/pageout PMD range madvise_cold_or_pageout_pte_range() resolves the folio backing a PMD via pmd_folio() and ages or reclaims it without checking for zone device memory. The surrounding guards do not cover every zone device case: - can_madv_lru_vma() excludes VM_PFNMAP and VM_HUGETLB VMAs (so device DAX is filtered) - !pmd_present() branch above rejects device-private and migration entries, which are non-present. - A present zone device PMD - e.g. a device-coherent THP - is not filtered by any of these, nor by pmd_folio() (unlike vm_normal_page_pmd(), it performs no special/pfnmap check), and would be aged or paged out here. Skip ZONE_DEVICE folios explicitly during MADV_COLD/PAGEOUT. Link: https://lore.kernel.org/20260817220810.1175596-3-gourry@gourry.net Fixes: a30b48bf1b24 ("mm/migrate_device: implement THP migration of zone device pages") Signed-off-by: Gregory Price (Meta) Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Balbir Singh Tested-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Barry Song Cc: Byungchul Park Cc: Dev Jain Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Liam R. Howlett Cc: Matthew Brost Cc: Rakie Kim Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Zi Yan Cc: Signed-off-by: Andrew Morton --- mm/madvise.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/madvise.c b/mm/madvise.c index eeee82cf2b3f4b..73c2901b9adbf0 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -404,6 +404,9 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, folio = pmd_folio(orig_pmd); + if (folio_is_zone_device(folio)) + goto huge_unlock; + /* Do not interfere with other mappings of this folio */ if (folio_maybe_mapped_shared(folio)) goto huge_unlock; From 9ebb788ef079862fe586a90986c0ba1205c95961 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Mon, 17 Aug 2026 18:08:10 -0400 Subject: [PATCH 635/857] mm/mempolicy: skip zone device folios when queueing folios queue_folios_pte_range() already pairs vm_normal_folio() with an explicit folio_is_zone_device() check before adding folios to the migration pagelist. vm_normal_folio() alone does not reject zone device memory (a present device-coherent page in a normal VMA is returned as "normal"). Mirror the explicit check in queue_folios_pmd() as well. queue_folios_pmd() uses pmd_folio() directly and can encounter a present zone device PMD - e.g. a device-coherent THP. This is not filtered by existing checks: !pmd_present() - only rejects non-present device-private and migration entries vma_migratable() - excludes DAX and VM_PFNMAP. The early return also means such a folio is no longer counted in qp->nr_failed under MPOL_MF_STRICT. This is the same pattern used by queue_folios_pte_range() (skipping zone device without failing). Link: https://lore.kernel.org/20260817220810.1175596-4-gourry@gourry.net Fixes: a30b48bf1b24 ("mm/migrate_device: implement THP migration of zone device pages") Signed-off-by: Gregory Price (Meta) Reviewed-by: Balbir Singh Acked-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Tested-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Barry Song Cc: Byungchul Park Cc: Dev Jain Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Liam R. Howlett Cc: Matthew Brost Cc: Rakie Kim Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Zi Yan Cc: Signed-off-by: Andrew Morton --- mm/mempolicy.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 2961291261090a..2ad0a5f18280a0 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -679,6 +679,8 @@ static void queue_folios_pmd(pmd_t *pmd, struct mm_walk *walk) return; } folio = pmd_folio(pmdval); + if (folio_is_zone_device(folio)) + return; if (is_huge_zero_folio(folio)) { walk->action = ACTION_CONTINUE; return; From b84c17412e1e5101b96ff7c50d4872287e900da5 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Mon, 17 Aug 2026 14:19:55 +0800 Subject: [PATCH 636/857] selftests/mm: khugepaged: consolidate error exits via kselftest helpers Replace the perror()+exit(EXIT_FAILURE) pattern with ksft_exit_fail_perror() so failures are reported through the kselftest framework, consistent with the rest of the file. Link: https://lore.kernel.org/20260817061955.45454-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Acked-by: David Hildenbrand (Arm) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Zi Yan Reviewed-by: Lance Yang Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/khugepaged.c | 20 +++++++------------- 1 file changed, 7 insertions(+), 13 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 83d27d069c4139..f82673f5f6b47e 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -337,21 +337,15 @@ static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup) ksft_exit_fail_perror("open()"); size = nr_hpages * hpage_pmd_size; - if (ftruncate(fd, size)) { - perror("ftruncate()"); - exit(EXIT_FAILURE); - } + if (ftruncate(fd, size)) + ksft_exit_fail_perror("ftruncate()"); p = mmap(BASE_ADDR, size, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0); - if (p != BASE_ADDR) { - perror("mmap()"); - exit(EXIT_FAILURE); - } + if (p != BASE_ADDR) + ksft_exit_fail_perror("mmap()"); fill_memory(p, 0, size); - if (msync(p, size, MS_SYNC)) { - perror("msync()"); - exit(EXIT_FAILURE); - } + if (msync(p, size, MS_SYNC)) + ksft_exit_fail_perror("msync()"); close(fd); munmap(p, size); success("OK"); @@ -426,7 +420,7 @@ static bool file_check_huge(void *addr, size_t len, int nr_hpages, case VMA_SHMEM: return check_huge_shmem(addr, len, nr_hpages, hpage_size); default: - exit(EXIT_FAILURE); + ksft_exit_fail_msg("Unknown VMA type\n"); return false; } } From 3e647bfb0aa675ef2c40c476f093bad6e0c1a45f Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sun, 30 Aug 2026 08:20:43 +0800 Subject: [PATCH 637/857] memcg: acquire peaks_lock when reading memory.peak MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm, memcg: fix memory.peak reset clobbering other fds' watermark", v4. The memory.peak / memory.swap.peak per-fd watermark tracking has two issues. Each open fd is a watcher and reads back max(its own value, the shared local_watermark); both bugs live in that scheme. Worst case for both is the same and is userspace-visible: a reader of memory.peak (or memory.swap.peak) gets a value lower than the true peak, so a tool that sizes or bills a cgroup by its peak usage under-reports it. Patch 1 (read side) fixes the race Sashiko pointed out in the v1 review [1]: peak_show() inspects local_watermark and the per-fd values without holding peaks_lock, so a reader that races an unrelated peak_write() reset briefly observes the lowered value. Transient. It takes peaks_lock in the show path. Patch 2 (write side) fixes peak_write(): on a reset it stores the current usage into the other watchers instead of the old watermark, so once usage has dropped from a peak a reset on one fd drags every other fd's peak down too, even fds that never reset. This patch (of 2): Sashiko reported that a reader can transiently observe a lower peak within a race window [1]. peak_show() returns max(local_watermark, ofp->value), but peak_write() updates those two under peaks_lock while the reader takes no lock. The interleaving is: writer (reset on fd A) reader (fd B) ---------------------- ------------- usage = page_counter_read(pc) WRITE_ONCE(local_watermark, usage) // watermark lowered to usage lw = READ_ONCE(local_watermark) // sees the lowered usage val = READ_ONCE(ofp->value) // B's value not updated yet return max(lw, val) // both low -> low peak WRITE_ONCE(peer_ctx->value, usage) // B updated, but too late Fix it by acquiring peaks_lock when reading the peak, so the reader sees a consistent snapshot of local_watermark and the per-fd values. The same race applies to memory.swap.peak, which shares peaks_lock and the peak_write() path, so take the lock there as well. Link: https://lore.kernel.org/20260830002044.1938621-1-ridong.chen@linux.dev Link: https://lore.kernel.org/20260830002044.1938621-2-ridong.chen@linux.dev Link: https://sashiko.dev/#/patchset/20260730115314.1069089-1-ridong.chen@linux.dev?part=1 [1] Fixes: c6f53ed8f213 ("mm, memcg: cg2 memory{.swap,}.peak write handlers") Signed-off-by: Ridong Chen Assisted-by: Claude:claude-opus-4-8 Acked-by: Johannes Weiner Acked-by: Shakeel Butt Reviewed-by: Muchun Song Cc: David Finkel Cc: Michal Hocko Cc: Michal Koutný Cc: Roman Gushchin Cc: Tejun Heo Cc: Tao Cui Signed-off-by: Andrew Morton --- mm/memcontrol.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 30636b9d96739e..1ee974cb6d3727 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4745,6 +4745,7 @@ static int memory_peak_show(struct seq_file *sf, void *v) { struct mem_cgroup *memcg = mem_cgroup_from_css(seq_css(sf)); + guard(spinlock)(&memcg->peaks_lock); return peak_show(sf, v, &memcg->memory); } @@ -5888,6 +5889,7 @@ static int swap_peak_show(struct seq_file *sf, void *v) { struct mem_cgroup *memcg = mem_cgroup_from_css(seq_css(sf)); + guard(spinlock)(&memcg->peaks_lock); return peak_show(sf, v, &memcg->swap); } From 963fd31707f3fe13586d38a2e1a6c094930c6b71 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sun, 30 Aug 2026 08:20:44 +0800 Subject: [PATCH 638/857] mm, memcg: fix memory.peak reset clobbering other fds' watermark MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Writing to memory.peak resets the peak for that fd only. Each fd is a watcher and reads back max(its own value, the shared local_watermark). peak_write() resets by lowering local_watermark to the current usage. To keep the other watchers' peaks it then walks the watcher list, but it stores the current usage into them instead of the old watermark. So once usage has dropped from a peak, a reset on one fd wrongly drags every other fd's peak down too, even fds that never reset. Reproduced on 7.2.0-rc5-next under QEMU, two fds A and B on one cgroup: B sees the peak (410624 KB), usage drops, then A resets -- and B's peak collapses to 1060 KB although B never reset. With this patch B keeps reading 410624 KB. Fix: save the old watermark before lowering it and use that to floor the other watchers, so a reset only affects the fd that issued it. Link: https://lore.kernel.org/20260830002044.1938621-3-ridong.chen@linux.dev Fixes: c6f53ed8f213 ("mm, memcg: cg2 memory{.swap,}.peak write handlers") Signed-off-by: Ridong Chen Closes: https://sashiko.dev/#/patchset/20260807090000.1532495-1-ridong.chen@linux.dev Assisted-by: Claude:claude-opus-4-8 Acked-by: Tao Cui Acked-by: Johannes Weiner Acked-by: Shakeel Butt Cc: David Finkel Cc: Michal Hocko Cc: Michal Koutný Cc: Muchun Song Cc: Roman Gushchin Cc: Tejun Heo Signed-off-by: Andrew Morton --- mm/memcontrol.c | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1ee974cb6d3727..d5ebe83eae3efc 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4775,7 +4775,7 @@ static ssize_t peak_write(struct kernfs_open_file *of, char *buf, size_t nbytes, loff_t off, struct page_counter *pc, struct list_head *watchers) { - unsigned long usage; + unsigned long usage, old_watermark; struct cgroup_of_peak *peer_ctx; struct mem_cgroup *memcg = mem_cgroup_from_css(of_css(of)); struct cgroup_of_peak *ofp = of_peak(of); @@ -4783,11 +4783,12 @@ static ssize_t peak_write(struct kernfs_open_file *of, char *buf, size_t nbytes, spin_lock(&memcg->peaks_lock); usage = page_counter_read(pc); + old_watermark = READ_ONCE(pc->local_watermark); WRITE_ONCE(pc->local_watermark, usage); list_for_each_entry(peer_ctx, watchers, list) - if (usage > peer_ctx->value) - WRITE_ONCE(peer_ctx->value, usage); + if (peer_ctx != ofp && old_watermark > peer_ctx->value) + WRITE_ONCE(peer_ctx->value, old_watermark); /* initial write, register watcher */ if (ofp->value == OFP_PEAK_UNSET) From 759dd7e21959327a73ed180cac20dbd17d7958e1 Mon Sep 17 00:00:00 2001 From: Eamon Sippy Date: Sat, 15 Aug 2026 10:32:45 +0000 Subject: [PATCH 639/857] mm: cma: make mm/cma.h self-contained and conditionalize includes mm/cma.h uses types from , , and without explicitly including them, violating the kernel header self-containment guidelines. and are also included unconditionally even though they are only needed under CONFIG_CMA_DEBUGFS and CONFIG_CMA_SYSFS respectively. Move the struct cma_kobject definition and inside the CONFIG_CMA_SYSFS block, and move inside CONFIG_CMA_DEBUGFS. Remove spurious trailing semicolons after the empty inline function bodies in the CONFIG_CMA_SYSFS #else branch. Add so that MAX_CMA_AREAS and CMA_MAX_NAME are always available when this header is included. Link: https://lore.kernel.org/20260815103246.5315-1-eamon112009@gmail.com Signed-off-by: Eamon Sippy Reviewed-by: Barry Song Signed-off-by: Andrew Morton --- mm/cma.h | 23 +++++++++++++++++------ 1 file changed, 17 insertions(+), 6 deletions(-) diff --git a/mm/cma.h b/mm/cma.h index ab6d39898ea52e..68e574b0f95187 100644 --- a/mm/cma.h +++ b/mm/cma.h @@ -2,14 +2,24 @@ #ifndef __MM_CMA_H__ #define __MM_CMA_H__ +#include #include +#include +#include +#include + +#ifdef CONFIG_CMA_DEBUGFS #include +#endif + +#ifdef CONFIG_CMA_SYSFS #include struct cma_kobject { struct kobject kobj; struct cma *cma; }; +#endif /* * Multi-range support. This can be useful if the size of the allocation @@ -38,10 +48,10 @@ struct cma_memrange { #define CMA_MAX_RANGES 8 struct cma { - unsigned long count; - unsigned long available_count; + unsigned long count; + unsigned long available_count; unsigned int order_per_bit; /* Order of pages represented by one bit */ - spinlock_t lock; + spinlock_t lock; struct mutex alloc_mutex; #ifdef CONFIG_CMA_DEBUGFS struct hlist_head mem_head; @@ -87,10 +97,11 @@ void cma_sysfs_account_fail_pages(struct cma *cma, unsigned long nr_pages); void cma_sysfs_account_release_pages(struct cma *cma, unsigned long nr_pages); #else static inline void cma_sysfs_account_success_pages(struct cma *cma, - unsigned long nr_pages) {}; + unsigned long nr_pages) {} static inline void cma_sysfs_account_fail_pages(struct cma *cma, - unsigned long nr_pages) {}; + unsigned long nr_pages) {} static inline void cma_sysfs_account_release_pages(struct cma *cma, - unsigned long nr_pages) {}; + unsigned long nr_pages) {} #endif + #endif From 87aa108eb4aa27580f987edc17792de63ecf9afc Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Thu, 13 Aug 2026 17:38:09 +0800 Subject: [PATCH 640/857] mm/oom_kill: remove unreachable __GFP_THISNODE check in constrained_alloc() The __GFP_THISNODE check in constrained_alloc() is dead code: global OOM is never triggered with __GFP_THISNODE (blocked in __alloc_pages_may_oom before out_of_memory() is called), and memcg OOM returns CONSTRAINT_MEMCG at the top of the function before reaching this point. Remove the check, its stale comment, and update the following comment that referenced __GFP_THISNODE. Link: https://lore.kernel.org/20260813093810.573302-1-ye.liu@linux.dev Signed-off-by: Ye Liu Acked-by: Michal Hocko Acked-by: Shakeel Butt Cc: David Rientjes Cc: Liu Ye Signed-off-by: Andrew Morton --- mm/oom_kill.c | 13 +++---------- 1 file changed, 3 insertions(+), 10 deletions(-) diff --git a/mm/oom_kill.c b/mm/oom_kill.c index 5f372f6e26fa32..fd3c476846a302 100644 --- a/mm/oom_kill.c +++ b/mm/oom_kill.c @@ -267,18 +267,11 @@ static enum oom_constraint constrained_alloc(struct oom_control *oc) if (!oc->zonelist) return CONSTRAINT_NONE; - /* - * Reach here only when __GFP_NOFAIL is used. So, we should avoid - * to kill current.We have to random task kill in this case. - * Hopefully, CONSTRAINT_THISNODE...but no way to handle it, now. - */ - if (oc->gfp_mask & __GFP_THISNODE) - return CONSTRAINT_NONE; /* - * This is not a __GFP_THISNODE allocation, so a truncated nodemask in - * the page allocator means a mempolicy is in effect. Cpuset policy - * is enforced in get_page_from_freelist(). + * A truncated nodemask in the page allocator means a mempolicy + * is in effect. Cpuset policy is enforced in + * get_page_from_freelist(). */ if (oc->nodemask && !nodes_subset(node_states[N_MEMORY], *oc->nodemask)) { From 569f041bdc7ba1c462d18769a992b1b87cd50926 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Tue, 11 Aug 2026 11:36:08 +0800 Subject: [PATCH 641/857] mm/oom_kill, proc: replace magic number 1000 with OOM_SCORE_ADJ_MAX In oom_badness() and proc_oom_score(), the oom_score_adj normalization uses a hardcoded 1000, which is the value of OOM_SCORE_ADJ_MAX defined in include/uapi/linux/oom.h. Other code in the kernel (e.g. fs/proc/base.c oom_adj handling) already uses OOM_SCORE_ADJ_MAX for the same purpose. Replace the magic number with the macro for consistency and readability. No functional change. Link: https://lore.kernel.org/20260811033609.3992348-1-ye.liu@linux.dev Signed-off-by: Ye Liu Acked-by: Michal Hocko Cc: David Rientjes Cc: Liu Ye Cc: Shakeel Butt Cc: Song Hu Signed-off-by: Andrew Morton --- fs/proc/base.c | 3 ++- mm/oom_kill.c | 2 +- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/fs/proc/base.c b/fs/proc/base.c index 6a39de424f62a1..58be3894246054 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -594,7 +594,8 @@ static int proc_oom_score(struct seq_file *m, struct pid_namespace *ns, * exporting for a long time so userspace might depend on it. */ if (badness != LONG_MIN) - points = (1000 + badness * 1000 / (long)totalpages) * 2 / 3; + points = (OOM_SCORE_ADJ_MAX + + badness * OOM_SCORE_ADJ_MAX / (long)totalpages) * 2 / 3; seq_printf(m, "%lu\n", points); diff --git a/mm/oom_kill.c b/mm/oom_kill.c index fd3c476846a302..5d48bd862c27b2 100644 --- a/mm/oom_kill.c +++ b/mm/oom_kill.c @@ -230,7 +230,7 @@ long oom_badness(struct task_struct *p, unsigned long totalpages) task_unlock(p); /* Normalize to oom_score_adj units */ - adj *= totalpages / 1000; + adj *= totalpages / OOM_SCORE_ADJ_MAX; points += adj; return points; From b1a1a14e55ec6d4d09b0f81fbcd26069a9bcb318 Mon Sep 17 00:00:00 2001 From: Pedro Falcato Date: Tue, 11 Aug 2026 18:21:55 +0100 Subject: [PATCH 642/857] mm: replace custom bad page map ratelimiting logic Patch series "mm: replace custom ratelimiting logic". The kernel has a perfectly cromulent and mostly-equivalent variant in lib/ratelimit.c that can be used. This patch (of 2): The current logic (allow up to $BURST prints per minute) can be entirely replaced by the generic version in lib/ratelimit.c, used around the kernel. Do so. The only functional difference should be that the new logs will read something like: KERN_WARNING "print_bad_page_map: %d callbacks suppressed\n", ... But that should be fine enough. Link: https://lore.kernel.org/20260811172156.356053-1-pfalcato@suse.de Link: https://lore.kernel.org/20260811172156.356053-2-pfalcato@suse.de Signed-off-by: Pedro Falcato Acked-by: Johannes Weiner Acked-by: Zi Yan Reviewed-by: SJ Park Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Acked-by: Mike Rapoport (Microsoft) Cc: Brendan Jackman Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan Signed-off-by: Andrew Morton --- mm/memory.c | 30 +++--------------------------- 1 file changed, 3 insertions(+), 27 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 347db2acd0f83b..09ac784f8b7b39 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -492,32 +492,8 @@ static inline void add_mm_rss_vec(struct mm_struct *mm, int *rss) add_mm_counter(mm, i, rss[i]); } -static bool is_bad_page_map_ratelimited(void) -{ - static unsigned long resume; - static unsigned long nr_shown; - static unsigned long nr_unshown; - - /* - * Allow a burst of 60 reports, then keep quiet for that minute; - * or allow a steady drip of one report per second. - */ - if (nr_shown == 60) { - if (time_before(jiffies, resume)) { - nr_unshown++; - return true; - } - if (nr_unshown) { - pr_alert("BUG: Bad page map: %lu messages suppressed\n", - nr_unshown); - nr_unshown = 0; - } - nr_shown = 0; - } - if (nr_shown++ == 0) - resume = jiffies + 60 * HZ; - return false; -} +/* Allow a burst of 60 bad page map reports per minute. */ +static DEFINE_RATELIMIT_STATE(bad_page_map_ratelimit, 60 * HZ, 60); static void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size) { @@ -633,7 +609,7 @@ static void print_bad_page_map(struct vm_area_struct *vma, char entry_str[PTVAL_STR_MAX]; pgoff_t index, anon_index; - if (is_bad_page_map_ratelimited()) + if (!__ratelimit(&bad_page_map_ratelimit)) return; mapping = vma->vm_file ? vma->vm_file->f_mapping : NULL; From 29bc0b5d617b8ba5d137bd19f8bfd6dd48e94d16 Mon Sep 17 00:00:00 2001 From: Pedro Falcato Date: Tue, 11 Aug 2026 18:21:56 +0100 Subject: [PATCH 643/857] mm/page_alloc: replace custom bad page ratelimiting logic The current logic (allow up to $BURST prints per minute) can be entirely replaced by the generic version in lib/ratelimit.c, used around the kernel. Do so. The only functional difference should be that the new logs will read something like: KERN_WARNING "bad_page: %d callbacks suppressed\n", ... But that should be fine enough. Link: https://lore.kernel.org/20260811172156.356053-3-pfalcato@suse.de Signed-off-by: Pedro Falcato Acked-by: Johannes Weiner Acked-by: Zi Yan Reviewed-by: SJ Park Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Acked-by: Mike Rapoport (Microsoft) Cc: Brendan Jackman Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan Signed-off-by: Andrew Morton --- mm/page_alloc.c | 28 +++++----------------------- 1 file changed, 5 insertions(+), 23 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 146f7e0a9462a2..c4dc61ec663eea 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -613,31 +613,13 @@ static inline bool __maybe_unused bad_range(struct zone *zone, struct page *page } #endif +/* Allow a burst of 60 reports per minute */ +static DEFINE_RATELIMIT_STATE(bad_page_ratelimit, 60 * HZ, 60); + static void bad_page(struct page *page, const char *reason) { - static unsigned long resume; - static unsigned long nr_shown; - static unsigned long nr_unshown; - - /* - * Allow a burst of 60 reports, then keep quiet for that minute; - * or allow a steady drip of one report per second. - */ - if (nr_shown == 60) { - if (time_before(jiffies, resume)) { - nr_unshown++; - goto out; - } - if (nr_unshown) { - pr_alert( - "BUG: Bad page state: %lu messages suppressed\n", - nr_unshown); - nr_unshown = 0; - } - nr_shown = 0; - } - if (nr_shown++ == 0) - resume = jiffies + 60 * HZ; + if (!__ratelimit(&bad_page_ratelimit)) + goto out; pr_alert("BUG: Bad page state in process %s pfn:%05lx\n", current->comm, page_to_pfn(page)); From c9e49207314b2769f7b0dd05ec88725bcc530347 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Mon, 31 Aug 2026 14:42:11 +0100 Subject: [PATCH 644/857] mm/vmpressure: remove window size TODO There has been a steady stream of patches that have been submitted by newcomers to core mm 'fixing' this TODO, with all but the original having very likely been generated by LLMs. It appears that TODOs to LLMs are like red rags to a bull. In addition, TODOs in code often bitrot and are distracting - those who understand the code know what could be improved in future. Therefore remove the TODO. The work required to actually fix this TODO requires somebody who both has understanding of the code and significant real-world data to back their changes. Such a person doesn't require a TODO prompt to implement this change, so nothing of value is being lost here. Link: https://lore.kernel.org/all/20260831130316.448-1-tahasezer.is@gmail.com/ Link: https://lore.kernel.org/linux-mm/20260724054305.516126-1-cui.tao@linux.dev/ Link: https://lore.kernel.org/linux-mm/20260715143646.15828-1-gaikwad.dcg@gmail.com/ Link: https://lore.kernel.org/all/20260227221555.29969-1-mcq@disroot.org/ Link: https://lore.kernel.org/20260831-remove-vmpressure-todo-v1-1-498515e59cdf@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Signed-off-by: Andrew Morton --- mm/vmpressure.c | 3 --- 1 file changed, 3 deletions(-) diff --git a/mm/vmpressure.c b/mm/vmpressure.c index 9629240d77adc7..3de99fef392894 100644 --- a/mm/vmpressure.c +++ b/mm/vmpressure.c @@ -30,9 +30,6 @@ * * As the vmscan reclaimer logic works with chunks which are multiple of * SWAP_CLUSTER_MAX, it makes sense to use it for the window size as well. - * - * TODO: Make the window size depend on machine size, as we do for vmstat - * thresholds. Currently we set it to 512 pages (2MB for 4KB pages). */ const unsigned long vmpressure_win = SWAP_CLUSTER_MAX * 16; From 28c334d6531623a7814c68dd7ce3f8ce168310b4 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Mon, 31 Aug 2026 15:18:23 +0100 Subject: [PATCH 645/857] tools/testing/selftests/mm: add missing .gitignore entries Commit 2bee308f3adb ("selftests/mm: use pattern matching in .gitignore") switched to a pattern-matching mechanism to reduce churn in .gitignore. It however accidentally excluded the page_frag test's-generated module intermediate C file with .mod.c extension, and also the local_config.h header generated if liburing is available locally. Explicitly fix both the issues, fixing the module-generated C file as a general pattern as these are always intermediate files that should be ignored. Since this is a trivial .gitignore change it doesn't seem necessary to treat it as a hotfix. Link: https://lore.kernel.org/20260831-fix-mm-selftests-gitignore-v1-1-c984bbd4c5e4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Reviewed-by: Gregory Price (Meta) Reviewed-by: Sarthak Sharma Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/.gitignore | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/mm/.gitignore b/tools/testing/selftests/mm/.gitignore index fcd892ed21e32c..a306d775478690 100644 --- a/tools/testing/selftests/mm/.gitignore +++ b/tools/testing/selftests/mm/.gitignore @@ -2,7 +2,9 @@ * !/**/ !*.c +*.mod.c !*.h +local_config.h !*.sh !.gitignore !Makefile From 4b67e50e62979643460c7abab1fd12b4e7597e28 Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:52 -0700 Subject: [PATCH 646/857] mm: make per-VMA locks available universally MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm: Unconditional per-VMA locks and cleanups", v7. tl;dr: Make per-VMA locks available in all configs. Simplify some of the per-VMA lock users now that they can rely on them being always available. Binder and networking folks: Your code is the target of the cleanups. I'm cc'ing you now on v2 because there's emerging consensus on the mm side that the approach here is sane. I'm not quite sure how this pile would get merged, but ack/review tags would be appreciated if this looks good to you. Longer version: When working on some x86 shadow stack code, it was a real pain to avoid causing recursive locking problems with mmap_lock. One way to avoid those was to avoid mmap_lock and use per-VMA locks instead. They are great, but they are not available in all configs which makes them unusable in generic code, or if you want to completely avoid mmap_lock. Make per-VMA locks available in all configs. Right now, they are only available on select architectures when SMP and MMU are enabled. But all of the primitives that per-VMA locks are built on (RCU, maple trees, refcounts) work just fine without SMP or MMU. The only real downside is that making VMAs a wee bit bigger on !MMU and !SMP builds. The upside is much cleaner code, lower complexity and less #ifdeffery. Clean up a binder VMA locking site now that it can rely on per-VMA locks. Building on top of universally-available per-VMA locks, introduce a new helper. Since the new API does not require callers to have a fallback to mmap_lock, it's much easier to use. Callers can potentially replace this very common kernel idiom: mmap_read_lock(mm); vma = vma_lookup() // fiddle with vma mmap_read_unlock(mm); with: vma = vma_start_read_unlocked(mm, address); // fiddle with vma vma_end_read(vma); Which avoids mmap_lock entirely in the fast path. Use that new API for another binder site and one in the TCP code. This patch (of 7): The per-VMA locks have been around for several years. They've had some bugs worked out of them and have seen quite wide use. However, they are still only available when architectures explicitly enable them. Remove the conditional compilation around the per-VMA locks, making them available on all architectures and configs. The approach up to now seemed to be to add ARCH_SUPPORTS_PER_VMA_LOCK when the architecture started using per-VMA locks in the fault handler. But, contrary to the naming, the Kconfig option does not really indicate whether the architecture supports per-VMA locks or not. It is more of a marker for whether the architecture is likely to benefit from per-VMA locks. To me, the most important thing side-effect of universal availability is letting per-VMA locks be used in SMP=n configs. This lets us use per-VMA locking in all x86 code without fallbacks. Overall, this just generally makes the kernel simpler. Just look at the diffstat. It also opens the door to users that want to use the per-VMA locks in common code. Doing *that* brings additional simplifications. The downside of this is adding some fields to vm_area_struct and mm_struct. There are likely ways to optimize this, especially for things like SMP=n configs. For now, do the simplest thing: use the same implementation everywhere. == Considerations for NOMMU config == NOMMU systems do not write-lock VMAs, therefore read-locking a VMA would always succeed unless VMA is detached. Therefore for NOMMU config we make vma_mark_attached() a NOOP, which keeps VMAs always in detached state. This causes VMA read-locking to always fail and the caller falls back to locking mmap_lock. The following functions will have a different implementation in NOMMU config: - vma_mark_attached(), vma_mark_detached() are made NOOPs, keeping VMAs always in a detached state and preventing assertions and refcount underflows; - vma_start_write(), vma_start_write_killable() are made NOOPs to avoid warnings in __vma_start_write() due to VMAs being detached. These functions are not used in NOMMU code but __vma_start_write() is an exported function, therefore might be used by drivers. - vma_assert_attached() is made NOOP because it's reachable from NOMMU code via split_vma()->vma_iter_store_new()->vma_iter_store_overwrite(); - vma_assert_write_locked() is asserting vma->vm_mm is write-locked, as was done before this change; - vma_assert_locked() is asserting vma->vm_mm is locked, as was done before this change; The following functions work for both MMU and NOMMU configs: - vma_lock_init() performs the same initialization as for MMU config; - mm_lock_seqcount_init(), mm_lock_seqcount_begin(), mm_lock_seqcount_end() are called from mmap_write_{lock|unlock} and update mm_lock_seq correctly. - mmap_lock_speculate_try_begin(), mmap_lock_speculate_retry() work as is because mm_lock_seq is updated correctly; - vma_start_read(), vma_start_read_locked() will always fail because VMAs are always detached; - vma_end_read() will never be called because vma_start_read() never succeeds; - vma_is_attached() always return false because VMAs are always detached; - vma_assert_detached() will never trigger because VMAs are never attached; - vma_start_read_locked() always return false because VMAs are always detached; - lock_vma_under_rcu() will be safe as the attempted read lock will bail; Changes in the following files are not affecting NOMMU config: task_mmu.c - not compiled when CONFIG_MMU=n; pagewalk.c - not compiled when CONFIG_MMU=n; userfaultfd.c - not compiled when CONFIG_MMU=n (CONFIG_USERFAULTFD depends on CONFIG_MMU); The following changes in the BPF code are made to keep NOMMU config working like before: stack_map_lock_vma() - keeps mmap_lock in NOMMU config; bpf_iter_task_vma_new() - bails out in NOMMU config; Link: https://lore.kernel.org/20260831203056.838265-1-surenb@google.com Link: https://lore.kernel.org/20260831203056.838265-2-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Cc: Liam R. Howlett Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Signed-off-by: Andrew Morton --- arch/arm/Kconfig | 1 - arch/arm64/Kconfig | 1 - arch/loongarch/Kconfig | 1 - arch/powerpc/platforms/powernv/Kconfig | 1 - arch/powerpc/platforms/pseries/Kconfig | 1 - arch/riscv/Kconfig | 1 - arch/s390/Kconfig | 1 - arch/x86/Kconfig | 2 - fs/proc/internal.h | 2 - fs/proc/task_mmu.c | 93 -------------------------- include/linux/mm.h | 12 ---- include/linux/mm_types.h | 8 +-- include/linux/mmap_lock.h | 75 +++++++-------------- kernel/bpf/stackmap.c | 17 ++--- kernel/bpf/task_iter.c | 2 +- kernel/fork.c | 2 - mm/Kconfig | 12 ---- mm/Kconfig.debug | 1 - mm/debug.c | 4 -- mm/init-mm.c | 2 - mm/memory.c | 2 - mm/mmap_lock.c | 26 +------ mm/pagewalk.c | 2 - mm/rmap.c | 2 - mm/userfaultfd.c | 55 --------------- rust/kernel/mm.rs | 32 +++------ tools/testing/vma/include/dup.h | 5 +- tools/testing/vma/vma_internal.h | 1 - 28 files changed, 48 insertions(+), 316 deletions(-) diff --git a/arch/arm/Kconfig b/arch/arm/Kconfig index ffbc7f38613151..408aa58a2a5bbc 100644 --- a/arch/arm/Kconfig +++ b/arch/arm/Kconfig @@ -42,7 +42,6 @@ config ARM select ARCH_SUPPORTS_ATOMIC_RMW select ARCH_SUPPORTS_CFI select ARCH_SUPPORTS_HUGETLBFS if ARM_LPAE - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_RT select ARCH_USE_BUILTIN_BSWAP select ARCH_USE_CMPXCHG_LOCKREF diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index b5a51b0ef9440a..2bbeded33da0da 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -81,7 +81,6 @@ config ARM64 select ARCH_HAS_PTE_PROTNONE select ARCH_SUPPORTS_NUMA_BALANCING select ARCH_SUPPORTS_PAGE_TABLE_CHECK - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_HUGE_PFNMAP if TRANSPARENT_HUGEPAGE select ARCH_SUPPORTS_RT select ARCH_SUPPORTS_SCHED_SMT diff --git a/arch/loongarch/Kconfig b/arch/loongarch/Kconfig index a21f51e5815e96..9c5def7062222b 100644 --- a/arch/loongarch/Kconfig +++ b/arch/loongarch/Kconfig @@ -69,7 +69,6 @@ config LOONGARCH select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS select ARCH_HAS_PTE_PROTNONE if 64BIT select ARCH_SUPPORTS_NUMA_BALANCING if NUMA - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_RT select ARCH_SUPPORTS_SCHED_SMT if SMP select ARCH_SUPPORTS_SCHED_MC if SMP diff --git a/arch/powerpc/platforms/powernv/Kconfig b/arch/powerpc/platforms/powernv/Kconfig index b5ad7c173ef0c1..dd8f6060fb7a2e 100644 --- a/arch/powerpc/platforms/powernv/Kconfig +++ b/arch/powerpc/platforms/powernv/Kconfig @@ -17,7 +17,6 @@ config PPC_POWERNV select PPC_DOORBELL select MMU_NOTIFIER select FORCE_SMP - select ARCH_SUPPORTS_PER_VMA_LOCK select PPC_RADIX_BROADCAST_TLBIE if PPC_RADIX_MMU default y diff --git a/arch/powerpc/platforms/pseries/Kconfig b/arch/powerpc/platforms/pseries/Kconfig index 74910ce3a541c3..7d125e288f6ef7 100644 --- a/arch/powerpc/platforms/pseries/Kconfig +++ b/arch/powerpc/platforms/pseries/Kconfig @@ -23,7 +23,6 @@ config PPC_PSERIES select HOTPLUG_CPU select FORCE_SMP select SWIOTLB - select ARCH_SUPPORTS_PER_VMA_LOCK select PPC_RADIX_BROADCAST_TLBIE if PPC_RADIX_MMU default y diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index f8e26c4bed2bae..505eed4af932bd 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -72,7 +72,6 @@ config RISCV select ARCH_SUPPORTS_LTO_CLANG_THIN select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS if 64BIT && MMU select ARCH_SUPPORTS_PAGE_TABLE_CHECK if MMU - select ARCH_SUPPORTS_PER_VMA_LOCK if MMU select ARCH_HAS_PTE_PROTNONE if MMU select ARCH_SUPPORTS_RT select ARCH_SUPPORTS_SHADOW_CALL_STACK if HAVE_SHADOW_CALL_STACK diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index 4b51bc6e8948d7..b88b8504213692 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -156,7 +156,6 @@ config S390 select ARCH_HAS_PTE_PROTNONE select ARCH_SUPPORTS_NUMA_BALANCING select ARCH_SUPPORTS_PAGE_TABLE_CHECK - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_USES_CFI_GENERIC_LLVM_PASS if CC_IS_CLANG select ARCH_USE_BUILTIN_BSWAP select ARCH_USE_CMPXCHG_LOCKREF diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index 7aa74bcc72f9db..a8c3b3d31a2761 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -27,7 +27,6 @@ config X86_64 select ARCH_HAS_GIGANTIC_PAGE select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS select ARCH_SUPPORTS_INT128 if CC_HAS_INT128 - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_HUGE_PFNMAP if TRANSPARENT_HUGEPAGE select HAVE_ARCH_SOFT_DIRTY select MODULES_USE_ELF_RELA @@ -1848,7 +1847,6 @@ config X86_USER_SHADOW_STACK bool "X86 userspace shadow stack" depends on AS_WRUSS depends on X86_64 - depends on PER_VMA_LOCK select ARCH_USES_HIGH_VMA_FLAGS select ARCH_HAS_USER_SHADOW_STACK select X86_CET diff --git a/fs/proc/internal.h b/fs/proc/internal.h index 04bd6c9e65a722..623bb43ede5509 100644 --- a/fs/proc/internal.h +++ b/fs/proc/internal.h @@ -385,10 +385,8 @@ struct mem_size_stats; struct proc_maps_locking_ctx { struct mm_struct *mm; -#ifdef CONFIG_PER_VMA_LOCK bool mmap_locked; struct vm_area_struct *locked_vma; -#endif }; struct proc_maps_private { diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 5c54aebe211824..e671b4fd8dedd9 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -130,8 +130,6 @@ static void release_task_mempolicy(struct proc_maps_private *priv) } #endif -#ifdef CONFIG_PER_VMA_LOCK - static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) { int ret = mmap_read_lock_killable(lock_ctx->mm); @@ -233,46 +231,6 @@ static inline void reacquire_rcu(struct proc_maps_private *priv) vma_iter_set(&priv->iter, priv->lock_ctx.locked_vma->vm_end); } -#else /* CONFIG_PER_VMA_LOCK */ - -static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) -{ - return mmap_read_lock_killable(lock_ctx->mm); -} - -static inline void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) -{ - mmap_read_unlock(lock_ctx->mm); -} - -static inline bool lock_vma_range(struct seq_file *m, - struct proc_maps_locking_ctx *lock_ctx) -{ - return lock_ctx_mm(lock_ctx) == 0; -} - -static inline void unlock_vma_range(struct proc_maps_locking_ctx *lock_ctx) -{ - unlock_ctx_mm(lock_ctx); -} - -static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv, - loff_t last_pos) -{ - return vma_next(&priv->iter); -} - -static inline bool fallback_to_mmap_lock(struct proc_maps_private *priv, - loff_t pos) -{ - return false; -} - -static inline void drop_rcu(struct proc_maps_private *priv) {} -static inline void reacquire_rcu(struct proc_maps_private *priv) {} - -#endif /* CONFIG_PER_VMA_LOCK */ - static struct vm_area_struct *proc_get_vma(struct seq_file *m, loff_t *ppos) { struct proc_maps_private *priv = m->private; @@ -560,8 +518,6 @@ static int pid_maps_open(struct inode *inode, struct file *file) PROCMAP_QUERY_VMA_FLAGS \ ) -#ifdef CONFIG_PER_VMA_LOCK - static int query_vma_setup(struct proc_maps_locking_ctx *lock_ctx) { reset_lock_ctx(lock_ctx); @@ -612,26 +568,6 @@ static struct vm_area_struct *query_vma_find_by_addr(struct proc_maps_locking_ct return vma; } -#else /* CONFIG_PER_VMA_LOCK */ - -static int query_vma_setup(struct proc_maps_locking_ctx *lock_ctx) -{ - return mmap_read_lock_killable(lock_ctx->mm); -} - -static void query_vma_teardown(struct proc_maps_locking_ctx *lock_ctx) -{ - mmap_read_unlock(lock_ctx->mm); -} - -static struct vm_area_struct *query_vma_find_by_addr(struct proc_maps_locking_ctx *lock_ctx, - unsigned long addr) -{ - return find_vma(lock_ctx->mm, addr); -} - -#endif /* CONFIG_PER_VMA_LOCK */ - static struct vm_area_struct *query_matching_vma(struct proc_maps_locking_ctx *lock_ctx, unsigned long addr, u32 flags) { @@ -1314,8 +1250,6 @@ static const struct mm_walk_ops smaps_shmem_walk_ops = { .walk_lock = PGWALK_RDLOCK, }; -#ifdef CONFIG_PER_VMA_LOCK - static const struct mm_walk_ops smaps_walk_vma_lock_ops = { .pmd_entry = smaps_pte_range, .hugetlb_entry = smaps_hugetlb_range, @@ -1345,22 +1279,6 @@ get_smaps_shmem_walk_ops(struct proc_maps_private *priv) return &smaps_shmem_walk_vma_lock_ops; } -#else /* CONFIG_PER_VMA_LOCK */ - -static inline const struct mm_walk_ops * -get_smaps_walk_ops(struct proc_maps_private *priv) -{ - return &smaps_walk_ops; -} - -static inline const struct mm_walk_ops * -get_smaps_shmem_walk_ops(struct proc_maps_private *priv) -{ - return &smaps_shmem_walk_ops; -} - -#endif /* CONFIG_PER_VMA_LOCK */ - /* * Gather mem stats from @vma with the indicated beginning * address @start, and keep them in @mss. @@ -3497,7 +3415,6 @@ static const struct mm_walk_ops show_numa_ops = { .walk_lock = PGWALK_RDLOCK, }; -#ifdef CONFIG_PER_VMA_LOCK static const struct mm_walk_ops show_numa_vma_lock_ops = { .hugetlb_entry = gather_hugetlb_stats, .pmd_entry = gather_pte_stats, @@ -3512,16 +3429,6 @@ get_show_numa_ops(struct proc_maps_private *priv) return &show_numa_vma_lock_ops; } -#else /* CONFIG_PER_VMA_LOCK */ - -static inline const struct mm_walk_ops * -get_show_numa_ops(struct proc_maps_private *priv) -{ - return &show_numa_ops; -} - -#endif /* CONFIG_PER_VMA_LOCK */ - /* * Display pages allocated per node and memory policy via /proc. */ diff --git a/include/linux/mm.h b/include/linux/mm.h index b19711b6dbc69a..a9fbe26536f450 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -928,7 +928,6 @@ static inline void vma_numab_state_free(struct vm_area_struct *vma) {} * These must be here rather than mmap_lock.h as dependent on vm_fault type, * declared in this header. */ -#ifdef CONFIG_PER_VMA_LOCK static inline void release_fault_lock(struct vm_fault *vmf) { if (vmf->flags & FAULT_FLAG_VMA_LOCK) @@ -944,17 +943,6 @@ static inline void assert_fault_locked(const struct vm_fault *vmf) else mmap_assert_locked(vmf->vma->vm_mm); } -#else -static inline void release_fault_lock(struct vm_fault *vmf) -{ - mmap_read_unlock(vmf->vma->vm_mm); -} - -static inline void assert_fault_locked(const struct vm_fault *vmf) -{ - mmap_assert_locked(vmf->vma->vm_mm); -} -#endif /* CONFIG_PER_VMA_LOCK */ static inline bool mm_flags_test(int flag, const struct mm_struct *mm) { diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 6d815f6440c94e..5413bd10fff2c2 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -950,7 +950,6 @@ struct vm_area_struct { vma_flags_t flags; }; -#ifdef CONFIG_PER_VMA_LOCK /* * Can only be written (using WRITE_ONCE()) while holding both: * - mmap_lock (in write mode) @@ -966,7 +965,7 @@ struct vm_area_struct { * slowpath. */ unsigned int vm_lock_seq; -#endif + /* * Low 32-bits of anonymous page offset. * See vma_start_anon_pgoff() comment for details. @@ -1003,7 +1002,6 @@ struct vm_area_struct { #ifdef CONFIG_NUMA_BALANCING struct vma_numab_state *numab_state; /* NUMA Balancing state */ #endif -#ifdef CONFIG_PER_VMA_LOCK /* * Used to keep track of firstly, whether the VMA is attached, secondly, * if attached, how many read locks are taken, and thirdly, if the @@ -1046,7 +1044,6 @@ struct vm_area_struct { #ifdef CONFIG_DEBUG_LOCK_ALLOC struct lockdep_map vmlock_dep_map; #endif -#endif #ifdef CONFIG_64BIT /* * High 32-bits of anonymous page offset. @@ -1254,7 +1251,6 @@ struct mm_struct { * init_mm.mmlist, and are protected * by mmlist_lock */ -#ifdef CONFIG_PER_VMA_LOCK struct rcuwait vma_writer_wait; /* * This field has lock-like semantics, meaning it is sometimes @@ -1274,7 +1270,7 @@ struct mm_struct { * mmap_lock. */ seqcount_t mm_lock_seq; -#endif + struct futex_mm_data futex; unsigned long hiwater_rss; /* High-watermark of RSS usage */ diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index bec0eab6ef035d..27c00bac29f9e1 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -76,8 +76,6 @@ static inline void mmap_assert_write_locked(const struct mm_struct *mm) rwsem_assert_held_write(&mm->mmap_lock); } -#ifdef CONFIG_PER_VMA_LOCK - #ifdef CONFIG_LOCKDEP #define __vma_lockdep_map(vma) (&vma->vmlock_dep_map) #else @@ -297,6 +295,9 @@ int __vma_start_write(struct vm_area_struct *vma, int state); */ static inline void vma_start_write(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + if (__is_vma_write_locked(vma)) return; @@ -319,6 +320,9 @@ static inline void vma_start_write(struct vm_area_struct *vma) static inline __must_check int vma_start_write_killable(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return 0; + if (__is_vma_write_locked(vma)) return 0; @@ -331,6 +335,11 @@ int vma_start_write_killable(struct vm_area_struct *vma) */ static inline void vma_assert_write_locked(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) { + mmap_assert_write_locked(vma->vm_mm); + return; + } + VM_WARN_ON_ONCE_VMA(!__is_vma_write_locked(vma), vma); } @@ -343,6 +352,11 @@ static inline void vma_assert_locked(struct vm_area_struct *vma) { unsigned int refcnt; + if (!IS_ENABLED(CONFIG_MMU)) { + mmap_assert_locked(vma->vm_mm); + return; + } + if (IS_ENABLED(CONFIG_LOCKDEP)) { if (!lock_is_held(__vma_lockdep_map(vma))) vma_assert_write_locked(vma); @@ -432,6 +446,9 @@ static inline bool vma_is_attached(struct vm_area_struct *vma) */ static inline void vma_assert_attached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + WARN_ON_ONCE(!vma_is_attached(vma)); } @@ -442,6 +459,9 @@ static inline void vma_assert_detached(struct vm_area_struct *vma) static inline void vma_mark_attached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + vma_assert_write_locked(vma); vma_assert_detached(vma); refcount_set_release(&vma->vm_refcnt, 1); @@ -451,6 +471,9 @@ void __vma_exclude_readers_for_detach(struct vm_area_struct *vma); static inline void vma_mark_detached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + vma_assert_write_locked(vma); vma_assert_attached(vma); @@ -484,54 +507,6 @@ struct vm_area_struct *lock_next_vma(struct mm_struct *mm, struct vma_iterator *iter, unsigned long address); -#else /* CONFIG_PER_VMA_LOCK */ - -static inline void mm_lock_seqcount_init(struct mm_struct *mm) {} -static inline void mm_lock_seqcount_begin(struct mm_struct *mm) {} -static inline void mm_lock_seqcount_end(struct mm_struct *mm) {} - -static inline bool mmap_lock_speculate_try_begin(struct mm_struct *mm, unsigned int *seq) -{ - return false; -} - -static inline bool mmap_lock_speculate_retry(struct mm_struct *mm, unsigned int seq) -{ - return true; -} -static inline void vma_lock_init(struct vm_area_struct *vma, bool reset_refcnt) {} -static inline void vma_end_read(struct vm_area_struct *vma) {} -static inline void vma_start_write(struct vm_area_struct *vma) {} -static inline __must_check -int vma_start_write_killable(struct vm_area_struct *vma) { return 0; } -static inline void vma_assert_write_locked(struct vm_area_struct *vma) - { mmap_assert_write_locked(vma->vm_mm); } -static inline bool vma_is_attached(struct vm_area_struct *vma) - { return true; } -static inline void vma_assert_attached(struct vm_area_struct *vma) {} -static inline void vma_assert_detached(struct vm_area_struct *vma) {} -static inline void vma_mark_attached(struct vm_area_struct *vma) {} -static inline void vma_mark_detached(struct vm_area_struct *vma) {} - -static inline struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, - unsigned long address) -{ - return NULL; -} - -static inline void vma_assert_locked(struct vm_area_struct *vma) -{ - mmap_assert_locked(vma->vm_mm); -} - -static inline void vma_assert_stabilised(struct vm_area_struct *vma) -{ - /* If no VMA locks, then either mmap lock suffices to stabilise. */ - mmap_assert_locked(vma->vm_mm); -} - -#endif /* CONFIG_PER_VMA_LOCK */ - static inline void vma_assert_can_modify(struct vm_area_struct *vma) { if (vma_is_attached(vma)) diff --git a/kernel/bpf/stackmap.c b/kernel/bpf/stackmap.c index a839041e0d0082..8fb70c2a9c8e6a 100644 --- a/kernel/bpf/stackmap.c +++ b/kernel/bpf/stackmap.c @@ -272,13 +272,10 @@ struct stack_map_vma_lock { /* * Acquire a stable read-side reference on the VMA covering @ip. * - * With CONFIG_PER_VMA_LOCK=y this returns a VMA with its per-VMA read - * lock held and mmap_lock dropped, so the caller may sleep. - * - * With CONFIG_PER_VMA_LOCK=n it returns a VMA with mmap_lock still - * held; the caller must snapshot any fields it needs and pin vm_file - * with get_file() before stack_map_unlock_vma() drops mmap_lock, as - * the VMA may be split, merged, or freed after that. + * On NOMMU configurations, returns with the mmap_lock held. If the MMU + * is enabled, the per-VMA lock will be held instead. The lock + * should be released with stack_map_unlock_vma() which will release the + * appropriate lock. Once the lock is released, the VMA may be freed. * * Returns NULL on failure, in which case no lock is held. */ @@ -288,7 +285,6 @@ stack_map_lock_vma(struct stack_map_vma_lock *lock, unsigned long ip) struct mm_struct *mm = lock->mm; struct vm_area_struct *vma; - /* noop under !CONFIG_PER_VMA_LOCK */ vma = lock_vma_under_rcu(mm, ip); if (vma) { lock->vma = vma; @@ -308,21 +304,20 @@ stack_map_lock_vma(struct stack_map_vma_lock *lock, unsigned long ip) return NULL; } -#ifdef CONFIG_PER_VMA_LOCK +#ifdef CONFIG_MMU if (!vma_start_read_locked(vma)) { mmap_read_unlock(mm); return NULL; } mmap_read_unlock(mm); #endif - lock->vma = vma; return vma; } static void stack_map_unlock_vma(struct stack_map_vma_lock *lock) { -#ifdef CONFIG_PER_VMA_LOCK +#ifdef CONFIG_MMU vma_end_read(lock->vma); #else mmap_read_unlock(lock->mm); diff --git a/kernel/bpf/task_iter.c b/kernel/bpf/task_iter.c index 13e1aabe6f8868..c65ba1dcd86672 100644 --- a/kernel/bpf/task_iter.c +++ b/kernel/bpf/task_iter.c @@ -869,7 +869,7 @@ __bpf_kfunc int bpf_iter_task_vma_new(struct bpf_iter_task_vma *it, BUILD_BUG_ON(sizeof(struct bpf_iter_task_vma_kern) != sizeof(struct bpf_iter_task_vma)); BUILD_BUG_ON(__alignof__(struct bpf_iter_task_vma_kern) != __alignof__(struct bpf_iter_task_vma)); - if (!IS_ENABLED(CONFIG_PER_VMA_LOCK)) { + if (!IS_ENABLED(CONFIG_MMU)) { kit->data = NULL; return -EOPNOTSUPP; } diff --git a/kernel/fork.c b/kernel/fork.c index 416758c8a3d431..22283bf849e152 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1083,9 +1083,7 @@ static void mmap_init_lock(struct mm_struct *mm) { init_rwsem(&mm->mmap_lock); mm_lock_seqcount_init(mm); -#ifdef CONFIG_PER_VMA_LOCK rcuwait_init(&mm->vma_writer_wait); -#endif } static struct mm_struct *mm_init(struct mm_struct *mm, struct task_struct *p) diff --git a/mm/Kconfig b/mm/Kconfig index 2c385f8b29445e..c1ddf59c0d71a8 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -1425,18 +1425,6 @@ config LRU_GEN_WALKS_MMU depends on LRU_GEN && ARCH_HAS_HW_PTE_YOUNG # } -config ARCH_SUPPORTS_PER_VMA_LOCK - def_bool n - -config PER_VMA_LOCK - def_bool y - depends on ARCH_SUPPORTS_PER_VMA_LOCK && MMU && SMP - help - Allow per-vma locking during page fault handling. - - This feature allows locking each virtual memory area separately when - handling page faults instead of taking mmap_lock. - config LOCK_MM_AND_FIND_VMA bool depends on !STACK_GROWSUP diff --git a/mm/Kconfig.debug b/mm/Kconfig.debug index 15dca19dd07da9..9eaa25d1cf2340 100644 --- a/mm/Kconfig.debug +++ b/mm/Kconfig.debug @@ -310,7 +310,6 @@ config DEBUG_KMEMLEAK_VERBOSE config PER_VMA_LOCK_STATS bool "Statistics for per-vma locks" - depends on PER_VMA_LOCK help Say Y here to enable success, retry and failure counters of page faults handled under protection of per-vma locks. When enabled, the diff --git a/mm/debug.c b/mm/debug.c index 9a0297b3988d89..655e6bcc0e8d91 100644 --- a/mm/debug.c +++ b/mm/debug.c @@ -157,17 +157,13 @@ void dump_vma(const struct vm_area_struct *vma) pr_emerg("vma %px start %px end %px mm %px\n" "prot %lx anon_vma %px vm_ops %px\n" "pgoff %lx file %px private_data %px\n" -#ifdef CONFIG_PER_VMA_LOCK "refcnt %x\n" -#endif "flags: %#lx(%pGv)\n", vma, (void *)vma->vm_start, (void *)vma->vm_end, vma->vm_mm, (unsigned long)pgprot_val(vma->vm_page_prot), vma->anon_vma, vma->vm_ops, vma_start_pgoff(vma), vma->vm_file, vma->vm_private_data, -#ifdef CONFIG_PER_VMA_LOCK refcount_read(&vma->vm_refcnt), -#endif vma->vm_flags, &vma->vm_flags); } EXPORT_SYMBOL(dump_vma); diff --git a/mm/init-mm.c b/mm/init-mm.c index 3e792aad762616..a1bb2c2d0284a1 100644 --- a/mm/init-mm.c +++ b/mm/init-mm.c @@ -39,10 +39,8 @@ struct mm_struct init_mm = { .page_table_lock = __SPIN_LOCK_UNLOCKED(init_mm.page_table_lock), .arg_lock = __SPIN_LOCK_UNLOCKED(init_mm.arg_lock), .mmlist = LIST_HEAD_INIT(init_mm.mmlist), -#ifdef CONFIG_PER_VMA_LOCK .vma_writer_wait = __RCUWAIT_INITIALIZER(init_mm.vma_writer_wait), .mm_lock_seq = SEQCNT_ZERO(init_mm.mm_lock_seq), -#endif #ifdef CONFIG_SCHED_MM_CID .mm_cid.lock = __RAW_SPIN_LOCK_UNLOCKED(init_mm.mm_cid.lock), #endif diff --git a/mm/memory.c b/mm/memory.c index 09ac784f8b7b39..bc14cae3c49d72 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -6799,7 +6799,6 @@ static vm_fault_t sanitize_fault_flags(struct vm_area_struct *vma, !vma_is_cow_mapping(vma))) return VM_FAULT_SIGSEGV; } -#ifdef CONFIG_PER_VMA_LOCK /* * Per-VMA locks can't be used with FAULT_FLAG_RETRY_NOWAIT because of * the assumption that lock is dropped on VM_FAULT_RETRY. @@ -6808,7 +6807,6 @@ static vm_fault_t sanitize_fault_flags(struct vm_area_struct *vma, (FAULT_FLAG_VMA_LOCK | FAULT_FLAG_RETRY_NOWAIT)) == (FAULT_FLAG_VMA_LOCK | FAULT_FLAG_RETRY_NOWAIT))) return VM_FAULT_SIGSEGV; -#endif return 0; } diff --git a/mm/mmap_lock.c b/mm/mmap_lock.c index 898c2ef1e95803..272f9ac762b91e 100644 --- a/mm/mmap_lock.c +++ b/mm/mmap_lock.c @@ -43,9 +43,6 @@ void __mmap_lock_do_trace_released(struct mm_struct *mm, bool write) EXPORT_SYMBOL(__mmap_lock_do_trace_released); #endif /* CONFIG_TRACING */ -#ifdef CONFIG_MMU -#ifdef CONFIG_PER_VMA_LOCK - /* State shared across __vma_[start, end]_exclude_readers. */ struct vma_exclude_readers_state { /* Input parameters. */ @@ -299,6 +296,8 @@ struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, MA_STATE(mas, &mm->mm_mt, address, address); struct vm_area_struct *vma; + if (!IS_ENABLED(CONFIG_MMU)) + return NULL; retry: rcu_read_lock(); vma = mas_walk(&mas); @@ -431,7 +430,6 @@ struct vm_area_struct *lock_next_vma(struct mm_struct *mm, return vma; } -#endif /* CONFIG_PER_VMA_LOCK */ #ifdef CONFIG_LOCK_MM_AND_FIND_VMA #include @@ -548,23 +546,3 @@ struct vm_area_struct *lock_mm_and_find_vma(struct mm_struct *mm, return NULL; } #endif /* CONFIG_LOCK_MM_AND_FIND_VMA */ - -#else /* CONFIG_MMU */ - -/* - * At least xtensa ends up having protection faults even with no - * MMU.. No stack expansion, at least. - */ -struct vm_area_struct *lock_mm_and_find_vma(struct mm_struct *mm, - unsigned long addr, struct pt_regs *regs) -{ - struct vm_area_struct *vma; - - mmap_read_lock(mm); - vma = vma_lookup(mm, addr); - if (!vma) - mmap_read_unlock(mm); - return vma; -} - -#endif /* CONFIG_MMU */ diff --git a/mm/pagewalk.c b/mm/pagewalk.c index cc07fcf50e87b3..7411702a37f58d 100644 --- a/mm/pagewalk.c +++ b/mm/pagewalk.c @@ -444,7 +444,6 @@ static inline void process_mm_walk_lock(struct mm_struct *mm, static inline void process_vma_walk_lock(struct vm_area_struct *vma, enum page_walk_lock walk_lock) { -#ifdef CONFIG_PER_VMA_LOCK switch (walk_lock) { case PGWALK_WRLOCK: vma_start_write(vma); @@ -459,7 +458,6 @@ static inline void process_vma_walk_lock(struct vm_area_struct *vma, /* PGWALK_RDLOCK is handled by process_mm_walk_lock */ break; } -#endif } /* diff --git a/mm/rmap.c b/mm/rmap.c index d1819fd6993800..fed0362e0bd0e3 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -260,11 +260,9 @@ static void check_anon_vma_clone(struct vm_area_struct *dst, /* For the anon_vma to be compatible, it can only be singular. */ VM_WARN_ON_ONCE(operation == VMA_OP_MERGE_UNFAULTED && !list_is_singular(&src->anon_vma_chain)); -#ifdef CONFIG_PER_VMA_LOCK /* Only merging an unfaulted VMA leaves the destination attached. */ VM_WARN_ON_ONCE(operation != VMA_OP_MERGE_UNFAULTED && vma_is_attached(dst)); -#endif } static void maybe_reuse_anon_vma(struct vm_area_struct *dst, diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 95c1ed96df9a7a..4ccf472b41630a 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -122,7 +122,6 @@ struct vm_area_struct *find_vma_and_prepare_anon(struct mm_struct *mm, return vma; } -#ifdef CONFIG_PER_VMA_LOCK /* * uffd_lock_vma() - Lookup and lock vma corresponding to @address. * @mm: mm to search vma in. @@ -182,34 +181,6 @@ static void uffd_mfill_unlock(struct vm_area_struct *vma) vma_end_read(vma); } -#else - -static struct vm_area_struct *uffd_mfill_lock(struct mm_struct *dst_mm, - unsigned long dst_start, - unsigned long len) -{ - struct vm_area_struct *dst_vma; - - mmap_read_lock(dst_mm); - dst_vma = find_vma_and_prepare_anon(dst_mm, dst_start); - if (IS_ERR(dst_vma)) - goto out_unlock; - - if (validate_dst_vma(dst_vma, dst_start + len)) - return dst_vma; - - dst_vma = ERR_PTR(-ENOENT); -out_unlock: - mmap_read_unlock(dst_mm); - return dst_vma; -} - -static void uffd_mfill_unlock(struct vm_area_struct *vma) -{ - mmap_read_unlock(vma->vm_mm); -} -#endif - static void mfill_put_vma(struct mfill_state *state) { if (!state->vma) @@ -1850,7 +1821,6 @@ int find_vmas_mm_locked(struct mm_struct *mm, return 0; } -#ifdef CONFIG_PER_VMA_LOCK static int uffd_move_lock(struct mm_struct *mm, unsigned long dst_start, unsigned long src_start, @@ -1925,31 +1895,6 @@ static void uffd_move_unlock(struct vm_area_struct *dst_vma, vma_end_read(dst_vma); } -#else - -static int uffd_move_lock(struct mm_struct *mm, - unsigned long dst_start, - unsigned long src_start, - struct vm_area_struct **dst_vmap, - struct vm_area_struct **src_vmap) -{ - int err; - - mmap_read_lock(mm); - err = find_vmas_mm_locked(mm, dst_start, src_start, dst_vmap, src_vmap); - if (err) - mmap_read_unlock(mm); - return err; -} - -static void uffd_move_unlock(struct vm_area_struct *dst_vma, - struct vm_area_struct *src_vma) -{ - mmap_assert_locked(src_vma->vm_mm); - mmap_read_unlock(dst_vma->vm_mm); -} -#endif - /** * move_pages - move arbitrary anonymous pages of an existing vma * @ctx: pointer to the userfaultfd context diff --git a/rust/kernel/mm.rs b/rust/kernel/mm.rs index 4764d7b68f2a7f..f4fa54616085f2 100644 --- a/rust/kernel/mm.rs +++ b/rust/kernel/mm.rs @@ -170,30 +170,20 @@ impl MmWithUser { /// /// This is an optimistic trylock operation, so it may fail if there is contention. In that /// case, you should fall back to taking the mmap read lock. - /// - /// When per-vma locks are disabled, this always returns `None`. #[inline] pub fn lock_vma_under_rcu(&self, vma_addr: usize) -> Option> { - #[cfg(CONFIG_PER_VMA_LOCK)] - { - // SAFETY: Calling `bindings::lock_vma_under_rcu` is always okay given an mm where - // `mm_users` is non-zero. - let vma = unsafe { bindings::lock_vma_under_rcu(self.as_raw(), vma_addr) }; - if !vma.is_null() { - return Some(VmaReadGuard { - // SAFETY: If `lock_vma_under_rcu` returns a non-null ptr, then it points at a - // valid vma. The vma is stable for as long as the vma read lock is held. - vma: unsafe { VmaRef::from_raw(vma) }, - _nts: NotThreadSafe, - }); - } + // SAFETY: Calling `bindings::lock_vma_under_rcu` is always okay given an mm where + // `mm_users` is non-zero. + let vma = unsafe { bindings::lock_vma_under_rcu(self.as_raw(), vma_addr) }; + if vma.is_null() { + return None; } - - // Silence warnings about unused variables. - #[cfg(not(CONFIG_PER_VMA_LOCK))] - let _ = vma_addr; - - None + Some(VmaReadGuard { + // SAFETY: If `lock_vma_under_rcu` returns a non-null ptr, then it points at a + // valid vma. The vma is stable for as long as the vma read lock is held. + vma: unsafe { VmaRef::from_raw(vma) }, + _nts: NotThreadSafe, + }) } /// Lock the mmap read lock. diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 4c58487b764e9d..57046d8ac81d80 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -560,7 +560,6 @@ struct vm_area_struct { vma_flags_t flags; }; -#ifdef CONFIG_PER_VMA_LOCK /* * Can only be written (using WRITE_ONCE()) while holding both: * - mmap_lock (in write mode) @@ -576,7 +575,7 @@ struct vm_area_struct { * slowpath. */ unsigned int vm_lock_seq; -#endif + unsigned int __vm_anon_pgoff_lo; /* @@ -610,10 +609,8 @@ struct vm_area_struct { #ifdef CONFIG_NUMA_BALANCING struct vma_numab_state *numab_state; /* NUMA Balancing state */ #endif -#ifdef CONFIG_PER_VMA_LOCK /* Unstable RCU readers are allowed to read this. */ refcount_t vm_refcnt; -#endif #ifdef CONFIG_64BIT unsigned int __vm_anon_pgoff_hi; #endif diff --git a/tools/testing/vma/vma_internal.h b/tools/testing/vma/vma_internal.h index 8a48b231aa7abf..54d5c3360aa26b 100644 --- a/tools/testing/vma/vma_internal.h +++ b/tools/testing/vma/vma_internal.h @@ -15,7 +15,6 @@ #include #define CONFIG_MMU 1 -#define CONFIG_PER_VMA_LOCK 1 #ifdef __CONCAT #undef __CONCAT From 097f9391e5598785eea7134b5d7402d53ab0ccf2 Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:53 -0700 Subject: [PATCH 647/857] binder: make shrinker rely solely on per-VMA lock MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit tl;dr: lock_vma_under_rcu() is already a trylock. No need to do both it and mmap_read_trylock(). Long Version: == Background == Historically, binder used an mmap_read_trylock() in its shrinker code. This ensures that reclaim is not blocked on an mmap_lock. Commit 95bc2d4a9020 ("binder: use per-vma lock in page reclaiming") added support for the per-VMA lock, but left mmap_read_trylock() as a fallback. This was presumably because the per-VMA locking can fail for several reasons and most (all?) lock_vma_under_rcu() callers have a fallback to mmap_read_trylock(). == Problem == The fallback is not worth the complexity here. lock_vma_under_rcu() is essentially already a non-blocking trylock. The main reason it fails is also the reason mmap_read_trylock() fails: something is holding mmap_write_lock(). The only remedy for a collision with mmap_write_lock() is to wait, which this code can not do. So the "fallback" after lock_vma_under_rcu() failure is not really a fallback: it is really likely to just be retrying in vain. That retry in an of itself isn't horrible. But it adds complexity. == Solution == Now that per-VMA locks are universally available, lock_vma_under_rcu() will not persistently fail. Rely on it alone and simplify the code. The removal of the fallback does not affect NOMMU case because binder driver depends on CONFIG_MMU. While at it we also make the handling of the cases where the original binder VMA is gone consistent. There are two cases to consider when Binder VMA is gone: 1. there is no VMA at that location anymore. 2. there is now another unrelated VMA at that location. Before this change we handle case 1 by having the shrinker proceed to free the page, and just skip the zap_vma_range() call. And we handle case 2 by having the shrinker return LRU_SKIP. While either behavior is acceptable, we need to handle them in a consistent way. Handle both cases by freeing the page without touching the VMA (skipping the zap_vma_range()). Full disclosure: I originally tried to do this with lock_vma_under_rcu_wait(), but it did not fit well with the mmap_lock trylock semantics. Claude caught this in a review and suggested the approach in this path. It seemed sane to me. So, Suggesed-by: Claude, I guess. Link: https://lore.kernel.org/20260831203056.838265-3-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Reviewed-by: Alice Ryhl Acked-by: Lorenzo Stoakes (ARM) Acked-by: Carlos Llamas Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Signed-off-by: Andrew Morton --- drivers/android/binder_alloc.c | 46 ++++++++++++++++------------------ 1 file changed, 21 insertions(+), 25 deletions(-) diff --git a/drivers/android/binder_alloc.c b/drivers/android/binder_alloc.c index e4488ad86a6557..fcb744088e77c6 100644 --- a/drivers/android/binder_alloc.c +++ b/drivers/android/binder_alloc.c @@ -1142,7 +1142,6 @@ enum lru_status binder_alloc_free_page(struct list_head *item, struct vm_area_struct *vma; struct page *page_to_free; unsigned long page_addr; - int mm_locked = 0; size_t index; if (!mmget_not_zero(mm)) @@ -1151,27 +1150,25 @@ enum lru_status binder_alloc_free_page(struct list_head *item, index = mdata->page_index; page_addr = alloc->vm_start + index * PAGE_SIZE; - /* attempt per-vma lock first */ + /* + * Attempt per-vma lock. This is essentially a + * "trylock". It can fail even if the VMA exists + * for 'page_addr'. + */ vma = lock_vma_under_rcu(mm, page_addr); if (!vma) { - /* fall back to mmap_lock */ - if (!mmap_read_trylock(mm)) - goto err_mmap_read_lock_failed; - mm_locked = 1; - vma = vma_lookup(mm, page_addr); + /* + * If the vma exists, we can't continue because we cannot + * remove the page from the vma. However, if the vma was + * unmapped, it's okay to continue. + */ + if (binder_alloc_is_mapped(alloc)) + goto err_vma_lock_failed; } if (!mutex_trylock(&alloc->mutex)) goto err_get_alloc_mutex_failed; - /* - * Since a binder_alloc can only be mapped once, we ensure - * the vma corresponds to this mapping by checking whether - * the binder_alloc is still mapped. - */ - if (vma && !binder_alloc_is_mapped(alloc)) - goto err_invalid_vma; - trace_binder_unmap_kernel_start(alloc, index); page_to_free = alloc->pages[index]; @@ -1182,7 +1179,12 @@ enum lru_status binder_alloc_free_page(struct list_head *item, list_lru_isolate(lru, item); spin_unlock(&lru->lock); - if (vma) { + /* + * Since a binder_alloc can only be mapped once, we ensure + * the vma corresponds to this mapping by checking whether + * the binder_alloc is still mapped. + */ + if (vma && binder_alloc_is_mapped(alloc)) { trace_binder_unmap_user_start(alloc, index); zap_vma_range(vma, page_addr, PAGE_SIZE); @@ -1191,23 +1193,17 @@ enum lru_status binder_alloc_free_page(struct list_head *item, } mutex_unlock(&alloc->mutex); - if (mm_locked) - mmap_read_unlock(mm); - else + if (vma) vma_end_read(vma); mmput_async(mm); binder_free_page(page_to_free); return LRU_REMOVED_RETRY; -err_invalid_vma: - mutex_unlock(&alloc->mutex); err_get_alloc_mutex_failed: - if (mm_locked) - mmap_read_unlock(mm); - else + if (vma) vma_end_read(vma); -err_mmap_read_lock_failed: +err_vma_lock_failed: mmput_async(mm); err_mmget: return LRU_SKIP; From c072915e745cdfc95483f87292141193327f9d89 Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:54 -0700 Subject: [PATCH 648/857] mm: add RCU-based VMA lookup helper that waits for writers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit There are basically two parallel ways to look up a VMA: the traditional way, which is protected by mmap_read_lock, and the RCU-based per-VMA lock way which is based on RCU and refcounts. However, per-VMA locks will fail if the lock is help by a writer and therefore never waits. In a number of places we need to wait for the lock and it's done by falling back to mmap_read_lock, locking the VMA and releasing the mmap_lock once VMA is locked. Add vma_start_read_unlocked() - a variant of the RCU-based lookup that waits for writers. This is basically the same as the existing RCU-based lookup, but on a failure to lock it temporarily takes mmap_lock for read and waits for writers to finish before locking the VMA, dropping the mmap_read_lock and returning the locked VMA. This has some advantages: 1. Callers do not need to have a fallback path for when they collide with writers. 2. Its fast path does not require taking mmap_lock for read. Basically, when applied correctly, this approach results in faster *and* simpler code. While at it, fix the comments for vma_start_read_locked(), vma_start_read_locked_nested(), and uffd_lock_vma(). Link: https://lore.kernel.org/20260831203056.838265-4-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Suggested-by: Lorenzo Stoakes (ARM) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Cc: Liam R. Howlett Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Signed-off-by: Andrew Morton --- include/linux/mmap_lock.h | 19 +++++++++++++++---- mm/mmap_lock.c | 35 +++++++++++++++++++++++++++++++++++ mm/userfaultfd.c | 6 ++++-- 3 files changed, 54 insertions(+), 6 deletions(-) diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index 27c00bac29f9e1..00eae65b74bd62 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -228,10 +228,14 @@ static inline void vma_refcount_put(struct vm_area_struct *vma) } /* - * Use only while holding mmap read lock which guarantees that locking will not - * fail (nobody can concurrently write-lock the vma). vma_start_read() should + * Use only while holding mmap read lock which guarantees that vma lock is not + * contended (nobody can concurrently write-lock the vma). vma_start_read() should * not be used in such cases because it might fail due to mm_lock_seq overflow. * This functionality is used to obtain vma read lock and drop the mmap read lock. + * + * VMA can't be detached while we are holding mmap lock, therefore in practice this + * function can fail only when there are so many readers that vm_refcnt overflows. + * The failure case is very unlikely and is already annotated as such internally. */ static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int subclass) { @@ -247,16 +251,23 @@ static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int } /* - * Use only while holding mmap read lock which guarantees that locking will not - * fail (nobody can concurrently write-lock the vma). vma_start_read() should + * Use only while holding mmap read lock which guarantees that vma lock is not + * contended (nobody can concurrently write-lock the vma). vma_start_read() should * not be used in such cases because it might fail due to mm_lock_seq overflow. * This functionality is used to obtain vma read lock and drop the mmap read lock. + * + * VMA can't be detached while we are holding mmap lock, therefore in practice this + * function can fail only when there are so many readers that vm_refcnt overflows. + * The failure case is very unlikely and is already annotated as such internally. */ static inline bool vma_start_read_locked(struct vm_area_struct *vma) { return vma_start_read_locked_nested(vma, 0); } +struct vm_area_struct *vma_start_read_unlocked(struct mm_struct *mm, + unsigned long address); + static inline void vma_end_read(struct vm_area_struct *vma) { vma_refcount_put(vma); diff --git a/mm/mmap_lock.c b/mm/mmap_lock.c index 272f9ac762b91e..2f94ee0fdee2a2 100644 --- a/mm/mmap_lock.c +++ b/mm/mmap_lock.c @@ -340,6 +340,41 @@ struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, return NULL; } +/** + * vma_start_read_unlocked() - Find the VMA covering 'address' and read-lock it. + * @mm: the mm_struct of the address space to search + * @address: address that the vma should contain + * + * The fast path does not take mmap_lock. Waits for writers to finish if the + * VMA is being modified by taking mmap_lock. + * Use when mmap_lock is not held, otherwise use vma_start_read_locked(). + * Nothing prevents VMAs being unmapped/mapped before or after the VMA is + * looked up, if a stronger guarantee is required, take an mmap_lock. + * + * Return: If a VMA exists which spans @address, return that VMA, read-locked. + * If no VMA is mapped there or, very unlikely, a reference count overflow + * occurred, return NULL. + */ +struct vm_area_struct *vma_start_read_unlocked(struct mm_struct *mm, + unsigned long address) +{ + struct vm_area_struct *vma; + + /* Fast path: return stable VMA covering 'address': */ + vma = lock_vma_under_rcu(mm, address); + if (vma) + return vma; + + /* Slow path: preclude VMA writers by temporarily getting mmap read lock. */ + mmap_read_lock(mm); + vma = vma_lookup(mm, address); + if (vma && !vma_start_read_locked(vma)) + vma = NULL; + mmap_read_unlock(mm); + + return vma; +} + static struct vm_area_struct *lock_next_vma_under_mmap_lock(struct mm_struct *mm, struct vma_iterator *vmi, unsigned long from_addr) diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 4ccf472b41630a..79cc7b546f130e 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -129,8 +129,10 @@ struct vm_area_struct *find_vma_and_prepare_anon(struct mm_struct *mm, * * Should be called without holding mmap_lock. * - * Return: A locked vma containing @address, -ENOENT if no vma is found, or - * -ENOMEM if anon_vma couldn't be allocated. + * Return: A locked vma containing @address, -ENOENT if no vma is found, + * -ENOMEM if anon_vma couldn't be allocated, or -EAGAIN if vma refcount + * overflow happened due to high number of readers and the caller should + * retry later. */ static struct vm_area_struct *uffd_lock_vma(struct mm_struct *mm, unsigned long address) From b4f10dd60693a360ae5034cffda63dcb3b837d8c Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:55 -0700 Subject: [PATCH 649/857] binder: remove mmap_lock fallback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Previously, the per-VMA locking could fail in the face of writers which necessitate a fallback to mmap_lock. The new vma_start_read_unlocked() will wait for writers instead of failing. Use the new helper. Wait for writers. Remove the fallback to mmap_lock. Link: https://lore.kernel.org/20260831203056.838265-5-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Reviewed-by: Alice Ryhl Acked-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Signed-off-by: Andrew Morton --- drivers/android/binder/page_range.rs | 19 +++---------------- drivers/android/binder_alloc.c | 17 +++++------------ rust/kernel/mm.rs | 28 ++++++++++++++++++++++++++++ 3 files changed, 36 insertions(+), 28 deletions(-) diff --git a/drivers/android/binder/page_range.rs b/drivers/android/binder/page_range.rs index 52ffbf3504e7f7..71febd3d5b0730 100644 --- a/drivers/android/binder/page_range.rs +++ b/drivers/android/binder/page_range.rs @@ -439,22 +439,9 @@ impl ShrinkablePageRange { // workqueue. let mm = MmWithUser::into_mmput_async(self.mm.mmget_not_zero().ok_or(ESRCH)?); { - let vma_read; - let mmap_read; - let vma = if let Some(ret) = mm.lock_vma_under_rcu(vma_addr) { - vma_read = ret; - check_vma(&vma_read, self) - } else { - mmap_read = mm.mmap_read_lock(); - mmap_read - .vma_lookup(vma_addr) - .and_then(|vma| check_vma(vma, self)) - }; - - match vma { - Some(vma) => vma.vm_insert_page(user_page_addr, &new_page)?, - None => return Err(ESRCH), - } + let vma_read_guard = mm.vma_start_read_unlocked(vma_addr).ok_or(ESRCH)?; + let vma = check_vma(&vma_read_guard, self).ok_or(ESRCH)?; + vma.vm_insert_page(user_page_addr, &new_page)?; } let inner = self.lock.lock(); diff --git a/drivers/android/binder_alloc.c b/drivers/android/binder_alloc.c index fcb744088e77c6..d6eae0aa708541 100644 --- a/drivers/android/binder_alloc.c +++ b/drivers/android/binder_alloc.c @@ -259,21 +259,14 @@ static int binder_page_insert(struct binder_alloc *alloc, struct vm_area_struct *vma; int ret = -ESRCH; - /* attempt per-vma lock first */ - vma = lock_vma_under_rcu(mm, addr); - if (vma) { - if (binder_alloc_is_mapped(alloc)) - ret = vm_insert_page(vma, addr, page); - vma_end_read(vma); + vma = vma_start_read_unlocked(mm, addr); + if (!vma) return ret; - } - /* fall back to mmap_lock */ - mmap_read_lock(mm); - vma = vma_lookup(mm, addr); - if (vma && binder_alloc_is_mapped(alloc)) + if (binder_alloc_is_mapped(alloc)) ret = vm_insert_page(vma, addr, page); - mmap_read_unlock(mm); + + vma_end_read(vma); return ret; } diff --git a/rust/kernel/mm.rs b/rust/kernel/mm.rs index f4fa54616085f2..58bc1793fdaf5e 100644 --- a/rust/kernel/mm.rs +++ b/rust/kernel/mm.rs @@ -186,6 +186,34 @@ impl MmWithUser { }) } + /// Find the VMA covering 'address' and read-lock it. + /// + /// The fast path does not take mmap_lock. Waits for writers to finish if the + /// VMA is being modified by taking mmap_lock. + /// Use when mmap_lock is not held, otherwise use vma_start_read_locked(). + /// Nothing prevents VMAs being unmapped/mapped before or after the VMA is + /// looked up, if a stronger guarantee is required, take an mmap_lock. + /// + /// Return: If a VMA exists which spans @address, return that VMA, read-locked. + /// If no VMA is mapped there or, very unlikely, a reference count overflow + /// occurred, return NULL. + #[inline] + pub fn vma_start_read_unlocked(&self, vma_addr: usize) -> Option> { + // SAFETY: We may invoke `vma_start_read_unlocked` because we know this `mm` has non-zero + // `mm_users`. + let vma = unsafe { bindings::vma_start_read_unlocked(self.as_raw(), vma_addr) }; + if vma.is_null() { + return None; + } + // INVARIANT: We just acquired the VMA read lock. + Some(VmaReadGuard { + // SAFETY: If `vma_start_read_unlocked` returns a non-null ptr, then it points at a + // valid vma. The vma is stable for as long as the vma read lock is held. + vma: unsafe { VmaRef::from_raw(vma) }, + _nts: NotThreadSafe, + }) + } + /// Lock the mmap read lock. #[inline] pub fn mmap_read_lock(&self) -> MmapReadGuard<'_> { From 02f913f6627df60bcab06342a98cc731dc38de01 Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:56 -0700 Subject: [PATCH 650/857] tcp: remove mmap_lock fallback path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Previously, the per-VMA locking could fail in the face of writers which necessitates a fallback to mmap_lock. The new vma_start_read_unlocked() will wait for writers instead of failing. Use the new helper. Wait for writers. Remove the fallback to mmap_lock. The fallback removal does not affect NOMMU case because TCP_ZEROCOPY is gated on CONFIG_MMU. This really is a nice cleanup. It removes the need to pass the lock state back and forth to find_tcp_vma(). Link: https://lore.kernel.org/20260831203056.838265-6-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Acked-by: Lorenzo Stoakes Acked-by: Vlastimil Babka (SUSE) Tested-by: syzbot@syzkaller.appspotmail.com Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Arve Hjønnevåg Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Signed-off-by: Andrew Morton --- net/ipv4/tcp.c | 31 +++++++++---------------------- 1 file changed, 9 insertions(+), 22 deletions(-) diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c index b4237d0e994d6f..b9ffd993e2c599 100644 --- a/net/ipv4/tcp.c +++ b/net/ipv4/tcp.c @@ -2168,27 +2168,18 @@ static void tcp_zc_finalize_rx_tstamp(struct sock *sk, } static struct vm_area_struct *find_tcp_vma(struct mm_struct *mm, - unsigned long address, - bool *mmap_locked) + unsigned long address) { - struct vm_area_struct *vma = lock_vma_under_rcu(mm, address); + struct vm_area_struct *vma = vma_start_read_unlocked(mm, address); - if (vma) { - if (vma->vm_ops != &tcp_vm_ops) { - vma_end_read(vma); - return NULL; - } - *mmap_locked = false; - return vma; - } + if (!vma) + return NULL; - mmap_read_lock(mm); - vma = vma_lookup(mm, address); - if (!vma || vma->vm_ops != &tcp_vm_ops) { - mmap_read_unlock(mm); + if (vma->vm_ops != &tcp_vm_ops) { + vma_end_read(vma); return NULL; } - *mmap_locked = true; + return vma; } @@ -2209,7 +2200,6 @@ static int tcp_zerocopy_receive(struct sock *sk, u32 seq = tp->copied_seq; u32 total_bytes_to_map; int inq = tcp_inq(sk); - bool mmap_locked; int ret; zc->copybuf_len = 0; @@ -2234,7 +2224,7 @@ static int tcp_zerocopy_receive(struct sock *sk, return 0; } - vma = find_tcp_vma(current->mm, address, &mmap_locked); + vma = find_tcp_vma(current->mm, address); if (!vma) return -EINVAL; @@ -2316,10 +2306,7 @@ static int tcp_zerocopy_receive(struct sock *sk, zc, total_bytes_to_map); } out: - if (mmap_locked) - mmap_read_unlock(current->mm); - else - vma_end_read(vma); + vma_end_read(vma); /* Try to copy straggler data. */ if (!ret) copylen = tcp_zc_handle_leftover(zc, sk, skb, &seq, copybuf_len, tss); From 07088210a6fe30fae448933f9b2953d0e9e56622 Mon Sep 17 00:00:00 2001 From: Joe Damato Date: Mon, 31 Aug 2026 10:48:35 -0700 Subject: [PATCH 651/857] mm: memcontrol: raise MEMCG_MAX for charges that fail without reclaiming Charges that exceed memory.max and return through the nomem label can raise no event and simply return -ENOMEM. A non-blocking charge can hit the limit, get rejected, but is not visible in memory.events. This was noticed in a production setting where bpf_mem_alloc() attempted to refill its per-cpu freelists, which triggered a non-blocking charge while at the limit. Commit d6e103a757fa ("mm: memcontrol: do not miss MEMCG_MAX events for enforced allocations") added raised_max_event to cover charges that are force charged without ever reaching reclaim, but charges that are rejected outright were left out. Getting an allocation failure without the corresponding MEMCG_MAX event is unexpected and makes debugging and monitoring harder. Raise the event on the way out for rejected charges as well, by routing the -ENOMEM return through the same exit path that already covers forced charges. The existing behavior of raising a MEMCG_MAX event on every charge/reclaim/retry iteration is left unchanged. Tested with a module that performs accounted GFP_NOWAIT page allocations from a task in a cgroup at its memory.max, and measures the resulting memory.events:max delta. Without this patch the rejected charges raise no event at all; with it the delta matches the number of rejected charges exactly. A GFP_KERNEL|__GFP_NORETRY control, which reaches reclaim, raises the same two events per failed charge before and after, confirming the existing charge/reclaim/retry accounting is unchanged. Link: https://lore.kernel.org/20260831174836.3102406-1-joe@dama.to Fixes: d6e103a757fa ("mm: memcontrol: do not miss MEMCG_MAX events for enforced allocations") Signed-off-by: Joe Damato Suggested-by: Shakeel Butt Acked-by: Shakeel Butt Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Signed-off-by: Andrew Morton --- mm/memcontrol.c | 28 ++++++++++++++++------------ 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index d5ebe83eae3efc..aeaa09e01d70ea 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2685,10 +2685,11 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, bool raised_max_event = false; unsigned long pflags; bool allow_spinning = gfpflags_allow_spinning(gfp_mask); + int ret = 0; retry: if (consume_stock(memcg, nr_pages)) - return 0; + return ret; if (!allow_spinning) /* Avoid the refill and flush of the older stock */ @@ -2799,16 +2800,11 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, * put the burden of reclaim on regular allocation requests * and let these go through as privileged allocations. */ - if (!(gfp_mask & (__GFP_NOFAIL | __GFP_HIGH))) - return -ENOMEM; + if (!(gfp_mask & (__GFP_NOFAIL | __GFP_HIGH))) { + ret = -ENOMEM; + goto out; + } force: - /* - * If the allocation has to be enforced, don't forget to raise - * a MEMCG_MAX event. - */ - if (!raised_max_event) - __memcg_memory_event(mem_over_limit, MEMCG_MAX, allow_spinning); - /* * The allocation either can't fail or will lead to more memory * being freed very soon. Allow memory usage go over the limit @@ -2818,7 +2814,15 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, if (do_memsw_account()) page_counter_charge(&memcg->memsw, nr_pages); - return 0; +out: + /* + * Don't forget to raise a MEMCG_MAX event for forced or rejected + * requests. + */ + if (!raised_max_event) + __memcg_memory_event(mem_over_limit, MEMCG_MAX, allow_spinning); + + return ret; done_restock: if (batch > nr_pages) @@ -2877,7 +2881,7 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, !(current->flags & PF_MEMALLOC) && gfpflags_allow_blocking(gfp_mask)) __mem_cgroup_handle_over_high(gfp_mask); - return 0; + return ret; } static inline int try_charge(struct mem_cgroup *memcg, gfp_t gfp_mask, From b2b6fd80d31a130a9b19dc322a7cd7caaaca466a Mon Sep 17 00:00:00 2001 From: Jason Angelov Date: Mon, 31 Aug 2026 08:06:48 -0700 Subject: [PATCH 652/857] mm/damon/core-kunit: test probe_hits handling at region split and merge Patch series "mm/damon: add kunit tests for probe_hits handling and probe params validation", v2. DAMON recently introduced probes and probe weights. Add kunit tests for the propagation of probe_hits at region split and merge, and the rejection of invalid probe parameters by damon_valid_probe_params(). This patch (of 2): damon_split_region_at() copies probe_hits[] and last_probe_hits[] to the new split region. damon_merge_two_regions() sets probe_hits[] to the size-weighted average of the merged regions. Extend damon_test_split_at() and damon_test_merge_two() tests to cover those fields. Link: https://lore.kernel.org/20260831150650.84829-1-sj@kernel.org Link: https://lore.kernel.org/20260831150650.84829-2-sj@kernel.org Signed-off-by: Jason Angelov Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: David Gow Cc: Brendan Higgins Signed-off-by: Andrew Morton --- mm/damon/tests/core-kunit.h | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 4a536d41cdb2d0..2db94d49c9bae4 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -152,6 +152,8 @@ static void damon_test_split_at(struct kunit *test) } r->nr_accesses = 42; r->last_nr_accesses = 15; + r->probe_hits[0] = 7; + r->last_probe_hits[0] = 3; r->age = 10; damon_add_region(r, t); damon_split_region_at(t, r, 25); @@ -168,6 +170,8 @@ static void damon_test_split_at(struct kunit *test) KUNIT_EXPECT_EQ(test, r->nr_accesses, r_new->nr_accesses); KUNIT_EXPECT_EQ(test, r->last_nr_accesses, r_new->last_nr_accesses); + KUNIT_EXPECT_EQ(test, r->probe_hits[0], r_new->probe_hits[0]); + KUNIT_EXPECT_EQ(test, r->last_probe_hits[0], r_new->last_probe_hits[0]); KUNIT_EXPECT_EQ(test, r->age, r_new->age); out: @@ -189,6 +193,7 @@ static void damon_test_merge_two(struct kunit *test) kunit_skip(test, "region alloc fail"); } r->nr_accesses = 10; + r->probe_hits[0] = 6; r->age = 9; damon_add_region(r, t); r2 = damon_new_region(100, 300); @@ -197,6 +202,7 @@ static void damon_test_merge_two(struct kunit *test) kunit_skip(test, "second region alloc fail"); } r2->nr_accesses = 20; + r2->probe_hits[0] = 14; r2->age = 21; damon_add_region(r2, t); @@ -204,6 +210,7 @@ static void damon_test_merge_two(struct kunit *test) KUNIT_EXPECT_EQ(test, r->ar.start, 0ul); KUNIT_EXPECT_EQ(test, r->ar.end, 300ul); KUNIT_EXPECT_EQ(test, r->nr_accesses, 16u); + KUNIT_EXPECT_EQ(test, r->probe_hits[0], 11); KUNIT_EXPECT_EQ(test, r->age, 17u); i = 0; From 3376ed377991372356c750284cf9cda8b0538324 Mon Sep 17 00:00:00 2001 From: Jason Angelov Date: Mon, 31 Aug 2026 08:06:49 -0700 Subject: [PATCH 653/857] mm/damon/core-kunit: test damon_valid_probe_params() damon_valid_probe_params() makes damon_commit_ctx() reject probe configurations that could overflow a probe_hits counter, a single (weight * probe_hits) product, or the sum of those products. Add a kunit test covering each rejection at its boundary: - samples per aggregation interval: U8_MAX is allowed, one more could overflow a probe_hits counter - single weight: the largest whose product fits in unsigned int is allowed, one larger is rejected - multiple probes: each product fits, but their sum overflows - no weight set: the validation is skipped Link: https://lore.kernel.org/20260831150650.84829-3-sj@kernel.org Signed-off-by: Jason Angelov Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow Signed-off-by: Andrew Morton --- mm/damon/tests/core-kunit.h | 57 +++++++++++++++++++++++++++++++++++++ 1 file changed, 57 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 2db94d49c9bae4..b1ca4c8e03f091 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1338,6 +1338,62 @@ static void damon_test_commit_ctx(struct kunit *test) damon_destroy_ctx(dst); } +static void damon_test_valid_probe_params(struct kunit *test) +{ + struct damon_ctx *ctx; + struct damon_probe *probe, *probe2; + + ctx = damon_new_ctx(); + if (!ctx) + kunit_skip(test, "ctx alloc fail"); + probe = damon_new_probe(); + if (!probe) { + damon_destroy_ctx(ctx); + kunit_skip(test, "probe alloc fail"); + } + damon_add_probe(ctx, probe); + + /* Parameters are validated only if any probe weight is set. */ + ctx->attrs.sample_interval = 1; + ctx->attrs.aggr_interval = 1000000; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* Up to U8_MAX samples per aggregation interval are allowed. */ + probe->weight = 100; + ctx->attrs.aggr_interval = 255; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* More samples could overflow the probe_hits counters. */ + ctx->attrs.aggr_interval = 256; + KUNIT_EXPECT_FALSE(test, damon_valid_probe_params(ctx)); + + /* The largest weight whose weighted hit count fits in unsigned int. */ + ctx->attrs.aggr_interval = 255; + probe->weight = UINT_MAX / 255; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* Any larger weight could overflow its weighted hit count. */ + probe->weight = UINT_MAX / 255 + 1; + KUNIT_EXPECT_FALSE(test, damon_valid_probe_params(ctx)); + + /* With one sample per aggregation, even the largest weight fits. */ + ctx->attrs.aggr_interval = 1; + probe->weight = UINT_MAX; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* The sum of all probes' weighted hit counts could also overflow. */ + probe2 = damon_new_probe(); + if (!probe2) { + damon_destroy_ctx(ctx); + kunit_skip(test, "probe2 alloc fail"); + } + probe2->weight = 1; + damon_add_probe(ctx, probe2); + KUNIT_EXPECT_FALSE(test, damon_valid_probe_params(ctx)); + + damon_destroy_ctx(ctx); +} + static void damos_test_filter_out(struct kunit *test) { struct damon_target *t; @@ -1664,6 +1720,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damos_test_commit_migrate_hot), KUNIT_CASE(damon_test_commit_target_regions), KUNIT_CASE(damon_test_commit_ctx), + KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), KUNIT_CASE(damon_test_feed_loop_next_input), KUNIT_CASE(damon_test_set_filters_default_reject), From f5933b2b73c8666c4a8524cae923c18c0d9341c1 Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 31 Aug 2026 08:02:24 -0700 Subject: [PATCH 654/857] docs/mm/damon/design: accurate semantics of nr_snapshots Patch series "docs/mm/damon/design: add explanation of nr_snapshots", v3. Add an explanation of nr_snapshots to avoid misunderstandings. This patch (of 3): Change "tried to be applied" -> "completely tried to be applied" to maintain consistency between the documentation and the code. Link: https://lore.kernel.org/20260831150227.83416-1-sj@kernel.org Link: https://lore.kernel.org/20260831150227.83416-2-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/mm/damon/design.rst | 4 ++-- include/linux/damon.h | 3 ++- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index aed6cb1cf48310..1739aeec6eb95f 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -846,8 +846,8 @@ scheme's execution. - ``nr_applied``: Total number of regions that the scheme is applied. - ``sz_applied``: Total size of regions that the scheme is applied. - ``qt_exceeds``: Total number of times the quota of the scheme has exceeded. -- ``nr_snapshots``: Total number of DAMON snapshots that the scheme is tried to - be applied. +- ``nr_snapshots``: Total number of DAMON snapshots that the scheme is + completely tried to be applied. - ``max_nr_snapshots``: Upper limit of ``nr_snapshots``. "A scheme is tried to be applied to a region" means DAMOS core logic determined diff --git a/include/linux/damon.h b/include/linux/damon.h index 0c8b7ddef9abb3..cbdf5f77978e72 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -357,7 +357,8 @@ struct damos_watermarks { * Total bytes that passed ops layer-handled DAMOS filters. * @qt_exceeds: Total number of times the quota of the scheme has exceeded. * @nr_snapshots: - * Total number of DAMON snapshots that the scheme has tried. + * Total number of DAMON snapshots that the scheme is completely + * tried to be applied. * * "Tried an action to a region" in this context means the DAMOS core logic * determined the region as eligible to apply the action. The access pattern From 924f50d7a7cdcc2ee84b7431eb4a776cc3c7de68 Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 31 Aug 2026 08:02:25 -0700 Subject: [PATCH 655/857] docs/mm/damon/design: difference between watermarks and nr_snapshots Explain the difference between nr_snapshots reaches max_nr_snapshots and watermarks. Link: https://lore.kernel.org/20260831150227.83416-3-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/mm/damon/design.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 1739aeec6eb95f..e7977f005ac06e 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -872,7 +872,8 @@ the action to the region will fail. Unlike normal stats, ``max_nr_snapshots`` is set by users. If it is set as non-zero and ``nr_snapshots`` be same to or greater than ``nr_snapshots``, the -scheme is deactivated. +scheme is deactivated. Note that, unlike watermarks, even if a scheme's +``nr_snapshots`` reaches ``max_nr_snapshots``, monitoring will not stop. To know how user-space can read the stats via :ref:`DAMON sysfs interface `, refer to :ref:s`stats ` part of the From 52b8b4aac08fe13838d6ba63d287ea6f0d5d2679 Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 31 Aug 2026 08:02:26 -0700 Subject: [PATCH 656/857] docs/mm/damon/design: fix typo of max_nr_snapshots Fix a typo (nr_snapshots -> max_nr_snapshots) and corrects a grammar error. Link: https://lore.kernel.org/20260831150227.83416-4-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/mm/damon/design.rst | 4 ++-- include/linux/damon.h | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index e7977f005ac06e..d1dd9050ebf40b 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -871,8 +871,8 @@ action is ``pageout`` while all pages of the region are unreclaimable, applying the action to the region will fail. Unlike normal stats, ``max_nr_snapshots`` is set by users. If it is set as -non-zero and ``nr_snapshots`` be same to or greater than ``nr_snapshots``, the -scheme is deactivated. Note that, unlike watermarks, even if a scheme's +non-zero and ``nr_snapshots`` equals or is greater than ``max_nr_snapshots``, +the scheme is deactivated. Note that, unlike watermarks, even if a scheme's ``nr_snapshots`` reaches ``max_nr_snapshots``, monitoring will not stop. To know how user-space can read the stats via :ref:`DAMON sysfs interface diff --git a/include/linux/damon.h b/include/linux/damon.h index cbdf5f77978e72..4b0d2d2e4ea4eb 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -549,7 +549,7 @@ struct damos_migrate_dests { * * After applying the &action to each region, &stat is updated. * - * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots be same to or + * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots equals or is * greater than it, the scheme is deactivated. */ struct damos { From 3caa100a4fa19bdfd2cfa01c02b4e4db500aae48 Mon Sep 17 00:00:00 2001 From: Cheng-Han Wu Date: Mon, 31 Aug 2026 07:57:21 -0700 Subject: [PATCH 657/857] mm/damon/core: remove unused damon_targets_empty() Patch series "mm/damon/core: remove unused helper functions", v2. Both damon_targets_empty() and damon_nr_running_ctxs() have had no in-tree users since commit 5ec4333b1967 ("mm/damon: remove DAMON debugfs interface") removed their remaining callers. Remove the unused declarations and definitions. This patch (of 2): damon_targets_empty() has had no in-tree users since commit 5ec4333b1967 ("mm/damon: remove DAMON debugfs interface") removed its last caller. Remove the unused declaration and definition. Link: https://lore.kernel.org/20260831145724.82387-1-sj@kernel.org Link: https://lore.kernel.org/20260831145724.82387-2-sj@kernel.org Signed-off-by: Cheng-Han Wu Signed-off-by: SJ Park Reviewed-by: SJ Park Signed-off-by: Andrew Morton --- include/linux/damon.h | 1 - mm/damon/core.c | 5 ----- 2 files changed, 6 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 4b0d2d2e4ea4eb..89a41dea1d23f4 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1055,7 +1055,6 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src); struct damon_target *damon_new_target(void); void damon_add_target(struct damon_ctx *ctx, struct damon_target *t); -bool damon_targets_empty(struct damon_ctx *ctx); void damon_free_target(struct damon_target *t); void damon_destroy_target(struct damon_target *t, struct damon_ctx *ctx); unsigned int damon_nr_regions(struct damon_target *t); diff --git a/mm/damon/core.c b/mm/damon/core.c index 644daf5a165606..da97b2dc39d16d 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -795,11 +795,6 @@ void damon_add_target(struct damon_ctx *ctx, struct damon_target *t) list_add_tail(&t->list, &ctx->adaptive_targets); } -bool damon_targets_empty(struct damon_ctx *ctx) -{ - return list_empty(&ctx->adaptive_targets); -} - static void damon_del_target(struct damon_target *t) { list_del(&t->list); From a0e351bdda6d951dacc47b0888d24fa9662a97d0 Mon Sep 17 00:00:00 2001 From: Cheng-Han Wu Date: Mon, 31 Aug 2026 07:57:22 -0700 Subject: [PATCH 658/857] mm/damon/core: remove unused damon_nr_running_ctxs() damon_nr_running_ctxs() has had no in-tree users since commit 5ec4333b1967 ("mm/damon: remove DAMON debugfs interface") removed all of its callers. Remove the unused declaration and definition. Link: https://lore.kernel.org/20260831145724.82387-3-sj@kernel.org Signed-off-by: Cheng-Han Wu Signed-off-by: SJ Park Reviewed-by: SJ Park Signed-off-by: Andrew Morton --- include/linux/damon.h | 1 - mm/damon/core.c | 14 -------------- 2 files changed, 15 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 89a41dea1d23f4..6993dca0f355cc 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1065,7 +1065,6 @@ int damon_set_attrs(struct damon_ctx *ctx, struct damon_attrs *attrs); void damon_set_schemes(struct damon_ctx *ctx, struct damos **schemes, ssize_t nr_schemes); int damon_commit_ctx(struct damon_ctx *old_ctx, struct damon_ctx *new_ctx); -int damon_nr_running_ctxs(void); bool damon_is_registered_ops(enum damon_ops_id id); int damon_register_ops(struct damon_operations *ops); int damon_select_ops(struct damon_ctx *ctx, enum damon_ops_id id); diff --git a/mm/damon/core.c b/mm/damon/core.c index da97b2dc39d16d..891f851bcbbf6f 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1847,20 +1847,6 @@ int damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src) return err; } -/** - * damon_nr_running_ctxs() - Return number of currently running contexts. - */ -int damon_nr_running_ctxs(void) -{ - int nr_ctxs; - - mutex_lock(&damon_lock); - nr_ctxs = nr_running_ctxs; - mutex_unlock(&damon_lock); - - return nr_ctxs; -} - /* Returns the size upper limit for each monitoring region */ static unsigned long damon_region_sz_limit(struct damon_ctx *ctx) { From 7af0cf17367c6a20009044280f83dd39f20d32e5 Mon Sep 17 00:00:00 2001 From: Asier Gutierrez Date: Mon, 31 Aug 2026 07:47:28 -0700 Subject: [PATCH 659/857] mm/damon: introduce DAMOS_QUOTA_HUGEPAGE auto tuning Patch series "mm/damon: Introduce a huge page collapsing mechanism using auto tuning", v4. Overview ======== This patchset introduces a new autotuning which allows to collapse hot regions into hugepages. Motivation ========== Since TLB is a bottleneck for many systems[1], a way to optimize TLB misses (or hits) is to use huge pages. Unfortunately, using "always" in THP leads to memory fragmentation and memory waste. For this reason, most application guides and system administrators suggest to disable THP. Selective huge page collapse per process is possible using prctl and a launcher. However, this does not solve the issue with hot region detection. Additionally, it the sysadmin should create a launcher that uses PRCTL to enable THP for a particular process. We can use the DAMON support for DAMOS_HUGEPAGE and DAMOS_COLLAPSE, to target a certain process. DAMOS_COLLAPSE can also target the hot regions in that process. Still, there is an issue with the amount of huge page consumption. Since huge pages can lead to memory fragmentation and waste, there should be a way to limit the amount of huge page consumption. There is hugetlbfs, but it requires changes to the application code or the use of libhugetlbfs. DAMON has now a way to autotune some of the variables and adjust quotas automatically, so that DAMON is fired only under the right circumstances. It would be nice to have something similar, but for huge pages. Solution ======== A new autotuning quota goal[2], damos_hugepage_mem_bp, is introduced, which checks the huge page consumption to total memory consumption. This new quota mechanism reuses current autotuning architecture. In order to test this new mechanism, a sample module[3] was created, but not included in this patch series. To demonstrate the tool, damo user space tool was modified[4], which sets up huge pages collapse autotuning. Benchmarks ========== Setup: physical server with arm64 processor with 4 NUMA nodes, 1 TB RAM and running mariaDB 10.5.29. Sysbench was used for the benchmark, with 20 tables and 3 million rows per table. The database was pinned to one of the nodes, and the benchmark framework to a different node. No network traffic involved in the benchmark. Damo user space tool was forked and hugepage_mem_bp support added[4]. DAMON was lauched using this command line: sudo ./damo start $(pidof mariadbd) \ --monitoring_nr_regions_range 10 1000 \ --monitoring_intervals 5000 100000 60000000 \ --damos_quota_time 0 --damos_quota_space 128000000 \ --damos_quota_interval 1000 \ --damos_quota_weights 0 1 1 \ --damos_quota_goal hugepage_mem_bp \ --damos_quota_goal_tuner temporal \ --damos_apply_interval 50000 \ --damos_access_rate 0 max --damos_age 50 max \ --damos_action collapse --debug_damon was 1000 to taget 10% hugepage to total memory ratio, or 2500 to target 25%. Tuner was also tested with consistent and temporal. Results ======= After the last timestamp, there was no change in huge page use, and the total huge page to memory consumption ratio barely moved. hugepage_mem_bp: 1000 goal tuner: temporal +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 16945.04297 | 0 | 0 | | 7 | 17008.69531 | 74 | 0.435071583 | | 8 | 17036.40234 | 194 | 1.138738074 | | 9 | 17017.01563 | 314 | 1.845211916 | | 10 | 17029.67969 | 434 | 2.548491856 | | 61 | 17111.30859 | 584 | 3.412947623 | | 120 | 17071.05859 | 694 | 4.065360072 | | 180 | 17133.88281 | 804 | 4.692456513 | | 203 | 17088.16406 | 916 | 5.360435426 | | 204 | 17126.34766 | 1046 | 6.107548562 | | 205 | 17093.84375 | 1176 | 6.879669764 | | 206 | 17142.77734 | 1298 | 7.571701913 | | 209 | 17149.17969 | 1686 | 9.831374041 | | 210 | 17097.30859 | 1754 | 10.25892462 | +-----------+----------------+----------------+----------------------+ hugepage_mem_bp: 1000 goal tuner: consistent +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 16955.24609 | 0 | 0 | | 34 | 17039.71875 | 106 | 0.622075995 | | 78 | 17009.47656 | 554 | 3.257007927 | | 90 | 17048.92188 | 596 | 3.495822225 | | 150 | 17092.90625 | 706 | 4.130368409 | | 180 | 17053.08984 | 764 | 4.480126517 | | 233 | 17100.50391 | 1496 | 8.748280216 | | 239 | 17098.89063 | 2216 | 12.95990511 | | 240 | 17135.44531 | 2334 | 13.62088908 | | 245 | 17132.55078 | 2932 | 17.11362212 | | 246 | 17117.95313 | 3052 | 17.82923448 | | 250 | 17163.12109 | 3532 | 20.57900763 | +-----------+----------------+----------------+----------------------+ hugepage_mem_bp: 2500 goal tuner: temporal +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 17010.31641 | 0 | 0 | | 9 | 17063.6875 | 50 | 0.2930199 | | 10 | 17051.75781 | 170 | 0.996964664 | | 60 | 17133.85547 | 572 | 3.338419663 | | 90 | 17192.07813 | 626 | 3.641211932 | | 120 | 17221.44531 | 682 | 3.960178647 | | 181 | 17199.76172 | 790 | 4.593086886 | | 208 | 17222.77734 | 1206 | 7.002354939 | | 214 | 17245.17969 | 1904 | 11.04076637 | | 215 | 17240.45703 | 2024 | 11.73982799 | | 220 | 17234.79688 | 2624 | 15.22501262 | | 228 | 17222.83594 | 3584 | 20.80958103 | | 231 | 17247.55469 | 3944 | 22.86700968 | | 235 | 17229.37109 | 4424 | 25.67708349 | +-----------+----------------+----------------+----------------------+ hugepage_mem_bp: 1000 goal tuner: consist +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 17125.85156 | 0 | 0 | | 38 | 17081.23438 | 76 | 0.444932716 | | 39 | 17133.11719 | 196 | 1.143983304 | | 40 | 17119.83984 | 316 | 1.84581166 | | 60 | 17109.72656 | 554 | 3.237924335 | | 90 | 17164.11328 | 628 | 3.65879664 | | 180 | 17177.66016 | 792 | 4.610639591 | | 220 | 17180.86719 | 1378 | 8.020549749 | | 226 | 17187.82031 | 1980 | 11.51978531 | | 233 | 17143.48438 | 2818 | 16.4377319 | | 240 | 17137.38281 | 3656 | 21.33347921 | | 250 | 17175.5 | 4856 | 28.27283049 | | 260 | 17199.66406 | 6056 | 35.20999002 | | 270 | 17203.98438 | 7254 | 42.16465118 | | 275 | 17207.21875 | 7762 | 45.10897498 | +-----------+----------------+----------------+----------------------+ More detailed tables are provided here[5] From this, we can conclude that the huge page autotuner works fine, achieving the target. When using consistent autotuner, it actually over-achieves the target, which is expected, since quota esz_bp is not set to 0 to cap the DAMOS policy. Patches Sequence ================ Patch 1 -> Introduce DAMOS_QUOTA_HUGEPAGE_MEM_BP and autotuning Patch 2 -> sysfs support for the new quota goal Patch 3 -> Document hugepage_mem_bp parameter This patch (of 3): Introduce DAMOS_QUOTA_HUGEPAGE_MEM_BP auto tuning. Add a new DAMOS quota goal metric to measure the amount of huge page consumption to total memory consumption ratio. Vmstat may lag, which in some cases may lead to NR_FREE_PAGES being greater than or equal to the amount of RAM in the system. A guard is added to avoid the extremely unlikely case [6]. In the case, return 100% (10000 bp). Link: https://lore.kernel.org/20260831144732.80910-1-sj@kernel.org Link: https://lore.kernel.org/20260831144732.80910-2-sj@kernel.org Link: https://dl.acm.org/doi/pdf/10.1145/3307650.3322227 [1] Link: https://lore.kernel.org/e67f05ad-dbb9-45e6-ba30-b167a99ac67d@huawei-partners.com [2] Link: https://lore.kernel.org/20260616150316.580819-3-gutierrez.asier@huawei-partners.com [3] Link: https://github.com/asierHuawei/damo/commit/79ae1a4ab1c012a7161db85a000d14f08fa36736 [4] Link: https://lore.kernel.org/all/03f678dd-9ef3-4b97-b753-c2e4554c5159@huawei-partners.com/ [5] Link: https://lore.kernel.org/all/20260715151615.99767-1-sj@kernel.org/ [6] Signed-off-by: Asier Gutierrez Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/damon.h | 2 ++ mm/damon/core.c | 19 +++++++++++++++++++ 2 files changed, 21 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 6993dca0f355cc..955b9f614e5bcd 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -153,6 +153,7 @@ enum damos_action { * @DAMOS_QUOTA_INACTIVE_MEM_BP: Inactive to total LRU memory ratio. * @DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: Scheme-eligible memory ratio of a * node in basis points (0-10000). + * @DAMOS_QUOTA_HUGEPAGE_MEM_BP: Huge page to total used memory ratio. * @NR_DAMOS_QUOTA_GOAL_METRICS: Number of DAMOS quota goal metrics. * * Metrics equal to larger than @NR_DAMOS_QUOTA_GOAL_METRICS are unsupported. @@ -167,6 +168,7 @@ enum damos_quota_goal_metric { DAMOS_QUOTA_ACTIVE_MEM_BP, DAMOS_QUOTA_INACTIVE_MEM_BP, DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, + DAMOS_QUOTA_HUGEPAGE_MEM_BP, NR_DAMOS_QUOTA_GOAL_METRICS, }; diff --git a/mm/damon/core.c b/mm/damon/core.c index 891f851bcbbf6f..3632b06f1db58f 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2988,6 +2988,22 @@ static unsigned int damos_get_in_active_mem_bp(bool active_ratio) return mult_frac(inactive, 10000, total); } +static unsigned int damos_hugepage_mem_bp(void) +{ + unsigned long thp, total_pages, free_pages; + + total_pages = totalram_pages(); + free_pages = global_zone_page_state(NR_FREE_PAGES); + + if (total_pages <= free_pages) + return 10000; + + thp = global_node_page_state(NR_ANON_THPS) + + global_node_page_state(NR_SHMEM_THPS) + + global_node_page_state(NR_FILE_THPS); + return mult_frac(thp, 10000, total_pages - free_pages); +} + static void damos_set_quota_goal_current_value(struct damon_ctx *c, struct damos *s, struct damos_quota_goal *goal) { @@ -3019,6 +3035,9 @@ static void damos_set_quota_goal_current_value(struct damon_ctx *c, goal->current_value = damos_get_node_eligible_mem_bp(c, s, goal->nid); break; + case DAMOS_QUOTA_HUGEPAGE_MEM_BP: + goal->current_value = damos_hugepage_mem_bp(); + break; default: break; } From ce442c232fdeeb342126a1220b5c9821bc94b5a0 Mon Sep 17 00:00:00 2001 From: Asier Gutierrez Date: Mon, 31 Aug 2026 07:47:29 -0700 Subject: [PATCH 660/857] mm/damon/sysfs: support hugepage_mem_bp quota goal metric DAMOS has a new autotune policy metric: DAMOS_QUOTA_HUGEPAGE_MEM_BP. This patch exposes DAMOS_QUOTA_HUGEPAGE_MEM_BP through sysfs. Add the "hugepage_mem_bp" to the sysfs-schemes interface. Link: https://lore.kernel.org/20260831144732.80910-3-sj@kernel.org Signed-off-by: Asier Gutierrez Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/sysfs-schemes.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 32f495a96b17a8..d9b81d7b5910ed 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -1269,6 +1269,10 @@ struct damos_sysfs_qgoal_metric_name damos_sysfs_qgoal_metric_names[] = { .metric = DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, .name = "node_eligible_mem_bp", }, + { + .metric = DAMOS_QUOTA_HUGEPAGE_MEM_BP, + .name = "hugepage_mem_bp", + }, }; static ssize_t target_metric_show(struct kobject *kobj, From d7cf2d97dd81a16e114666b1b4290e9eaddc1f4d Mon Sep 17 00:00:00 2001 From: Asier Gutierrez Date: Mon, 31 Aug 2026 07:47:30 -0700 Subject: [PATCH 661/857] Docs/mm/damon/design: cocument hugepage_mem_bp target metric Document hugepage_mem_bp metric exposed by sysfs. Link: https://lore.kernel.org/20260831144732.80910-4-sj@kernel.org Signed-off-by: Asier Gutierrez Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/mm/damon/design.rst | 2 ++ 1 file changed, 2 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index d1dd9050ebf40b..63cbb7b536da20 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -713,6 +713,8 @@ mechanism tries to make ``current_value`` of ``target_metric`` be same to bp (1/10,000). - ``node_eligible_mem_bp``: Scheme target access pattern-eligible memory ratio of a node in bp (1/10,000). +- ``hugepage_mem_bp``: Total huge page to total used memory ratio in bp + (1/10,000). ``nid`` is optionally required for ``node_mem_used_bp``, ``node_mem_free_bp``, ``node_memcg_used_bp``, ``node_memcg_free_bp`` and ``node_eligible_mem_bp`` to From a80729eea91375af1b8df333abba797b54aca4fd Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Mon, 31 Aug 2026 07:26:03 -0700 Subject: [PATCH 662/857] mm/damon/core: remove declaration of __damon_commit_ctx() Patch series "mm/damon: misc cleanups". Cleanup the code, tests and samples for clarifications and readability. The patches are individually sent by the authors. I'm reposting those as one series for convenience of handling. For this reason, changelog is on each patch's commentary section. This patch (of 7): __damon_commit_ctx() was added by commit b1471afe4d10 ("mm/damon/core: do parameter testing commit on damon_start()") but is actually not needed. Remove it. Link: https://lore.kernel.org/20260831142611.77572-1-sj@kernel.org Link: https://lore.kernel.org/20260831142611.77572-2-sj@kernel.org Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Shuah Khan Cc: Zenghui Yu Cc: Enze Li Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: zhaozhengzhuo Signed-off-by: Andrew Morton --- mm/damon/core.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 3632b06f1db58f..718195fbeb686a 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1935,8 +1935,6 @@ static int __damon_start(struct damon_ctx *ctx) return err; } -static int __damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src); - /** * damon_start() - Starts the monitorings for a given group of contexts. * @ctxs: an array of the pointers for contexts to start monitoring From 15d3028924447d5ae70889b6b6cd945ca93cbe7d Mon Sep 17 00:00:00 2001 From: Enze Li Date: Mon, 31 Aug 2026 07:26:04 -0700 Subject: [PATCH 663/857] mm/damon/core: introduce damon_set_target_pid() The logic that finds the struct pid for a given pid number and assigns it to a damon_target is duplicated in multiple places. Including damon_sysfs_add_target() of mm/damon/sysfs.c and the start functions of the two sample modules, samples/damon/wsse.c and samples/damon/prcl.c. Add a function that does the work, and replace the duplicated code in the places with calls to the function. Link: https://lore.kernel.org/20260831142611.77572-3-sj@kernel.org Signed-off-by: Enze Li Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo Signed-off-by: Andrew Morton --- include/linux/damon.h | 1 + mm/damon/core.c | 12 ++++++++++++ mm/damon/sysfs.c | 6 ++---- samples/damon/prcl.c | 5 +---- samples/damon/wsse.c | 5 +---- 5 files changed, 17 insertions(+), 12 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 955b9f614e5bcd..7b1b6050a8286f 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1057,6 +1057,7 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src); struct damon_target *damon_new_target(void); void damon_add_target(struct damon_ctx *ctx, struct damon_target *t); +int damon_set_target_pid(struct damon_target *t, int pid); void damon_free_target(struct damon_target *t); void damon_destroy_target(struct damon_target *t, struct damon_ctx *ctx); unsigned int damon_nr_regions(struct damon_target *t); diff --git a/mm/damon/core.c b/mm/damon/core.c index 718195fbeb686a..3f89cfdf5f0221 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -10,6 +10,7 @@ #include #include #include +#include #include #include #include @@ -795,6 +796,17 @@ void damon_add_target(struct damon_ctx *ctx, struct damon_target *t) list_add_tail(&t->list, &ctx->adaptive_targets); } +/* + * Assign the struct pid of the given pid number to the given target. + */ +int damon_set_target_pid(struct damon_target *t, int pid) +{ + t->pid = find_get_pid(pid); + if (!t->pid) + return -EINVAL; + return 0; +} + static void damon_del_target(struct damon_target *t) { list_del(&t->list); diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index e3858ffab4b227..3c81b4c91ac0dd 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -3,7 +3,6 @@ * DAMON sysfs Interface */ -#include #include #include @@ -2035,9 +2034,8 @@ static int damon_sysfs_add_target(struct damon_sysfs_target *sys_target, return -ENOMEM; damon_add_target(ctx, t); if (damon_target_has_pid(ctx)) { - t->pid = find_get_pid(sys_target->pid); - if (!t->pid) - /* caller will destroy targets */ + /* caller will destroy targets */ + if (damon_set_target_pid(t, sys_target->pid)) return -EINVAL; } t->obsolete = sys_target->obsolete; diff --git a/samples/damon/prcl.c b/samples/damon/prcl.c index 842099bd622861..83ddf12811d57f 100644 --- a/samples/damon/prcl.c +++ b/samples/damon/prcl.c @@ -32,7 +32,6 @@ module_param_cb(enabled, &enabled_param_ops, &enabled, 0600); MODULE_PARM_DESC(enabled, "Enable or disable DAMON_SAMPLE_PRCL"); static struct damon_ctx *ctx; -static struct pid *target_pidp; static int damon_sample_prcl_repeat_call_fn(void *data) { @@ -79,12 +78,10 @@ static int damon_sample_prcl_start(void) return -ENOMEM; } damon_add_target(ctx, target); - target_pidp = find_get_pid(target_pid); - if (!target_pidp) { + if (damon_set_target_pid(target, target_pid)) { damon_destroy_ctx(ctx); return -EINVAL; } - target->pid = target_pidp; scheme = damon_new_scheme( &(struct damos_access_pattern) { diff --git a/samples/damon/wsse.c b/samples/damon/wsse.c index 37fd5da2015885..53944aea8428ea 100644 --- a/samples/damon/wsse.c +++ b/samples/damon/wsse.c @@ -33,7 +33,6 @@ module_param_cb(enabled, &enabled_param_ops, &enabled, 0600); MODULE_PARM_DESC(enabled, "Enable or disable DAMON_SAMPLE_WSSE"); static struct damon_ctx *ctx; -static struct pid *target_pidp; static int damon_sample_wsse_repeat_call_fn(void *data) { @@ -79,12 +78,10 @@ static int damon_sample_wsse_start(void) return -ENOMEM; } damon_add_target(ctx, target); - target_pidp = find_get_pid(target_pid); - if (!target_pidp) { + if (damon_set_target_pid(target, target_pid)) { damon_destroy_ctx(ctx); return -EINVAL; } - target->pid = target_pidp; err = damon_start(&ctx, 1, true); if (err) { From 7004c36d7b858dfb078d89041c5978345ede98fc Mon Sep 17 00:00:00 2001 From: Li Youhong Date: Mon, 31 Aug 2026 07:26:05 -0700 Subject: [PATCH 664/857] mm/damon/ops-common: factor out damon_putback_folio_list() The putback loop is duplicated in damon_migrate_folio_list() and on the invalid-nid path of damon_migrate_pages(). Factor it into a small helper for readability. No functional change. Link: https://lore.kernel.org/20260831142611.77572-4-sj@kernel.org Signed-off-by: Li Youhong Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: Enze Li Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo Signed-off-by: Andrew Morton --- mm/damon/ops-common.c | 30 +++++++++++++++--------------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index fbda70d8ea4d05..7a2e40bc7baede 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -330,6 +330,19 @@ static unsigned int __damon_migrate_folio_list( return nr_succeeded; } +static void damon_putback_folio_list(struct list_head *folio_list) +{ + struct folio *folio; + + while (!list_empty(folio_list)) { + folio = lru_to_folio(folio_list); + list_del(&folio->lru); + node_stat_sub_folio(folio, NR_ISOLATED_ANON + + folio_is_file_lru(folio)); + folio_putback_lru(folio); + } +} + static unsigned int damon_migrate_folio_list(struct list_head *folio_list, struct pglist_data *pgdat, int target_nid) @@ -371,13 +384,7 @@ static unsigned int damon_migrate_folio_list(struct list_head *folio_list, list_splice(&ret_folios, folio_list); - while (!list_empty(folio_list)) { - folio = lru_to_folio(folio_list); - list_del(&folio->lru); - node_stat_sub_folio(folio, NR_ISOLATED_ANON + - folio_is_file_lru(folio)); - folio_putback_lru(folio); - } + damon_putback_folio_list(folio_list); return nr_migrated; } @@ -394,14 +401,7 @@ unsigned long damon_migrate_pages(struct list_head *folio_list, int target_nid) if (target_nid < 0 || target_nid >= MAX_NUMNODES || !node_state(target_nid, N_MEMORY)) { - while (!list_empty(folio_list)) { - struct folio *folio = lru_to_folio(folio_list); - - list_del(&folio->lru); - node_stat_sub_folio(folio, NR_ISOLATED_ANON + - folio_is_file_lru(folio)); - folio_putback_lru(folio); - } + damon_putback_folio_list(folio_list); return nr_migrated; } From 7131fc2a6fe5fbbb800516e72666c14844ae38ad Mon Sep 17 00:00:00 2001 From: Hari Mishal Date: Mon, 31 Aug 2026 07:26:06 -0700 Subject: [PATCH 665/857] selftests/damon/sysfs.py: clean up sh processes used for obsolete_target test The obsolete_target test spawns three sh processes and uses their pids as DAMON monitoring targets. These processes are never terminated or waited on, so they are left running (or become zombies) as orphaned children after the test program exits. Terminate each process and communicate() with it after the targets are no longer needed, so it exits and gets reaped instead of being leaked. Link: https://lore.kernel.org/20260831142611.77572-5-sj@kernel.org Signed-off-by: Hari Mishal Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Enze Li Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo Signed-off-by: Andrew Morton --- tools/testing/selftests/damon/sysfs.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs.py b/tools/testing/selftests/damon/sysfs.py index 3ffa054b63867d..88a26422ff44cf 100755 --- a/tools/testing/selftests/damon/sysfs.py +++ b/tools/testing/selftests/damon/sysfs.py @@ -385,6 +385,10 @@ def main(): assert_ctxs_committed(kdamonds) kdamonds.stop() + for proc in (proc1, proc2, proc3): + proc.terminate() + proc.communicate() + test_memcg_filter_memcg_path_staging() if __name__ == '__main__': From 29770f2f629acfbfcc1a453f55fa3172d56b93a0 Mon Sep 17 00:00:00 2001 From: Jaeyeon Lee Date: Mon, 31 Aug 2026 07:26:07 -0700 Subject: [PATCH 666/857] mm/damon/tests: use scoped_guard() for damon_test_ops_registration Replace manual mutex_lock() and mutex_unlock() calls with the scoped_guard() macro. This simplifies the code, improves readability, and ensures that the lock is automatically released when the scope ends, preventing potential lock leaks in the future. Link: https://lore.kernel.org/20260831142611.77572-6-sj@kernel.org Signed-off-by: Jaeyeon Lee Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: Enze Li Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo Signed-off-by: Andrew Morton --- mm/damon/tests/core-kunit.h | 21 ++++++++++----------- 1 file changed, 10 insertions(+), 11 deletions(-) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index b1ca4c8e03f091..7071ec277b0072 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -441,17 +441,17 @@ static void damon_test_ops_registration(struct kunit *test) KUNIT_EXPECT_EQ(test, damon_select_ops(c, NR_DAMON_OPS), -EINVAL); /* Registration should success after unregistration */ - mutex_lock(&damon_ops_lock); - bak = damon_registered_ops[DAMON_OPS_VADDR]; - damon_registered_ops[DAMON_OPS_VADDR] = (struct damon_operations){}; - mutex_unlock(&damon_ops_lock); + scoped_guard(mutex, &damon_ops_lock) { + bak = damon_registered_ops[DAMON_OPS_VADDR]; + damon_registered_ops[DAMON_OPS_VADDR] = + (struct damon_operations){}; + } ops.id = DAMON_OPS_VADDR; KUNIT_EXPECT_EQ(test, damon_register_ops(&ops), 0); - mutex_lock(&damon_ops_lock); - damon_registered_ops[DAMON_OPS_VADDR] = bak; - mutex_unlock(&damon_ops_lock); + scoped_guard(mutex, &damon_ops_lock) + damon_registered_ops[DAMON_OPS_VADDR] = bak; /* Check double-registration failure again */ KUNIT_EXPECT_EQ(test, damon_register_ops(&ops), -EINVAL); @@ -459,10 +459,9 @@ static void damon_test_ops_registration(struct kunit *test) damon_destroy_ctx(c); if (need_cleanup) { - mutex_lock(&damon_ops_lock); - damon_registered_ops[DAMON_OPS_VADDR] = - (struct damon_operations){}; - mutex_unlock(&damon_ops_lock); + scoped_guard(mutex, &damon_ops_lock) + damon_registered_ops[DAMON_OPS_VADDR] = + (struct damon_operations){}; } } From d04036d67bbb678a594f8c625004e64f9231188e Mon Sep 17 00:00:00 2001 From: zhaozhengzhuo Date: Mon, 31 Aug 2026 07:26:08 -0700 Subject: [PATCH 667/857] selftests/damon: prevent remaining cross-object state pollution _damon_sysfs.py defines constructors with mutable default arguments, including DamosAccessPattern(), DamosQuota(), DamosWatermarks(), DamosDests(), IntervalsGoal(), and empty lists. Default arguments are evaluated once at function definition time. Damos() instances created without explicit arguments therefore share the same DamosQuota(), and the other default-constructed sub-objects and lists are shared in the same way. The sub-objects keep back-pointers to their owner scheme, so constructing the second Damos() rebinds the shared quota's scheme pointer to the second object. An item appended to one object's default contexts or filters list is also visible from other default-constructed objects. The shared state can corrupt test configurations. DamosQuota.sysfs_dir() derives the sysfs directory from its scheme pointer, so operating on the first scheme's default quota may write to the second scheme's directory. The wrong values often match the defaults, so tests still pass, but the behavior depends on object creation order. Commit 8319dadcbd81 ("selftests/damon: prevent cross-context state pollution in DamonCtx") fixed the same pattern in DamonCtx only. Fix the remaining constructors by defaulting to None and creating fresh objects or lists inside each constructor. Explicit arguments keep their previous behavior. Link: https://lore.kernel.org/20260831142611.77572-7-sj@kernel.org Signed-off-by: zhaozhengzhuo Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: Enze Li Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Signed-off-by: Andrew Morton --- tools/testing/selftests/damon/_damon_sysfs.py | 38 ++++++++++++++----- 1 file changed, 28 insertions(+), 10 deletions(-) diff --git a/tools/testing/selftests/damon/_damon_sysfs.py b/tools/testing/selftests/damon/_damon_sysfs.py index e6a2265d721e8b..f604b7d6530b3c 100644 --- a/tools/testing/selftests/damon/_damon_sysfs.py +++ b/tools/testing/selftests/damon/_damon_sysfs.py @@ -321,8 +321,10 @@ class DamosFilters: filters = None scheme = None # owner scheme - def __init__(self, name, filters=[]): + def __init__(self, name, filters=None): self.name = name + if filters is None: + filters = [] self.filters = filters for idx, filter_ in enumerate(self.filters): filter_.idx = idx @@ -368,7 +370,9 @@ class DamosDests: dests = None scheme = None # owner scheme - def __init__(self, dests=[]): + def __init__(self, dests=None): + if dests is None: + dests = [] self.dests = dests for idx, dest in enumerate(self.dests): dest.idx = idx @@ -426,15 +430,21 @@ class Damos: stats = None tried_regions = None - def __init__(self, action='stat', access_pattern=DamosAccessPattern(), - quota=DamosQuota(), watermarks=DamosWatermarks(), - core_filters=[], ops_filters=[], filters=[], target_nid=0, - dests=DamosDests(), apply_interval_us=0): + def __init__(self, action='stat', access_pattern=None, quota=None, + watermarks=None, core_filters=None, ops_filters=None, + filters=None, target_nid=0, dests=None, + apply_interval_us=0): self.action = action + if access_pattern is None: + access_pattern = DamosAccessPattern() self.access_pattern = access_pattern self.access_pattern.scheme = self + if quota is None: + quota = DamosQuota() self.quota = quota self.quota.scheme = self + if watermarks is None: + watermarks = DamosWatermarks() self.watermarks = watermarks self.watermarks.scheme = self @@ -448,6 +458,8 @@ def __init__(self, action='stat', access_pattern=DamosAccessPattern(), self.filters.scheme = self self.target_nid = target_nid + if dests is None: + dests = DamosDests() self.dests = dests self.dests.scheme = self @@ -568,10 +580,12 @@ class DamonAttrs: context = None def __init__(self, sample_us=5000, aggr_us=100000, - intervals_goal=IntervalsGoal(), update_us=1000000, - min_nr_regions=10, max_nr_regions=1000): + intervals_goal=None, update_us=1000000, min_nr_regions=10, + max_nr_regions=1000): self.sample_us = sample_us self.aggr_us = aggr_us + if intervals_goal is None: + intervals_goal = IntervalsGoal() self.intervals_goal = intervals_goal self.intervals_goal.attrs = self self.update_us = update_us @@ -703,7 +717,9 @@ class Kdamond: idx = None # index of this kdamond between siblings kdamonds = None # parent - def __init__(self, contexts=[], refresh_ms=None): + def __init__(self, contexts=None, refresh_ms=None): + if contexts is None: + contexts = [] self.contexts = contexts self.refresh_ms = refresh_ms for idx, context in enumerate(self.contexts): @@ -853,7 +869,9 @@ def commit_schemes_quota_goals(self): class Kdamonds: kdamonds = [] - def __init__(self, kdamonds=[]): + def __init__(self, kdamonds=None): + if kdamonds is None: + kdamonds = [] self.kdamonds = kdamonds for idx, kdamond in enumerate(self.kdamonds): kdamond.idx = idx From 6865bb62ea6014cd8531ce842d20497df4ba0f16 Mon Sep 17 00:00:00 2001 From: Enze Li Date: Mon, 31 Aug 2026 07:26:09 -0700 Subject: [PATCH 668/857] samples/damon/mtier: add comment for struct region_range The mtier sample defines a local struct region_range using phys_addr_t instead of damon_addr_range which uses unsigned long. Add a comment explaining the rationale: on 32-bit systems with more than 4GiB memory, phys_addr_t will be 64-bit while unsigned long is 32-bit. Link: https://lore.kernel.org/20260831142611.77572-8-sj@kernel.org Signed-off-by: Enze Li Signed-off-by: SJ Park Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo Signed-off-by: Andrew Morton --- samples/damon/mtier.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/samples/damon/mtier.c b/samples/damon/mtier.c index d1123ebbfab906..bea45c87cc9bed 100644 --- a/samples/damon/mtier.c +++ b/samples/damon/mtier.c @@ -52,6 +52,11 @@ module_param(detect_node_addresses, bool, 0600); static struct damon_ctx *ctxs[2]; +/* + * Use phys_addr_t instead of damon_addr_range (unsigned long) for physical + * addresses. On 32-bit systems with more than 4GB memory, phys_addr_t will + * be 64-bit while unsigned long is 32-bit. + */ struct region_range { phys_addr_t start; phys_addr_t end; From c6093edf00825f312d15be572149ce8edddae764 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:32 +0800 Subject: [PATCH 669/857] mm: fix stale ZONE_DEVICE refcount comment Patch series "mm: optimize zone-device memmap initialization", v11. memmap_init_zone_device() can take a noticeable amount of time when large pmem namespaces are bound or rebound, because it initializes nearly identical struct page descriptors one PFN at a time. This series reduces that ZONE_DEVICE memmap initialization overhead by reusing prepared struct page templates and, on x86, using memcpy_nontemporal() for the template copy path. The main target is large fsdax/devdax pmem configurations, where the cost of initializing the memmap shows up directly in nd_pmem/dax_pmem bind and rebind latency. This matters because the cost is paid in the synchronous probe/bind path for large DAX/PMEM ZONE_DEVICE mappings. Userspace workflows such as provisioning or reconfiguring nd_pmem/dax_pmem namespaces, bringing hot-added PMEM-backed capacity online, and recovering or rebinding a device after driver or device changes all wait for this initialization to finish. Reducing this cost will yield benefits as lower user-visible provisioning, hot-add, recovery, and rebind latency for large DAX/PMEM devices. Patches 1-2 are preparatory cleanups and helper extraction. Patches 3-4 add the template-copy path for head pages and compound tails. Patch 5 introduces memcpy_nontemporal(). Patch 6 switches the ZONE_DEVICE template-copy path over to memcpy_nontemporal(). Patch 7 extends the x86 fixed-size memcpy_flushcache() inline cases used by the x86 memcpy_nontemporal() backend for struct page sized copies. Architectures without a specialized memcpy_nontemporal() backend fall back to memcpy(), so the generic template-copy optimization remains available without arch-specific support. On x86, memcpy_nontemporal() maps to the existing memcpy_flushcache() backend and can use the fixed-size MOVNTI paths added by this series for struct page sized copies. memcpy_nontemporal() is only a copy primitive. It does not imply a drain or a publication barrier. Callers that use it before a producer-consumer or device-visible handoff must provide the required ordering. The ZONE_DEVICE template-copy path uses it only while initializing struct page metadata, so the copy primitive itself does not grow a separate drain contract. The numbers below measure the time spent in memmap_init_zone_device() during driver bind/rebind. They are not measurements of the full nd_pmem or dax_pmem bind/rebind operation. Tested in an x86_64 QEMU/KVM VM with a 100 GB fsdax namespace device configured with map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem drivers 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of nd_pmem rebinds: 221.07 ms Average of dax_pmem rebinds: 191.20 ms With this series applied: Average of nd_pmem rebinds: 71.93 ms Average of dax_pmem rebinds: 87.37 ms This reduces the average memmap initialization time measured during rebind by about 67.5% for nd_pmem and 54.3% for dax_pmem. As an additional x86_64 data point, I also ran measurements on the same physical host with a 100 GB PMEM region created via the memmap= kernel command line, configured as fsdax and devdax namespaces with map=dev and 2 MiB alignment. For brevity, the individual patches keep only the VM results rather than including a second set of physical-host measurements throughout the series. The physical-host numbers below are included only as supplemental evidence that the same optimization also provides a similar benefit on a non-virtualized system. Test procedure: Reconfigure the namespace mode, rebind the nd_pmem or dax_pmem driver 30 times, and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base (v7.3-rc1): nd_pmem / fsdax: 205.90 ms dax_pmem / devdax: 225.43 ms With this series applied: nd_pmem / fsdax: 69.13 ms dax_pmem / devdax: 90.67 ms This reduces the measured memmap initialization time during rebind by about 66.4% for nd_pmem and 59.8% for dax_pmem on that setup, which is broadly consistent with the VM results above. As another supplemental data point, I measured the test_hmm.ko module on the same physical x86_64 host, using the test_hmm.ko setup from the previous discussion that times ten 64 GB memremap_pages()/memunmap_pages() iterations during module insertion[1]. By default, module insertion initializes two DEVICE_PRIVATE dmirror devices, so two avg memremap values are reported; each value is the average for one 64 GB chunk. This is not the primary target workload of the series, but it exercises the same large ZONE_DEVICE memmap initialization path and shows the same direction of improvement. Base (v7.3-rc1): avg memremap reported during module insertion: 116500596 ns, 116438028 ns With this series applied: avg memremap reported during module insertion: 46953088 ns, 46428399 ns This corresponds to about a 59.9% reduction based on the mean of the reported values, which is again consistent with the pmem bind/rebind results above. I also include an arm64 data point for the generic template-copy part. It was measured on an arm64 QEMU virt VM with 64 KB pages and a 100 GB ACPI NVDIMM sparse backend. This setup does not use the x86 MOVNTI fast paths, so it exercises the architecture-independent part of the optimization. For devdax, 2 MiB alignment is rejected in this 64 KB page setup, so the devdax namespace was tested with the supported default 512 MiB alignment. Base (v7.3-rc1): Average of rebinds for nd_pmem driver: 27.93 ms Average of rebinds for dax_pmem driver: 27.87 ms With this series applied: Average of rebinds for nd_pmem driver: 14.53 ms Average of rebinds for dax_pmem driver: 16.27 ms This reduces the average memmap initialization time measured during rebind by about 48.0% for nd_pmem and 41.6% for dax_pmem on that arm64 VM setup. Since this arm64 setup does not use the x86 MOVNTI fast paths, the result also suggests that the generic template-copy optimization can benefit architectures without an architecture-specific memcpy_nontemporal() backend. This patch (of 7): The comment in __init_zone_device_page() still uses the old MEMORY_TYPE_* names and implies that FS_DAX pages regain a refcount of 1 in the free path. That no longer matches the code. Update the comment to describe the current policy correctly: MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free path, while the remaining ZONE_DEVICE types start from 0 here and raise the count again when the allocator or driver hands the page out. No functional change intended. Link: https://lore.kernel.org/20260831111638.76012-1-lizhe.67@bytedance.com Link: https://lore.kernel.org/20260831111638.76012-2-lizhe.67@bytedance.com Link: https://lore.kernel.org/all/aiEoByaQdRR3xtM5@nvdebian.thelocal/ [1] Signed-off-by: Li Zhe Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Alistair Popple Reviewed-by: Muchun Song Reviewed-by: Mike Rapoport (Microsoft) Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: Ingo Molnar Cc: Kees Cook Signed-off-by: Andrew Morton --- mm/mm_init.c | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index e2a16d83363557..ac9cf1a59d6207 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1012,13 +1012,9 @@ static void __ref __init_zone_device_page(struct page *page, unsigned long pfn, page->zone_device_data = NULL; /* - * ZONE_DEVICE pages other than MEMORY_TYPE_GENERIC are released - * directly to the driver page allocator which will set the page count - * to 1 when allocating the page. - * - * MEMORY_TYPE_GENERIC and MEMORY_TYPE_FS_DAX pages automatically have - * their refcount reset to one whenever they are freed (ie. after - * their refcount drops to 0). + * MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free + * path. The remaining ZONE_DEVICE types start from 0 here and raise + * the count again when the allocator or driver hands the page out. */ switch (pgmap->type) { case MEMORY_DEVICE_FS_DAX: From 90c516a00c84c663a81ccc0e19517d8fef2b09f3 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:33 +0800 Subject: [PATCH 670/857] mm: add a set_page_section_from_pfn() helper Callers that want to update section bits from a PFN currently need to open-code: set_page_section(page, pfn_to_section_nr(pfn)); and guard that sequence with #ifdef SECTION_IN_PAGE_FLAGS. Add set_page_section_from_pfn() to wrap that update in one place. When section bits are stored in page flags, the helper derives the section number from the PFN and updates the page flags. Otherwise keep it as a no-op so callers can use one helper without open-coding SECTION_IN_PAGE_FLAGS. Convert set_page_links() to use the new helper so later ZONE_DEVICE fast-path patches can also update section bits without open-coding SECTION_IN_PAGE_FLAGS at each callsite. This keeps the PFN-to-section translation local to the configurations that actually store section bits in struct page flags, and avoids exposing that detail to generic callers. No functional change intended. Link: https://lore.kernel.org/20260831111638.76012-3-lizhe.67@bytedance.com Signed-off-by: Li Zhe Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Muchun Song Reviewed-by: Balbir Singh Cc: Alistair Popple Cc: Arnd Bergmann Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Signed-off-by: Andrew Morton --- include/linux/mm.h | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index a9fbe26536f450..1b28e6fc8d5dd1 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -2632,12 +2632,23 @@ static inline void set_page_section(struct page *page, unsigned long section) page->flags.f |= (section & SECTIONS_MASK) << SECTIONS_PGSHIFT; } +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ + set_page_section(page, pfn_to_section_nr(pfn)); +} + static inline unsigned long memdesc_section(const memdesc_flags_t *mdf) { ASSERT_EXCLUSIVE_BITS(mdf->f, SECTIONS_MASK << SECTIONS_PGSHIFT); return (mdf->f >> SECTIONS_PGSHIFT) & SECTIONS_MASK; } #else /* !SECTION_IN_PAGE_FLAGS */ +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ +} + static inline unsigned long memdesc_section(const memdesc_flags_t *mdf) { return 0; @@ -2860,9 +2871,7 @@ static inline void set_page_links(struct page *page, enum zone_type zone, { set_page_zone(page, zone); set_page_node(page, node); -#ifdef SECTION_IN_PAGE_FLAGS - set_page_section(page, pfn_to_section_nr(pfn)); -#endif + set_page_section_from_pfn(page, pfn); } /** From ee90c18d5bbeb458cf71970aff51639c553c7d75 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:34 +0800 Subject: [PATCH 671/857] mm: add a template-based fast path for zone-device page init memmap_init_zone_device() repeats nearly identical head-page initialization for each PFN. Initialize the first real ZONE_DEVICE head page through the existing path, copy that final state into a reusable template, refresh the PFN-dependent fields in that template before each copy, and copy it into the remaining destination pages. Use the template path unconditionally. The page_ref_set tracepoint is primarily a debugging aid, while this code is still initializing struct pages before they are handed out. From the perspective of users of those pages, the initialization-time refcount transitions are not part of the observable page lifetime. This means page_ref_set will no longer observe every initialization-time refcount assignment for copied ZONE_DEVICE head pages. The impact is controlled because the final initialized struct page state is unchanged, and keeping a separate non-template path only for this local tracepoint observability would add complexity to the common path. This patch accelerates head-page initialization. The pfns_per_compound == 1 case gets the full benefit here, compound tails are handled in the next patch. Tested in a VM with a 100 GB fsdax namespace device configured with map=dev on Intel Ice Lake server. This test exercises the nd_pmem rebind path (pfns_per_compound == 1). Test procedure: Rebind the nd_pmem driver 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of rebinds for nd_pmem driver: 221.07 ms With this patch and its prerequisites applied: Average of rebinds for nd_pmem driver: 155.00 ms This reduces the average memmap initialization time measured during rebind from 221.07 ms to 155.00 ms, or about 29.9%. Link: https://lore.kernel.org/20260831111638.76012-4-lizhe.67@bytedance.com Signed-off-by: Li Zhe Reviewed-by: Mike Rapoport (Microsoft) Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Muchun Song Signed-off-by: Andrew Morton --- mm/mm_init.c | 38 +++++++++++++++++++++++++++++++++++--- 1 file changed, 35 insertions(+), 3 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index ac9cf1a59d6207..be9a588b0f540b 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1029,6 +1029,17 @@ static void __ref __init_zone_device_page(struct page *page, unsigned long pfn, } } +static void zone_device_page_init_from_template(struct page *page, + unsigned long pfn, struct page *template) +{ + set_page_section_from_pfn(template, pfn); +#ifdef WANT_PAGE_VIRTUAL + if (!is_highmem_idx(ZONE_DEVICE)) + set_page_address(template, __va(pfn << PAGE_SHIFT)); +#endif + memcpy(page, template, sizeof(*page)); +} + /* * With compound page geometry and when struct pages are stored in ram most * tail pages are reused. Consequently, the amount of unique struct pages to @@ -1091,6 +1102,8 @@ void __ref memmap_init_zone_device(struct zone *zone, unsigned long zone_idx = zone_idx(zone); unsigned long start = jiffies; int nid = pgdat->node_id; + struct page template; + struct page *page; if (WARN_ON_ONCE(!pgmap || zone_idx != ZONE_DEVICE)) return; @@ -1105,10 +1118,29 @@ void __ref memmap_init_zone_device(struct zone *zone, nr_pages = end_pfn - start_pfn; } - for (pfn = start_pfn; pfn < end_pfn; pfn += pfns_per_compound) { - struct page *page = pfn_to_page(pfn); + if (!nr_pages) + return; - __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + /* + * Seed the reusable head-page template from the first real struct + * page. The normal page-init and refcount helpers must operate on + * a real memmap entry rather than a stack object. + */ + pfn = start_pfn; + page = pfn_to_page(pfn); + __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + memcpy(&template, page, sizeof(*page)); + if (pfns_per_compound != 1) + memmap_init_compound(page, pfn, zone_idx, nid, pgmap, + compound_nr_pages(pfn, altmap, pgmap)); + pfn += pfns_per_compound; + + /* Initialize the remaining head pages from template. */ + for (; pfn < end_pfn; pfn += pfns_per_compound) { + page = pfn_to_page(pfn); + + zone_device_page_init_from_template(page, pfn, + &template); if (IS_ALIGNED(pfn, PAGES_PER_SECTION)) cond_resched(); From 67ff3a088bea90c31c90f7848e83f4329e6b41ac Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:35 +0800 Subject: [PATCH 672/857] mm: extend the template fast path to zone-device compound tails The template fast path from the previous patch only accelerates head pages. Compound tails in memmap_init_compound() still go through the normal initialization path one by one. Build separate head and tail templates and reuse one prepared tail template across the tail pages in a compound range. Head pages preserve the existing refcount policy, while compound tails always start with a refcount of 0 after prep_compound_tail(). This extends the template-copy fast path to pfns_per_compound > 1. Tail-page PFN-dependent fields are refreshed in the reusable tail template before each copy. Do not keep a separate non-template fallback for compound tails either. These pages are still under memmap initialization, and the initialization-time refcount updates are not part of the observable lifetime of pages handed out later. The impact is controlled for the same reason as for head pages. The first tail page still seeds the reusable tail template through the normal tail initialization sequence, and the copied tail pages have the same final initialized state except for the PFN-dependent fields refreshed before each copy. Tested in a VM with a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. This test exercises the dax_pmem rebind path and measures memmap initialization latency. Test procedure: Unbind and rebind the dax_pmem driver 30 times, collect memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of rebinds for dax_pmem driver: 191.20 ms With this patch and its prerequisites applied: Average of rebinds for dax_pmem driver: 176.87 ms This reduces the average memmap initialization time measured during rebind from 191.20 ms to 176.87 ms, or about 7.5%. Link: https://lore.kernel.org/20260831111638.76012-5-lizhe.67@bytedance.com Signed-off-by: Li Zhe Reviewed-by: Mike Rapoport (Microsoft) Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Muchun Song Signed-off-by: Andrew Morton --- mm/mm_init.c | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index be9a588b0f540b..0d83034cfa97b0 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1072,6 +1072,8 @@ static void __ref memmap_init_compound(struct page *head, { unsigned long pfn, end_pfn = head_pfn + nr_pages; unsigned int order = pgmap->vmemmap_shift; + struct page template; + struct page *page; /* * We have to initialize the pages, including setting up page links. @@ -1080,13 +1082,23 @@ static void __ref memmap_init_compound(struct page *head, * the pages in the same go. */ __SetPageHead(head); - for (pfn = head_pfn + 1; pfn < end_pfn; pfn++) { - struct page *page = pfn_to_page(pfn); - __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); - prep_compound_tail(page, head, order); - set_page_count(page, 0); - } + /* + * All tails of the same compound page share the state established by + * prep_compound_tail(). Reuse one tail template for the whole range and + * refresh only the PFN-dependent fields in that template before each copy. + */ + pfn = head_pfn + 1; + page = pfn_to_page(pfn); + __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + prep_compound_tail(page, head, order); + set_page_count(page, 0); + memcpy(&template, page, sizeof(*page)); + + /* Initialize the remaining tail pages from template. */ + for (pfn = head_pfn + 2; pfn < end_pfn; pfn++) + zone_device_page_init_from_template(pfn_to_page(pfn), pfn, + &template); prep_compound_head(head, order); } From 9b2b8fd4eb1fbc0ed95ba3a514d702565e4b5721 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:36 +0800 Subject: [PATCH 673/857] string: introduce memcpy_nontemporal() Introduce memcpy_nontemporal() for write-once copy sites that want a named non-temporal copy primitive. On x86_64, override the helper in arch/x86/include/asm/string_64.h using the usual self-macro pattern, next to the existing memcpy_flushcache() backend that memcpy_nontemporal() wraps. include/linux/string.h provides the generic memcpy_nontemporal() fallback as #define memcpy_nontemporal(dst, src, len) \ ((void)memcpy(dst, src, len)) instead of an inline wrapper, so architectures without a specialized backend keep the usual memcpy() FORTIFY coverage when the compiler can still see object sizes at the original call site. It also makes the memcpy_nontemporal() API uniformly void, matching memcpy_flushcache() and the x86 backend, so callers cannot accidentally depend on a return value on fallback architectures. memcpy_nontemporal() is only a copy primitive. It does not imply a drain or a publication barrier. Callers that use it before a producer-consumer or device-visible handoff must provide the required ordering at that handoff point. The immediate user is the ZONE_DEVICE template-copy path. It populates struct page descriptors in a write-once pattern, so a regular cached memcpy() can incur avoidable write-allocate traffic and cache pollution for data with little near-term reuse. Link: https://lore.kernel.org/20260831111638.76012-6-lizhe.67@bytedance.com Signed-off-by: Li Zhe Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Mike Rapoport (Microsoft) Cc: Muchun Song Signed-off-by: Andrew Morton --- arch/x86/include/asm/string_64.h | 12 ++++++++++++ include/linux/string.h | 13 +++++++++++++ 2 files changed, 25 insertions(+) diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h index 4635616863f53d..21ae515ae35a3d 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -100,6 +100,18 @@ static __always_inline void memcpy_flushcache(void *dst, const void *src, size_t } __memcpy_flushcache(dst, src, cnt); } + +#define memcpy_nontemporal memcpy_nontemporal +/* + * Reuse the existing x86 flushcache backend as the non-temporal copy + * primitive. + */ +static __always_inline void memcpy_nontemporal(void *dst, const void *src, + size_t cnt) +{ + memcpy_flushcache(dst, src, cnt); +} + #endif #endif /* __KERNEL__ */ diff --git a/include/linux/string.h b/include/linux/string.h index 5702daca4326b7..6cb5cdd01158b2 100644 --- a/include/linux/string.h +++ b/include/linux/string.h @@ -278,6 +278,19 @@ static inline void memcpy_flushcache(void *dst, const void *src, size_t cnt) } #endif +#ifndef memcpy_nontemporal +/* + * memcpy_nontemporal() requests a non-temporal copy when the + * architecture has a suitable backend. Architectures without a + * specialized backend fall back to memcpy(). Keep this as a + * function-like macro so the compiler can still see the original + * memcpy() call site and preserve the usual FORTIFY coverage when + * object sizes remain visible there, while keeping the API void. + */ +#define memcpy_nontemporal(dst, src, len) \ + ((void)memcpy(dst, src, len)) +#endif + void *memchr_inv(const void *s, int c, size_t n); char *strreplace(char *str, char old, char new); From fe94f568e5ecb485a401cde3447c481f8e984e49 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:37 +0800 Subject: [PATCH 674/857] mm: use memcpy_nontemporal() in zone-device template copies The template fast path currently uses memcpy() for the actual struct page copy. Switch zone_device_page_init_from_template() to memcpy_nontemporal(). ZONE_DEVICE memmap initialization is largely write-once: each struct page is populated once, and most destination cachelines are not expected to be reused immediately afterwards. On x86, a regular cached memcpy() can therefore incur write-allocate traffic by pulling destination cachelines into the cache before writeback, and can populate the cache with data that has little near-term reuse. Using memcpy_nontemporal() lets this path request nontemporal stores for that copy pattern, which can reduce cache pollution and avoid part of the associated write-allocate overhead, while architectures without a specialized backend still fall back to memcpy(). Do not add a KASAN/KMSAN-specific fallback around this call site. As Muchun pointed out, special KASAN handling for memcpy_flushcache() or memcpy_nontemporal(), if needed, belongs in the low-level helper rather than in this ZONE_DEVICE caller. No separate drain is added here. memcpy_nontemporal() is used only as the copy primitive while memmap_init_zone_device() is still initializing the struct page array. The ordinary stores that follow in this path, such as compound-page setup, are part of the same CPU's initialization sequence; they are not used as a publication store that tells another CPU or device to consume data written by the non-temporal copy. Therefore this call site does not need a helper-level drain for correctness. Callers that use memcpy_nontemporal() as part of a producer-consumer or device-visible handoff must add the required ordering themselves. Tested in a VM with a 100 GB fsdax namespace device configured with map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem driver 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of rebinds for nd_pmem driver: 221.07 ms Average of rebinds for dax_pmem driver: 191.20 ms With this patch and its prerequisites applied: Average of rebinds for nd_pmem driver: 150.40 ms Average of rebinds for dax_pmem driver: 161.83 ms This reduces the average memmap initialization time measured during rebind by about 32.0% for nd_pmem and 15.4% for dax_pmem. Link: https://lore.kernel.org/20260831111638.76012-7-lizhe.67@bytedance.com Signed-off-by: Li Zhe Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Mike Rapoport (Microsoft) Cc: Muchun Song Signed-off-by: Andrew Morton --- mm/mm_init.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 0d83034cfa97b0..2ed17cc707edc7 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1037,7 +1037,7 @@ static void zone_device_page_init_from_template(struct page *page, if (!is_highmem_idx(ZONE_DEVICE)) set_page_address(template, __va(pfn << PAGE_SHIFT)); #endif - memcpy(page, template, sizeof(*page)); + memcpy_nontemporal(page, template, sizeof(*page)); } /* From 47d559e1655301a0d13c241dd683532017441f2f Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:38 +0800 Subject: [PATCH 675/857] x86/string: extend memcpy_flushcache() fixed-size fastpaths The x86 memcpy_nontemporal() helper maps to memcpy_flushcache(), and the ZONE_DEVICE template-copy path uses it to copy one struct page at a time. The relevant copy size is sizeof(struct page). On x86_64, the base struct page layout is 64 bytes. Adding either the KMSAN metadata pointers or an out-of-flags last_cpupid field can make it 80 bytes after alignment, and enabling both can make it 96 bytes. memcpy_flushcache() currently only has inline fixed-size cases for 4, 8, and 16 bytes. As a result, these constant-sized struct page copies fall through to __memcpy_flushcache() even though the compiler knows the copy size at the call site. Add fixed-size MOVNTI cases up to 96 bytes so the ZONE_DEVICE template-copy path can keep these struct page copies in the inline memcpy_flushcache() path. This matters for ZONE_DEVICE memmap initialization because the copy happens once per initialized struct page. For a 100 GB fsdax namespace with map=dev, this is about 25 million struct page copies during nd_pmem binding or rebinding. Tested in a VM with a 100 GB fsdax namespace device configured with map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem drivers 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). With memcpy_nontemporal() used by the ZONE_DEVICE template-copy path: Average of rebinds for nd_pmem driver: 150.40 ms Average of rebinds for dax_pmem driver: 161.83 ms With this x86 fixed-size fastpath patch applied: Average of rebinds for nd_pmem driver: 71.93 ms Average of rebinds for dax_pmem driver: 87.37 ms This further reduces the average memmap initialization time measured during rebind by about 52.2% for nd_pmem and 46.0% for dax_pmem. Link: https://lore.kernel.org/20260831111638.76012-8-lizhe.67@bytedance.com Signed-off-by: Li Zhe Suggested-by: Borislav Petkov Acked-by: Borislav Petkov (AMD) Acked-by: Dave Hansen Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Mike Rapoport (Microsoft) Cc: Muchun Song Signed-off-by: Andrew Morton --- arch/x86/include/asm/string_64.h | 71 +++++++++++++++++++++++++------- 1 file changed, 56 insertions(+), 15 deletions(-) diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h index 21ae515ae35a3d..831d3dda3b380e 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -82,23 +82,64 @@ int strcmp(const char *cs, const char *ct); #ifdef CONFIG_ARCH_HAS_UACCESS_FLUSHCACHE #define __HAVE_ARCH_MEMCPY_FLUSHCACHE 1 void __memcpy_flushcache(void *dst, const void *src, size_t cnt); -static __always_inline void memcpy_flushcache(void *dst, const void *src, size_t cnt) + +static __always_inline void movnti_4(void *dst, const void *src) +{ + asm volatile("movntil %1, %0" + : "=m"(*(u32 *)dst) + : "r"(*(const u32 *)src) + : "memory"); +} + +static __always_inline void movnti_8(void *dst, const void *src) +{ + asm volatile("movntiq %1, %0" + : "=m"(*(u64 *)dst) + : "r"(*(const u64 *)src) + : "memory"); +} + +static __always_inline void movnti_16(void *dst, const void *src) +{ + movnti_8(dst, src); + movnti_8(dst + 8, src + 8); +} + +static __always_inline void movnti_32(void *dst, const void *src) +{ + movnti_16(dst, src); + movnti_16(dst + 16, src + 16); +} + +static __always_inline void movnti_64(void *dst, const void *src) +{ + movnti_32(dst, src); + movnti_32(dst + 32, src + 32); +} + +static __always_inline void memcpy_flushcache(void *dst, const void *src, + size_t cnt) { - if (__builtin_constant_p(cnt)) { - switch (cnt) { - case 4: - asm ("movntil %1, %0" : "=m"(*(u32 *)dst) : "r"(*(u32 *)src)); - return; - case 8: - asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src)); - return; - case 16: - asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src)); - asm ("movntiq %1, %0" : "=m"(*(u64 *)(dst + 8)) : "r"(*(u64 *)(src + 8))); - return; - } + if (!__builtin_constant_p(cnt)) + return __memcpy_flushcache(dst, src, cnt); + + /* + * The relevant fixed-size copies here are the x86_64 struct page sizes: + * 64, 80, and 96 bytes. Keep 32-byte and 48-byte copies inline as well + * instead of sending those nearby fixed-size cases back to + * __memcpy_flushcache(). + */ + switch (cnt) { + case 4: movnti_4(dst, src); break; + case 8: movnti_8(dst, src); break; + case 16: movnti_16(dst, src); break; + case 32: movnti_32(dst, src); break; + case 48: movnti_32(dst, src); movnti_16(dst + 32, src + 32); break; + case 64: movnti_64(dst, src); break; + case 80: movnti_64(dst, src); movnti_16(dst + 64, src + 64); break; + case 96: movnti_64(dst, src); movnti_32(dst + 64, src + 64); break; + default: __memcpy_flushcache(dst, src, cnt); break; } - __memcpy_flushcache(dst, src, cnt); } #define memcpy_nontemporal memcpy_nontemporal From 96c8ebc0c539eee4530410254ba6301baea5e6e8 Mon Sep 17 00:00:00 2001 From: Dev Jain Date: Mon, 31 Aug 2026 08:28:47 +0000 Subject: [PATCH 676/857] mm/rmap: remove stale hugetlb check in try_to_unmap_one Post commit d4ec5572825a ("mm/rmap: add try_to_unmap_poisoned_hugetlb_one") try_to_unmap_one() cannot be called with a hugetlb folio. Therefore remove the folio_test_hugetlb() check. Link: https://lore.kernel.org/20260831082849.3573957-1-dev.jain@arm.com Signed-off-by: Dev Jain Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Lance Yang Reviewed-by: Kunwu Chan Cc: David Hildenbrand Cc: Harry Yoo Cc: Jann Horn Cc: Liam R. Howlett Cc: Rik van Riel Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/rmap.c | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/mm/rmap.c b/mm/rmap.c index fed0362e0bd0e3..b5cc9273fe5a47 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -2297,11 +2297,8 @@ static bool try_to_unmap_one(struct folio *folio, struct vm_area_struct *vma, VM_BUG_ON_FOLIO(!pvmw.pte, folio); address = pvmw.address; - if (folio_test_hugetlb(folio)) { - pteval = huge_ptep_get(mm, address, pvmw.pte); - } else { - pteval = ptep_get(pvmw.pte); - } + pteval = ptep_get(pvmw.pte); + if (likely(pte_present(pteval))) { pfn = pte_pfn(pteval); } else { From f4bb278a0bb905c7703993060e0d9271820f7532 Mon Sep 17 00:00:00 2001 From: David Stevens Date: Mon, 31 Aug 2026 16:43:39 -0700 Subject: [PATCH 677/857] memcg: don't call schedule_work when no spinning is allowed Memcg charging can be done from any context, but calling schedule_work() isn't safe from an NMI. If memory.high is breached from a context where spinning isn't allowed, use irq_work to schedule the reclaim work. Link: https://lore.kernel.org/20260831234339.280376-1-stevensd@google.com Fixes: 3ac4638a734a ("memcg: make memcg_rstat_updated nmi safe") Signed-off-by: David Stevens Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/memcontrol.h | 1 + mm/memcontrol.c | 12 +++++++++++- 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index c799926435560f..945d7dbf1bb2c8 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -206,6 +206,7 @@ struct mem_cgroup { spinlock_t peaks_lock; /* Range enforcement for interrupt charges */ + struct irq_work high_irq_work; struct work_struct high_work; #ifdef CONFIG_ZSWAP diff --git a/mm/memcontrol.c b/mm/memcontrol.c index aeaa09e01d70ea..d8c22070f24a5b 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2440,6 +2440,11 @@ static void high_work_func(struct work_struct *work) reclaim_high(memcg, MEMCG_CHARGE_BATCH, GFP_KERNEL); } +static void high_irq_work_func(struct irq_work *work) +{ + schedule_work(&container_of(work, struct mem_cgroup, high_irq_work)->high_work); +} + /* * Clamp the maximum sleep time per allocation batch to 2 seconds. This is * enough to still cause a significant slowdown in most cases, while still @@ -2848,7 +2853,10 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, /* Don't bother a random interrupted task */ if (!in_task()) { if (mem_high) { - schedule_work(&memcg->high_work); + if (allow_spinning) + schedule_work(&memcg->high_work); + else + irq_work_queue(&memcg->high_irq_work); break; } continue; @@ -4223,6 +4231,7 @@ static struct mem_cgroup *mem_cgroup_alloc(struct mem_cgroup *parent) goto fail; INIT_WORK(&memcg->high_work, high_work_func); + init_irq_work(&memcg->high_irq_work, high_irq_work_func); vmpressure_init(&memcg->vmpressure); INIT_LIST_HEAD(&memcg->memory_peaks); INIT_LIST_HEAD(&memcg->swap_peaks); @@ -4431,6 +4440,7 @@ static void mem_cgroup_css_free(struct cgroup_subsys_state *css) static_branch_dec(&memcg_bpf_enabled_key); vmpressure_cleanup(&memcg->vmpressure); + irq_work_sync(&memcg->high_irq_work); cancel_work_sync(&memcg->high_work); free_shrinker_info(memcg); mem_cgroup_free(memcg); From ea813276f8ab9b9a01397a751f67da3ae89f3a4f Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Mon, 31 Aug 2026 15:43:04 +0530 Subject: [PATCH 678/857] mm/gup_test: report actual pinned bytes __gup_test_ioctl() advances addr to the end of the current batch before checking if GUP pinned the entire requested batch. If GUP pins more than 0 pages but less than the requested batch size, addr still advances by the requested batch size. The next iteration detects the partial pinning and breaks out of the loop. Again gup->size is calculated using addr - gup->addr, so it also includes the unpinned pages of the requested batch. Calculate gup->size using the actual number of pages pinned multiplied by PAGE_SIZE. Link: https://lore.kernel.org/20260831101304.162867-1-sarthak.sharma@arm.com Fixes: 64c349f4ae78 ("mm: add infrastructure for get_user_pages_fast() benchmarking") Signed-off-by: Sarthak Sharma Reviewed-by: Kiryl Shutsemau (Meta) Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: David Hildenbrand Signed-off-by: Andrew Morton --- mm/gup_test.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/gup_test.c b/mm/gup_test.c index 44c1cdfb9c3717..185ba3bb8ed10b 100644 --- a/mm/gup_test.c +++ b/mm/gup_test.c @@ -188,7 +188,7 @@ static int __gup_test_ioctl(unsigned int cmd, nr_pages = i; gup->get_delta_usec = ktime_us_delta(end_time, start_time); - gup->size = addr - gup->addr; + gup->size = nr_pages * PAGE_SIZE; /* * Take an un-benchmark-timed moment to verify DMA pinned From 2f00a89742412b23852bdfa68e976d705f577276 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 31 Aug 2026 10:15:13 +0100 Subject: [PATCH 679/857] mm/huge_memory: do not touch frozen folios in deferred_split_isolate() Patch series "Fix deferred_split_isolate() and drop the split workaround", v2. deferred_split_isolate() probes each queued folio with folio_try_get(). folio_try_get() failure is treated as a lost race with folio_put(). It leads to wrong results when !folio_try_get() was not caused by folio_put(): for a frozen folio, PG_partially_mapped gets wrongfully cleared and the folio dropped from the queue. It came up in the review of my collapse RFC series: https://lore.kernel.org/all/20260824131224.73344-1-lance.yang@linux.dev/ The bug is inert in upstream code: - __folio_split() works around it; - __folio_migrate_mapping() freezes a folio it is about to replace; - reclaim freezes only what try_to_unmap() already unmapped. No cc:stable needed. But my collapse rework steps on it, so it is worth fixing. The branch the first patch removes also hid an inert, pre-existing bug in the zone device path: https://lore.kernel.org/all/20260827163838.1813081-1-usama.arif@linux.dev/ The first patch fixes deferred_split_isolate(). The second patch removes the workaround for this deferred_split_isolate() behaviour from __folio_freeze_and_split_unmapped(). Tested in a VM: split_huge_page_test, folio_split_race_test and cow pass. Also ran a test that leaves 16 partially mapped THPs on the deferred split queue and drives thp-deferred_split through debugfs, checking nr_anon_partially_mapped. This patch (of 2): deferred_split_isolate() probes each queued folio with folio_try_get(). folio_try_get() failure is treated as a lost race with folio_put(): clear PG_partially_mapped, correct MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, take the folio off the queue. The folio_put() race is the most common case for !folio_try_get(), but it is not the only option. Another scenario is folio_ref_freeze(). A zero refcount in such cases does not mean the folio is going away. It means "don't touch me" and current deferred_split_isolate() doesn't respect it. It can lead to unqueueing folios from the deferred list for no reason: CPU 0 CPU 1 --------------------------- ------------------------------ freeze a mapped folio deferred_split_scan() folio_ref_freeze() folio_try_get() fails folio_clear_partially_mapped() NR_ANON_PARTIALLY_MAPPED-- folio off the queue give up, put it back folio_ref_unfreeze() The folio is still partially mapped, but it is no longer a split candidate. Nothing queues it again until part of it is unmapped once more. Skip the folio instead: whoever freezes the folio, owns it and owner is responsible for its fate. It also covers the folio_put() case: __folio_put() unqueues the folio via folio_unqueue_deferred_split(). Nothing is lost by skipping. Everything that frees a queued folio unqueues it first, and folio_unqueue_deferred_split() clears PG_partially_mapped and brings MTHP_STAT_NR_ANON_PARTIALLY_MAPPED down on the way: __folio_put(), folios_put_refs() mm/folio.c __folio_migrate_mapping() mm/migrate.c shrink_folio_list() mm/vmscan.c __folio_freeze_and_split_unmapped() does the same by hand, under the list_lru lock it holds across the freeze. A freeze that ends in folio_ref_unfreeze() leaves a folio that is still partially mapped and still belongs on the queue. Link: https://lore.kernel.org/20260831091514.1879786-1-kirill@shutemov.name Link: https://lore.kernel.org/20260831091514.1879786-2-kirill@shutemov.name Fixes: 8422acdc97ed ("mm: introduce a pageflag for partially mapped folios") Signed-off-by: Kiryl Shutsemau (Meta) Reported-by: Lance Yang Closes: https://lore.kernel.org/all/20260824131224.73344-1-lance.yang@linux.dev/ Assisted-by: Claude-Code:claude-opus-5 Reviewed-by: Zi Yan Reviewed-by: Johannes Weiner Reviewed-by: Lance Yang Acked-by: Usama Arif Reviewed-by: Baolin Wang Cc: Balbir Singh Cc: Barry Song Cc: David Hildenbrand Cc: Dev Jain Cc: Hugh Dickins Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ryan Roberts Signed-off-by: Andrew Morton --- mm/huge_memory.c | 19 ++++--------------- 1 file changed, 4 insertions(+), 15 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 54494c3fa9835e..779c02e0bf6cc3 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4638,22 +4638,11 @@ static enum lru_status deferred_split_isolate(struct list_head *item, struct folio *folio = container_of(item, struct folio, _deferred_list); struct list_head *freeable = cb_arg; - if (folio_try_get(folio)) { - list_lru_isolate_move(lru, item, freeable); - return LRU_REMOVED; - } + /* Lost race to folio_put() or the folio is under folio_ref_freeze() */ + if (!folio_try_get(folio)) + return LRU_SKIP; - /* - * We lost race with folio_put(). Read folio state before the - * isolate: folio_unqueue_deferred_split() checks list_empty() - * locklessly, so once removed the folio can be freed any time. - */ - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(folio_order(folio), - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } - list_lru_isolate(lru, item); + list_lru_isolate_move(lru, item, freeable); return LRU_REMOVED; } From 9a7162b965b3cc6b4e96c52514d8dba911127476 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 31 Aug 2026 10:15:14 +0100 Subject: [PATCH 680/857] mm/huge_memory: dequeue the deferred split after the split freeze __folio_freeze_and_split_unmapped() takes the deferred split list_lru lock across the freeze. It is only there to stop deferred_split_scan() from touching the folio under split. With deferred_split_isolate() fixed, the workaround can be dropped. Unqueue the folio after folio_ref_freeze(), the way __folio_migrate_mapping() does: folio_unqueue_deferred_split() needs a zero refcount and a memcg still set, and both hold there. If the split is called from deferred_split_scan(), the unqueue is a no-op -- the folio is already removed from the list. But PG_partially_mapped is still set, so it has to be cleared here or MTHP_STAT_NR_ANON_PARTIALLY_MAPPED never comes back down. Assisted-by: Claude-Code:claude-opus-5 Link: https://lore.kernel.org/20260831091514.1879786-3-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Reviewed-by: Zi Yan Reviewed-by: Johannes Weiner Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Cc: Balbir Singh Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Hugh Dickins Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ryan Roberts Cc: Usama Arif Signed-off-by: Andrew Morton --- mm/huge_memory.c | 46 ++++++++++++++-------------------------------- 1 file changed, 14 insertions(+), 32 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 779c02e0bf6cc3..c5d11147b69aec 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3979,41 +3979,27 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n struct folio *end_folio = folio_next(folio); struct folio *new_folio, *next; int old_order = folio_order(folio); - struct list_lru_one *lru; - bool dequeue_deferred; int ret = 0; VM_WARN_ON_ONCE(!mapping && end); - /* - * If this folio can be on the deferred split queue, lock out - * the shrinker before freezing the ref. If the shrinker sees - * a 0-ref folio, it assumes it beat folio_put() to the list - * lock and must clean up the LRU state - the same dequeue we - * will do below as part of the split. - */ - dequeue_deferred = folio_test_anon(folio) && old_order > 1; - if (dequeue_deferred) { - struct mem_cgroup *memcg; - - rcu_read_lock(); - memcg = folio_memcg(folio); - lru = list_lru_lock(&deferred_split_lru, - folio_nid(folio), &memcg); - } + if (folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { struct swap_cluster_info *ci = NULL; struct lruvec *lruvec; - if (dequeue_deferred) { - __list_lru_del(&deferred_split_lru, lru, - &folio->_deferred_list, folio_nid(folio)); - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(old_order, - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } - list_lru_unlock(lru); - rcu_read_unlock(); + /* Take off the deferred split queue while frozen and memcg set */ + folio_unqueue_deferred_split(folio); + + /* + * deferred_split_scan() takes the folio off the queue before it + * splits it, so the unqueue above finds an empty list and + * leaves PG_partially_mapped set. + * Clear it here: the flag does not survive the split. + */ + if (folio_test_partially_mapped(folio)) { + folio_clear_partially_mapped(folio); + mod_mthp_stat(old_order, + MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); } if (mapping) { @@ -4115,10 +4101,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n if (ci) swap_cluster_unlock(ci); } else { - if (dequeue_deferred) { - list_lru_unlock(lru); - rcu_read_unlock(); - } return -EAGAIN; } From e316241b2d4f9db91e278cf84a3b1e9852b2dca7 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Mon, 31 Aug 2026 21:35:18 +0800 Subject: [PATCH 681/857] mm/hugetlb: preserve source surplus accounting during demotion Patch series "mm/hugetlb: fix surplus accounting and availability checks during demotion", v2. Fix surplus accounting and availability checks in the hugetlb demote path. Patch 1 fixes source hstate accounting when the free folio selected for demotion accounts for a surplus page. Patch 2 prevents demotion from removing free huge pages that back reservations. Both fixes were tested with x86_64 QEMU guests. The commands below use: hstate=/sys/kernel/mm/hugepages/hugepages-1048576kB Patch 1: surplus accounting A vmemmap restoration failure is difficult to trigger deterministically. For this test only, add a one-shot fault injection that makes the first attempt to restore the vmemmap of an optimized 1 GiB folio fail: /* TEST ONLY: fail the first optimized 1G folio restore. */ static atomic_t fail_next_1g_restore = ATOMIC_INIT(1); /* In __hugetlb_vmemmap_restore_folio(). */ if (huge_page_size(h) == SZ_1G && atomic_cmpxchg(&fail_next_1g_restore, 1, 0) == 1) { pr_info("TEST ONLY: forcing one 1G vmemmap " "restore failure\n"); return -ENOMEM; } The injection does not modify the demotion or accounting code. It is one-shot so that the later restore performed during demotion can succeed. 1. Boot QEMU with: hugepagesz=1G hugepages=0 hugetlb_cma=1G hugetlb_free_vmemmap=on 2. Enable overcommit: echo 1 > "$hstate/nr_overcommit_hugepages" 3. Allocate one 1 GiB huge page: nr=1 surplus=1 free=0 resv=0 4. Unmap it. The forced restoration failure leaves the folio on the freelist while it is still accounted as surplus: nr=1 surplus=1 free=1 resv=0 5. Demote one page: echo 1 > "$hstate/demote" Before this fix: nr=0 surplus=1 free=0 resv=0 surplus > nr After this fix: nr=0 surplus=0 free=0 resv=0 Patch 2: cap demotion This reproducer requires no kernel instrumentation. 1. Boot QEMU with: hugepagesz=1G hugepages=2 nr=2 surplus=0 free=2 resv=0 2. Reserve one 1 GiB huge page with an untouched hugetlbfs mapping: nr=2 surplus=0 free=2 resv=1 3. Request demotion of two pages: echo 2 > "$hstate/demote" Before this fix: nr=0 surplus=0 free=0 resv=1 resv > free After this fix: nr=1 surplus=0 free=1 resv=1 resv == free 4. Touch the reserved page and let the process exit. Before this fix, the access fails with SIGBUS and leaves: nr=0 surplus=0 free=0 resv=0 After this fix, the access succeeds and leaves: nr=1 surplus=0 free=1 resv=0 This patch (of 2): demote_pool_huge_page() currently removes every source folio as a persistent folio. A free folio can instead account for one of the source hstate's surplus pages, for example after a vmemmap restoration failure. Removing such a folio without adjusting surplus_huge_pages makes the persistent count underflow, and later subtracting it from max_huge_pages can underflow that counter as well. Classify selected folios against the node's surplus count while holding hugetlb_lock, and preserve that classification on rollback. Track the number of successfully demoted persistent folios separately so only those folios reduce the source max_huge_pages target. All successfully demoted folios still increase the destination target because the new destination folios are added as persistent pages. Testing: Tested on an x86_64 QEMU guest booted with: hugepagesz=1G hugepages=0 hugetlb_cma=1G hugetlb_free_vmemmap=on For testing only, add a one-shot fault injection that makes the first call to __hugetlb_vmemmap_restore_folio() for an optimized 1 GiB folio return -ENOMEM. Set nr_overcommit_hugepages to 1, then allocate one 1 GiB huge page: nr=1 surplus=1 free=0 resv=0 Unmap it. The failed restoration leaves the folio on the freelist while it is still accounted as surplus: nr=1 surplus=1 free=1 resv=0 Demote one page. Before this fix, the result is: nr=0 surplus=1 free=0 resv=0 After this fix, the result is: nr=0 surplus=0 free=0 resv=0 The fault injection is one-shot, so the restore performed during demotion can succeed. Link: https://lore.kernel.org/20260831133519.2505020-2-xialonglong2025@163.com Fixes: 8531fc6f52f5 ("hugetlb: add hugetlb demote page support") Signed-off-by: Longlong Xia Cc: David Hildenbrand Cc: Muchun Song Cc: Oscar Salvador Cc: Yu Zhao Assisted-by: Codex:gpt-5.6-sol Signed-off-by: Andrew Morton --- mm/hugetlb.c | 35 +++++++++++++++++++++++++++++++---- 1 file changed, 31 insertions(+), 4 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 8fa1bafa03d91b..2692fb760c3f1a 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3999,6 +3999,7 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, struct hstate *dst; long rc = 0; long nr_demoted = 0; + long nr_persistent = 0; lockdep_assert_held(&hugetlb_lock); @@ -4011,22 +4012,40 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, for_each_node_mask_to_free(src, nr_nodes, node, nodes_allowed) { LIST_HEAD(list); + LIST_HEAD(surplus_list); struct folio *folio, *next; list_for_each_entry_safe(folio, next, &src->hugepage_freelists[node], lru) { + bool adjust_surplus; + if (folio_test_hwpoison(folio)) continue; - remove_hugetlb_folio(src, folio, false); - list_add(&folio->lru, &list); + /* Surplus accounting is maintained per node, not per folio. */ + adjust_surplus = src->surplus_huge_pages_node[node] > 0; + remove_hugetlb_folio(src, folio, adjust_surplus); + list_add(&folio->lru, adjust_surplus ? &surplus_list : &list); + if (!adjust_surplus) + nr_persistent++; if (++nr_demoted == nr_to_demote) break; } + if (list_empty(&list) && list_empty(&surplus_list)) + continue; + spin_unlock_irq(&hugetlb_lock); - rc = demote_free_hugetlb_folios(src, dst, &list); + if (!list_empty(&list)) + rc = demote_free_hugetlb_folios(src, dst, &list); + if (!list_empty(&surplus_list)) { + long tmp_rc; + + tmp_rc = demote_free_hugetlb_folios(src, dst, &surplus_list); + if (rc >= 0) + rc = tmp_rc; + } spin_lock_irq(&hugetlb_lock); @@ -4034,6 +4053,14 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, list_del(&folio->lru); add_hugetlb_folio(src, folio, false); + nr_demoted--; + nr_persistent--; + } + + list_for_each_entry_safe(folio, next, &surplus_list, lru) { + list_del(&folio->lru); + add_hugetlb_folio(src, folio, true); + nr_demoted--; } @@ -4045,7 +4072,7 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, * Not absolutely necessary, but for consistency update max_huge_pages * based on pool changes for the demoted page. */ - src->max_huge_pages -= nr_demoted; + src->max_huge_pages -= nr_persistent; dst->max_huge_pages += nr_demoted << (huge_page_order(src) - huge_page_order(dst)); if (rc < 0) From d07fce65bfbe04603df372cc7f243b16e730ef71 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Mon, 31 Aug 2026 21:35:19 +0800 Subject: [PATCH 682/857] mm/hugetlb: cap demotion at currently available free pages Demotion must not remove free huge pages that back existing reservations. The sysfs path checks whether any page is available, but passes the entire request to demote_pool_huge_page(). For example, with two free pages and one reservation, a request for two pages removes both and leaves the reservation without a backing page. Cap the sysfs request by both global availability and the selected node's free pages. Recheck global availability in demote_pool_huge_page() before each node batch because that function drops hugetlb_lock while restoring vmemmap and reservations can change before the next batch. Testing: Tested on an x86_64 QEMU guest booted with: hugepagesz=1G hugepages=2 Reserve one 1 GiB huge page with an untouched hugetlbfs mapping: nr=2 surplus=0 free=2 resv=1 Request demotion of two pages. Before this fix, both free pages are demoted: nr=0 surplus=0 free=0 resv=1 Touching the reserved mapping then fails with SIGBUS. After this fix, the request is capped at the single available page: nr=1 surplus=0 free=1 resv=1 Touching the reserved mapping succeeds. After the process exits, the counters are: nr=1 surplus=0 free=1 resv=0 Link: https://lore.kernel.org/20260831133519.2505020-3-xialonglong2025@163.com Fixes: c0f398c3b2cf ("mm/hugetlb_vmemmap: batch HVO work when demoting") Signed-off-by: Longlong Xia Assisted-by: Codex:gpt-5.6-sol Cc: David Hildenbrand Cc: Longlong Xia Cc: Muchun Song Cc: Oscar Salvador Cc: Yu Zhao Signed-off-by: Andrew Morton --- mm/hugetlb.c | 22 +++++++++++++++++++++- mm/hugetlb_sysfs.c | 10 +++++----- 2 files changed, 26 insertions(+), 6 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 2692fb760c3f1a..bfc0184ed95c72 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -4014,6 +4014,26 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, LIST_HEAD(list); LIST_HEAD(surplus_list); struct folio *folio, *next; + unsigned long nr_available, nr_target; + + /* + * Re-check available each node batch: the previous + * batch released hugetlb_lock for vmemmap restore/split, + * and a new reservation could have been added in that + * window, shrinking the budget. available is global + * (resv is not per-node), so 0 means no node can + * contribute -- stop the whole scan. + */ + nr_available = available_huge_pages(src); + if (!nr_available) + break; + + /* + * Cap this batch at the current budget; expressed as a + * cumulative stop point because nr_demoted is running. + */ + nr_target = nr_demoted + min_t(unsigned long, + nr_to_demote - nr_demoted, nr_available); list_for_each_entry_safe(folio, next, &src->hugepage_freelists[node], lru) { bool adjust_surplus; @@ -4028,7 +4048,7 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, if (!adjust_surplus) nr_persistent++; - if (++nr_demoted == nr_to_demote) + if (++nr_demoted == nr_target) break; } diff --git a/mm/hugetlb_sysfs.c b/mm/hugetlb_sysfs.c index 79ece91406bfa4..326a54b4d991c9 100644 --- a/mm/hugetlb_sysfs.c +++ b/mm/hugetlb_sysfs.c @@ -211,15 +211,15 @@ static ssize_t demote_store(struct kobject *kobj, * Check for available pages to demote each time thorough the * loop as demote_pool_huge_page will drop hugetlb_lock. */ + nr_available = h->free_huge_pages - h->resv_huge_pages; if (nid != NUMA_NO_NODE) - nr_available = h->free_huge_pages_node[nid]; - else - nr_available = h->free_huge_pages; - nr_available -= h->resv_huge_pages; + nr_available = min(nr_available, + h->free_huge_pages_node[nid]); if (!nr_available) break; - rc = demote_pool_huge_page(h, n_mask, nr_demote); + rc = demote_pool_huge_page(h, n_mask, + min(nr_demote, nr_available)); if (rc < 0) { err = rc; break; From 2bc489ef89179363e4889e0e87ee680fb7de5f8c Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:23 +0530 Subject: [PATCH 683/857] mm: make ptval_to_str() generally available Patch series "mm: Drop pxd_ERROR()". pxd_ERROR() macros have been provided by all platforms, which are very much identical and can be dropped off completely if these pgtable printing could be moved to callers in generic MM aka all pxd_clear_bad(). But first cleanups and re-organizations are required in some platforms that are using these macros internally. Afterwards [pte|pmd|pud|p4d|pgd]_ERROR() macros have been completely dropped from the entire tree. This patch (of 8): Move ptval_to_str() inside a header thus making the helper more generally available for new users which are being added later. While here, also move another related string size macro PTVAL_STR_MAX inside the header as well. Link: https://lore.kernel.org/20260831054331.625505-1-anshuman.khandual@arm.com Link: https://lore.kernel.org/20260831054331.625505-2-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Lorenzo Stoakes Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato Cc: Geert Uytterhoeven Signed-off-by: Andrew Morton --- include/linux/pgtable.h | 14 ++++++++++++++ mm/memory.c | 15 +-------------- 2 files changed, 15 insertions(+), 14 deletions(-) diff --git a/include/linux/pgtable.h b/include/linux/pgtable.h index 8c093c119e5a82..e3c8ab96941c5e 100644 --- a/include/linux/pgtable.h +++ b/include/linux/pgtable.h @@ -2313,6 +2313,20 @@ static inline const char *pgtable_level_to_str(enum pgtable_level level) } } +void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size); + +#define ptval_to_str(buf, val) \ + do { \ + auto __val = (val); \ + \ + ptval_bytes_to_hex_str((buf), sizeof(buf), &__val, sizeof(__val)); \ + } while (0) + +#if defined(__SIZEOF_INT128__) +#define PTVAL_STR_MAX (32 + 1) /* Max 128-bit value in hex + NUL */ +#else +#define PTVAL_STR_MAX (16 + 1) /* Max 64-bit value in hex + NUL */ +#endif #endif /* !__ASSEMBLER__ */ #if !defined(MAX_POSSIBLE_PHYSMEM_BITS) && !defined(CONFIG_64BIT) diff --git a/mm/memory.c b/mm/memory.c index bc14cae3c49d72..ec63dd6212ac5a 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -495,7 +495,7 @@ static inline void add_mm_rss_vec(struct mm_struct *mm, int *rss) /* Allow a burst of 60 bad page map reports per minute. */ static DEFINE_RATELIMIT_STATE(bad_page_map_ratelimit, 60 * HZ, 60); -static void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size) +void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size) { if (WARN_ON_ONCE(buf_size < entry_size * 2 + 1)) { snprintf(buf, buf_size, "overflow"); @@ -522,19 +522,6 @@ static void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry } } -#define ptval_to_str(buf, val) \ - do { \ - auto __val = (val); \ - \ - ptval_bytes_to_hex_str((buf), sizeof(buf), &__val, sizeof(__val)); \ - } while (0) - -#if defined(__SIZEOF_INT128__) -#define PTVAL_STR_MAX (32 + 1) /* Max 128-bit value in hex + NUL */ -#else -#define PTVAL_STR_MAX (16 + 1) /* Max 64-bit value in hex + NUL */ -#endif - static void __print_bad_page_map_pgtable(struct mm_struct *mm, unsigned long addr) { char pgd_str[PTVAL_STR_MAX]; From 9a42d9bd298244fa3f16f800482961ddde5d0b4d Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:24 +0530 Subject: [PATCH 684/857] mm: stop using pxd_ERROR() pxd_ERROR() has been used in generic mm just to print the page table entry in pxd_clear_bad() before clearing those out with pxd_clear() later. These pxd_ERROR() macros have been provided by all platforms which basically did the same thing. Make pxd_clear_bad() use recently added ptval_to_str() instead for printing page table entries thus completely dropping dependency on platform provided pxd_ERROR() macros which can then be dropped off later on. Link: https://lore.kernel.org/20260831054331.625505-3-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Lorenzo Stoakes Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato Signed-off-by: Andrew Morton --- mm/pgtable-generic.c | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c index b91b1a98029c7f..224c444cf45d97 100644 --- a/mm/pgtable-generic.c +++ b/mm/pgtable-generic.c @@ -26,14 +26,20 @@ void pgd_clear_bad(pgd_t *pgd) { - pgd_ERROR(*pgd); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pgd_val(*pgd)); + pr_err("bad pgd %s.\n", str); pgd_clear(pgd); } #ifndef __PAGETABLE_P4D_FOLDED void p4d_clear_bad(p4d_t *p4d) { - p4d_ERROR(*p4d); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, p4d_val(*p4d)); + pr_err("bad p4d %s.\n", str); p4d_clear(p4d); } #endif @@ -41,7 +47,10 @@ void p4d_clear_bad(p4d_t *p4d) #ifndef __PAGETABLE_PUD_FOLDED void pud_clear_bad(pud_t *pud) { - pud_ERROR(*pud); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pud_val(*pud)); + pr_err("bad pud %s.\n", str); pud_clear(pud); } #endif @@ -53,7 +62,10 @@ void pud_clear_bad(pud_t *pud) */ void pmd_clear_bad(pmd_t *pmd) { - pmd_ERROR(*pmd); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pmd_val(*pmd)); + pr_err("bad pmd %s.\n", str); pmd_clear(pmd); } From 8d5c1366213824421b83164d8de8bfc711bde777 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:25 +0530 Subject: [PATCH 685/857] loongarch/mm: stop using pte_ERROR() Directly use pr_err() in __set_fixmap() and drop pte_ERROR() which helps in eventually dropping pte_ERROR() macro across the tree. In this new printing __FILE__ and __LINE__ has been dropped because they are always the same and don't really add any value. The new ptval_to_str() helper is being used for converting pgtable entry value into a string. The error message itself has been cleaned up as well. Link: https://lore.kernel.org/20260831054331.625505-4-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Huacai Chen Cc: WANG Xuerui Cc: Geert Uytterhoeven Cc: Helge Deller Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: Yoshinori Sato Signed-off-by: Andrew Morton --- arch/loongarch/include/asm/pgtable.h | 2 -- arch/loongarch/mm/init.c | 4 +++- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h index a05f6a4928dc68..eddd8906b77dfc 100644 --- a/arch/loongarch/include/asm/pgtable.h +++ b/arch/loongarch/include/asm/pgtable.h @@ -135,8 +135,6 @@ struct vm_area_struct; #define ptep_get(ptep) READ_ONCE(*(ptep)) #define pmdp_get(pmdp) READ_ONCE(*(pmdp)) -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) #ifndef __PAGETABLE_PMD_FOLDED #define pmd_ERROR(e) \ pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) diff --git a/arch/loongarch/mm/init.c b/arch/loongarch/mm/init.c index 4b46c5d30708d8..f801f7097f0379 100644 --- a/arch/loongarch/mm/init.c +++ b/arch/loongarch/mm/init.c @@ -197,13 +197,15 @@ void __init __set_fixmap(enum fixed_addresses idx, phys_addr_t phys, pgprot_t flags) { unsigned long addr = __fix_to_virt(idx); + char str[PTVAL_STR_MAX]; pte_t *ptep; BUG_ON(idx <= FIX_HOLE || idx >= __end_of_fixed_addresses); ptep = populate_kernel_pte(addr); if (!pte_none(ptep_get(ptep))) { - pte_ERROR(*ptep); + ptval_to_str(str, pte_val(*ptep)); + pr_err("unexpected set PTE at %lx in %s: %s\n", addr, __func__, str); return; } From 8e9b5e98c0e07c101079b2fa43659a600a1e2388 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:26 +0530 Subject: [PATCH 686/857] parisc/mm: directly use generic [pmd|pgd]_clear_bad() Drop [pmd|pgd]_ERROR() followed by [pmd|pgd]_clear() instances. But instead directly use semantically equivalent generic helpers [pmd|pgd]_clear_bad() in unmap_uncached_[pte|pmd]() which helps in dropping their corresponding [pmd|pgd]_ERROR() macros across the tree. Link: https://lore.kernel.org/20260831054331.625505-5-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Reviewed-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: James E.J. Bottomley Cc: Helge Deller Cc: Geert Uytterhoeven Cc: Huacai Chen Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato Signed-off-by: Andrew Morton --- arch/parisc/kernel/pci-dma.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/arch/parisc/kernel/pci-dma.c b/arch/parisc/kernel/pci-dma.c index bf9f192c826ebe..84e7309a826696 100644 --- a/arch/parisc/kernel/pci-dma.c +++ b/arch/parisc/kernel/pci-dma.c @@ -160,8 +160,7 @@ static inline void unmap_uncached_pte(pmd_t * pmd, unsigned long vaddr, if (pmd_none(*pmd)) return; if (pmd_bad(*pmd)) { - pmd_ERROR(*pmd); - pmd_clear(pmd); + pmd_clear_bad(pmd); return; } pte = pte_offset_kernel(pmd, vaddr); @@ -196,8 +195,7 @@ static inline void unmap_uncached_pmd(pgd_t * dir, unsigned long vaddr, if (pgd_none(*dir)) return; if (pgd_bad(*dir)) { - pgd_ERROR(*dir); - pgd_clear(dir); + pgd_clear_bad(dir); return; } pmd = pmd_offset(pud_offset(p4d_offset(dir, vaddr), vaddr), vaddr); From 91618c4e30a9a7e982962989d91fdbbd265a5b75 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:27 +0530 Subject: [PATCH 687/857] sh/mm: stop using pte_ERROR() Directly use pr_err() in set_pte_phys() and drop pte_ERROR() which helps in eventually dropping pte_ERROR() macro across the tree. In this new printing __FILE__ and __LINE__ has been dropped because they are always the same and don't really add any value. Besides ptrval_to_str() has been able to handle different PTE representation with and without CONFIG_X2TLB, which helped in unifying error message printing. Link: https://lore.kernel.org/20260831054331.625505-6-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Yoshinori Sato Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: Lorenzo Stoakes Cc: Samuel Holland Cc: WANG Xuerui Signed-off-by: Andrew Morton --- arch/sh/include/asm/pgtable_32.h | 5 ----- arch/sh/mm/init.c | 6 +++++- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/arch/sh/include/asm/pgtable_32.h b/arch/sh/include/asm/pgtable_32.h index 5f51af18997b57..c8eb9a7a4c4c78 100644 --- a/arch/sh/include/asm/pgtable_32.h +++ b/arch/sh/include/asm/pgtable_32.h @@ -401,14 +401,9 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #define pmd_page(pmd) (virt_to_page(pmd_val(pmd))) #ifdef CONFIG_X2TLB -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %p(%08lx%08lx).\n", __FILE__, __LINE__, \ - &(e), (e).pte_high, (e).pte_low) #define pgd_ERROR(e) \ printk("%s:%d: bad pgd %016llx.\n", __FILE__, __LINE__, pgd_val(e)) #else -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) #define pgd_ERROR(e) \ printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #endif diff --git a/arch/sh/mm/init.c b/arch/sh/mm/init.c index 110308bdef01d0..9466ae6f9f164d 100644 --- a/arch/sh/mm/init.c +++ b/arch/sh/mm/init.c @@ -84,7 +84,11 @@ static void set_pte_phys(unsigned long addr, unsigned long phys, pgprot_t prot) pte = __get_pte_phys(addr); if (!pte_none(*pte)) { - pte_ERROR(*pte); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pte_val(*pte)); + pr_err("unexpected set PTE at %lx in %s: bad pte %p(%s).\n", + addr, __func__, pte, str); return; } From b2ab8317ddf1d37f5590c6fdcc62e7602fddedc9 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:28 +0530 Subject: [PATCH 688/857] sh/mm: stop using [p4d|pud|pmd]_ERROR() Stop using [p4d|pud|pmd]_ERROR() in __get_pte_phys() as the pgtable entries are known to be NULL and hence could not really be accessed. Link: https://lore.kernel.org/20260831054331.625505-7-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Reviewed-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Yoshinori Sato Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: Lorenzo Stoakes Cc: Samuel Holland Cc: WANG Xuerui Signed-off-by: Andrew Morton --- arch/sh/mm/init.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/sh/mm/init.c b/arch/sh/mm/init.c index 9466ae6f9f164d..8d65e60688dcbd 100644 --- a/arch/sh/mm/init.c +++ b/arch/sh/mm/init.c @@ -59,19 +59,19 @@ static pte_t *__get_pte_phys(unsigned long addr) p4d = p4d_alloc(NULL, pgd, addr); if (unlikely(!p4d)) { - p4d_ERROR(*p4d); + pr_err("allocating p4d table failed\n"); return NULL; } pud = pud_alloc(NULL, p4d, addr); if (unlikely(!pud)) { - pud_ERROR(*pud); + pr_err("allocating pud table failed\n"); return NULL; } pmd = pmd_alloc(NULL, pud, addr); if (unlikely(!pmd)) { - pmd_ERROR(*pmd); + pr_err("allocating pmd table failed\n"); return NULL; } From a7f9e233c6d457888d9589d29f3a51e3b5dc32cc Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:29 +0530 Subject: [PATCH 689/857] sh/mm: stop using pgd_ERROR() Stop using pgd_ERROR() in __get_pte_phys() when page table entry is already known to be empty. Link: https://lore.kernel.org/20260831054331.625505-8-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Reviewed-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Yoshinori Sato Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: Lorenzo Stoakes Cc: Samuel Holland Cc: WANG Xuerui Signed-off-by: Andrew Morton --- arch/sh/mm/init.c | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/arch/sh/mm/init.c b/arch/sh/mm/init.c index 8d65e60688dcbd..93921109f4e64b 100644 --- a/arch/sh/mm/init.c +++ b/arch/sh/mm/init.c @@ -52,10 +52,8 @@ static pte_t *__get_pte_phys(unsigned long addr) pmd_t *pmd; pgd = pgd_offset_k(addr); - if (pgd_none(*pgd)) { - pgd_ERROR(*pgd); + if (pgd_none(*pgd)) return NULL; - } p4d = p4d_alloc(NULL, pgd, addr); if (unlikely(!p4d)) { From 3381159234488f92a68f5d8ef9aa6c52be16e8cb Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:30 +0530 Subject: [PATCH 690/857] mm: drop pxd_ERROR() There are no more users left for any pxd_ERROR() either in generic MM or in the platform MM. Hence all these platform macros along with their generic fallback could be dropped across the tree. Link: https://lore.kernel.org/20260831054331.625505-9-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Acked-by: Geert Uytterhoeven # m68k Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato Signed-off-by: Andrew Morton --- arch/alpha/include/asm/pgtable.h | 7 ------- arch/arc/include/asm/pgtable-levels.h | 11 ----------- arch/arm/include/asm/pgtable.h | 7 ------- arch/arm/kernel/traps.c | 17 ----------------- arch/arm64/include/asm/pgtable.h | 15 --------------- arch/csky/include/asm/pgtable.h | 4 ---- arch/hexagon/include/asm/pgtable.h | 3 --- arch/loongarch/include/asm/pgtable.h | 11 ----------- arch/m68k/include/asm/mcf_pgtable.h | 6 ------ arch/m68k/include/asm/motorola_pgtable.h | 8 -------- arch/m68k/include/asm/sun3_pgtable.h | 7 ------- arch/microblaze/include/asm/pgtable.h | 7 ------- arch/mips/include/asm/pgtable-32.h | 10 ---------- arch/mips/include/asm/pgtable-64.h | 13 ------------- arch/nios2/include/asm/pgtable.h | 7 ------- arch/openrisc/include/asm/pgtable.h | 7 ------- arch/parisc/include/asm/pgtable.h | 9 --------- arch/powerpc/include/asm/book3s/32/pgtable.h | 2 -- arch/powerpc/include/asm/book3s/64/pgtable.h | 7 ------- arch/powerpc/include/asm/nohash/32/pgtable.h | 2 -- .../powerpc/include/asm/nohash/64/pgtable-4k.h | 3 --- arch/powerpc/include/asm/nohash/64/pgtable.h | 5 ----- arch/riscv/include/asm/page.h | 6 ------ arch/riscv/include/asm/pgtable-64.h | 9 --------- arch/riscv/include/asm/pgtable.h | 4 ---- arch/s390/include/asm/pgtable.h | 11 ----------- arch/sh/include/asm/pgtable-3level.h | 3 --- arch/sh/include/asm/pgtable_32.h | 8 -------- arch/sparc/include/asm/pgtable_32.h | 3 --- arch/sparc/include/asm/pgtable_64.h | 10 ---------- arch/um/include/asm/pgtable-2level.h | 7 ------- arch/um/include/asm/pgtable-4level.h | 13 ------------- arch/x86/include/asm/pgtable-2level.h | 5 ----- arch/x86/include/asm/pgtable-3level.h | 11 ----------- arch/x86/include/asm/pgtable_64.h | 18 ------------------ arch/xtensa/include/asm/pgtable.h | 4 ---- include/asm-generic/pgtable-nop4d.h | 1 - include/asm-generic/pgtable-nopmd.h | 1 - include/asm-generic/pgtable-nopud.h | 1 - 39 files changed, 283 deletions(-) diff --git a/arch/alpha/include/asm/pgtable.h b/arch/alpha/include/asm/pgtable.h index 8e00cf9dc39dea..7cac8241ee674c 100644 --- a/arch/alpha/include/asm/pgtable.h +++ b/arch/alpha/include/asm/pgtable.h @@ -357,13 +357,6 @@ static inline pte_t pte_swp_clear_exclusive(pte_t pte) return pte; } -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern void paging_init(void); /* We have our own get_unmapped_area */ diff --git a/arch/arc/include/asm/pgtable-levels.h b/arch/arc/include/asm/pgtable-levels.h index c8f9273372c073..167b82fcfafe36 100644 --- a/arch/arc/include/asm/pgtable-levels.h +++ b/arch/arc/include/asm/pgtable-levels.h @@ -98,8 +98,6 @@ /* * 1st level paging: pgd */ -#define pgd_ERROR(e) \ - pr_crit("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #if CONFIG_PGTABLE_LEVELS > 3 @@ -115,9 +113,6 @@ /* * 2nd level paging: pud */ -#define pud_ERROR(e) \ - pr_crit("%s:%d: bad pud %08lx.\n", __FILE__, __LINE__, pud_val(e)) - #endif #if CONFIG_PGTABLE_LEVELS > 2 @@ -137,9 +132,6 @@ /* * 3rd level paging: pmd */ -#define pmd_ERROR(e) \ - pr_crit("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) - #define pmd_pfn(pmd) ((pmd_val(pmd) & PMD_MASK) >> PAGE_SHIFT) #define pfn_pmd(pfn,prot) __pmd(((pfn) << PAGE_SHIFT) | pgprot_val(prot)) @@ -165,9 +157,6 @@ /* * 4th level paging: pte */ -#define pte_ERROR(e) \ - pr_crit("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) - #define PFN_PTE_SHIFT PAGE_SHIFT #define pte_none(x) (!pte_val(x)) #define pte_present(x) (pte_val(x) & _PAGE_PRESENT) diff --git a/arch/arm/include/asm/pgtable.h b/arch/arm/include/asm/pgtable.h index 982795cf45637e..8dd17d20faa33f 100644 --- a/arch/arm/include/asm/pgtable.h +++ b/arch/arm/include/asm/pgtable.h @@ -44,13 +44,6 @@ #define LIBRARY_TEXT_START 0x0c000000 #ifndef __ASSEMBLY__ -extern void __pte_error(const char *file, int line, pte_t); -extern void __pmd_error(const char *file, int line, pmd_t); -extern void __pgd_error(const char *file, int line, pgd_t); - -#define pte_ERROR(pte) __pte_error(__FILE__, __LINE__, pte) -#define pmd_ERROR(pmd) __pmd_error(__FILE__, __LINE__, pmd) -#define pgd_ERROR(pgd) __pgd_error(__FILE__, __LINE__, pgd) /* * This is the lowest virtual address we can permit any user space diff --git a/arch/arm/kernel/traps.c b/arch/arm/kernel/traps.c index afbd2ebe5c39dc..ad04c806cc9d8f 100644 --- a/arch/arm/kernel/traps.c +++ b/arch/arm/kernel/traps.c @@ -753,23 +753,6 @@ void __readwrite_bug(const char *fn) } EXPORT_SYMBOL(__readwrite_bug); -#ifdef CONFIG_MMU -void __pte_error(const char *file, int line, pte_t pte) -{ - pr_err("%s:%d: bad pte %08llx.\n", file, line, (long long)pte_val(pte)); -} - -void __pmd_error(const char *file, int line, pmd_t pmd) -{ - pr_err("%s:%d: bad pmd %08llx.\n", file, line, (long long)pmd_val(pmd)); -} - -void __pgd_error(const char *file, int line, pgd_t pgd) -{ - pr_err("%s:%d: bad pgd %08llx.\n", file, line, (long long)pgd_val(pgd)); -} -#endif - asmlinkage void __div0(void) { pr_err("Division by zero in kernel.\n"); diff --git a/arch/arm64/include/asm/pgtable.h b/arch/arm64/include/asm/pgtable.h index 6000905a2e865e..e89ec5f4787b49 100644 --- a/arch/arm64/include/asm/pgtable.h +++ b/arch/arm64/include/asm/pgtable.h @@ -107,9 +107,6 @@ static inline void arch_leave_lazy_mmu_mode(void) __flush_tlb_range(vma, address, address + PMD_SIZE, PMD_SIZE, 2, \ TLBF_NOBROADCAST | TLBF_NONOTIFY | TLBF_NOWALKCACHE) -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %016llx.\n", __FILE__, __LINE__, pte_val(e)) - #ifdef CONFIG_ARM64_PA_BITS_52 static inline phys_addr_t __pte_to_phys(pte_t pte) { @@ -866,9 +863,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #if CONFIG_PGTABLE_LEVELS > 2 -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016llx.\n", __FILE__, __LINE__, pmd_val(e)) - #define pud_none(pud) (!pud_val(pud)) #define pud_bad(pud) ((pud_val(pud) & PUD_TYPE_MASK) != \ PUD_TYPE_TABLE) @@ -960,9 +954,6 @@ static inline bool mm_pud_folded(const struct mm_struct *mm) } #define mm_pud_folded mm_pud_folded -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016llx.\n", __FILE__, __LINE__, pud_val(e)) - #define p4d_none(p4d) (pgtable_l4_enabled() && !p4d_val(p4d)) #define p4d_bad(p4d) (pgtable_l4_enabled() && \ ((p4d_val(p4d) & P4D_TYPE_MASK) != \ @@ -1088,9 +1079,6 @@ static inline bool mm_p4d_folded(const struct mm_struct *mm) } #define mm_p4d_folded mm_p4d_folded -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %016llx.\n", __FILE__, __LINE__, p4d_val(e)) - #define pgd_none(pgd) (pgtable_l5_enabled() && !pgd_val(pgd)) #define pgd_bad(pgd) (pgtable_l5_enabled() && \ ((pgd_val(pgd) & PGD_TYPE_MASK) != \ @@ -1217,9 +1205,6 @@ p4d_t *p4d_offset_lockless_folded(pgd_t *pgdp, pgd_t pgd, unsigned long addr) #endif /* CONFIG_PGTABLE_LEVELS > 4 */ -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %016llx.\n", __FILE__, __LINE__, pgd_val(e)) - #define pgd_set_fixmap(addr) ((pgd_t *)set_fixmap_offset(FIX_PGD, addr)) #define pgd_clear_fixmap() clear_fixmap(FIX_PGD) diff --git a/arch/csky/include/asm/pgtable.h b/arch/csky/include/asm/pgtable.h index bafcd5823531a5..5ca77ff89ef939 100644 --- a/arch/csky/include/asm/pgtable.h +++ b/arch/csky/include/asm/pgtable.h @@ -23,10 +23,6 @@ #define PTRS_PER_PMD 1 #define PTRS_PER_PTE (PAGE_SIZE / sizeof(pte_t)) -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, (e).pte_low) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #define PFN_PTE_SHIFT PAGE_SHIFT #define pmd_pfn(pmd) (pmd_phys(pmd) >> PAGE_SHIFT) diff --git a/arch/hexagon/include/asm/pgtable.h b/arch/hexagon/include/asm/pgtable.h index 27b269e2870d37..2fdb27afe70329 100644 --- a/arch/hexagon/include/asm/pgtable.h +++ b/arch/hexagon/include/asm/pgtable.h @@ -94,9 +94,6 @@ #endif /* Any bigger and the PTE disappears. */ -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__,\ - pgd_val(e)) /* * Page Protection Constants. Includes (in this variant) cache attributes. diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h index eddd8906b77dfc..cf29a4c8ac593a 100644 --- a/arch/loongarch/include/asm/pgtable.h +++ b/arch/loongarch/include/asm/pgtable.h @@ -135,17 +135,6 @@ struct vm_area_struct; #define ptep_get(ptep) READ_ONCE(*(ptep)) #define pmdp_get(pmdp) READ_ONCE(*(pmdp)) -#ifndef __PAGETABLE_PMD_FOLDED -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#endif -#ifndef __PAGETABLE_PUD_FOLDED -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) -#endif -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern pte_t invalid_pte_table[PTRS_PER_PTE]; #ifndef __PAGETABLE_PUD_FOLDED diff --git a/arch/m68k/include/asm/mcf_pgtable.h b/arch/m68k/include/asm/mcf_pgtable.h index 189bb7b1e6630f..f45a882238dbb6 100644 --- a/arch/m68k/include/asm/mcf_pgtable.h +++ b/arch/m68k/include/asm/mcf_pgtable.h @@ -137,12 +137,6 @@ static inline int pmd_bad2(pmd_t *pmd) { return 0; } #define pmd_present(pmd) (!pmd_none2(&(pmd))) static inline void pmd_clear(pmd_t *pmdp) { pmd_val(*pmdp) = 0; } -#define pte_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pte %08lx.\n", \ - __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %08lx.\n", \ - __FILE__, __LINE__, pgd_val(e)) /* * The following only work if pte_present() is true. diff --git a/arch/m68k/include/asm/motorola_pgtable.h b/arch/m68k/include/asm/motorola_pgtable.h index dcf6829b3eab97..d9393b310add6a 100644 --- a/arch/m68k/include/asm/motorola_pgtable.h +++ b/arch/m68k/include/asm/motorola_pgtable.h @@ -131,14 +131,6 @@ static inline void pud_set(pud_t *pudp, pmd_t *pmdp) #define pud_clear(pudp) ({ pud_val(*pudp) = 0; }) #define pud_page(pud) (mem_map + ((unsigned long)(__va(pud_val(pud)) - PAGE_OFFSET) >> PAGE_SHIFT)) -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - - /* * The following only work if pte_present() is true. * Undefined behaviour if not.. diff --git a/arch/m68k/include/asm/sun3_pgtable.h b/arch/m68k/include/asm/sun3_pgtable.h index 80ca185a18a193..704442a391fd74 100644 --- a/arch/m68k/include/asm/sun3_pgtable.h +++ b/arch/m68k/include/asm/sun3_pgtable.h @@ -119,13 +119,6 @@ static inline int pmd_present2 (pmd_t *pmd) { return pmd_val (*pmd) & SUN3_PMD_V #define pmd_present(pmd) (!pmd_none2(&(pmd))) static inline void pmd_clear (pmd_t *pmdp) { pmd_val (*pmdp) = 0; } - -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - - /* * The following only work if pte_present() is true. * Undefined behaviour if not... diff --git a/arch/microblaze/include/asm/pgtable.h b/arch/microblaze/include/asm/pgtable.h index 7678c040a2fd36..72708f9af1c0b8 100644 --- a/arch/microblaze/include/asm/pgtable.h +++ b/arch/microblaze/include/asm/pgtable.h @@ -103,13 +103,6 @@ extern pte_t *va_to_pte(unsigned long address); #define USER_PGD_PTRS (PAGE_OFFSET >> PGDIR_SHIFT) #define KERNEL_PGD_PTRS (PTRS_PER_PGD-USER_PGD_PTRS) -#define pte_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pte "PTE_FMT".\n", \ - __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %08lx.\n", \ - __FILE__, __LINE__, pgd_val(e)) - /* * Bits in a linux-style PTE. These match the bits in the * (hardware-defined) PTE as closely as possible. diff --git a/arch/mips/include/asm/pgtable-32.h b/arch/mips/include/asm/pgtable-32.h index 92b7591aac2acd..ef1001ab09c5be 100644 --- a/arch/mips/include/asm/pgtable-32.h +++ b/arch/mips/include/asm/pgtable-32.h @@ -104,16 +104,6 @@ extern int add_temporary_entry(unsigned long entrylo0, unsigned long entrylo1, # define VMALLOC_END (FIXADDR_START-2*PAGE_SIZE) #endif -#ifdef CONFIG_PHYS_ADDR_T_64BIT -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %016Lx.\n", __FILE__, __LINE__, pte_val(e)) -#else -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#endif -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern void load_pgd(unsigned long pg_dir); extern pte_t invalid_pte_table[PTRS_PER_PTE]; diff --git a/arch/mips/include/asm/pgtable-64.h b/arch/mips/include/asm/pgtable-64.h index 6e854bb11f37de..785fc37bab9417 100644 --- a/arch/mips/include/asm/pgtable-64.h +++ b/arch/mips/include/asm/pgtable-64.h @@ -151,19 +151,6 @@ #define MODULES_END (FIXADDR_START-2*PAGE_SIZE) #endif -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) -#ifndef __PAGETABLE_PMD_FOLDED -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#endif -#ifndef __PAGETABLE_PUD_FOLDED -#define pud_ERROR(e) \ - printk("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) -#endif -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern pte_t invalid_pte_table[PTRS_PER_PTE]; #ifndef __PAGETABLE_PUD_FOLDED diff --git a/arch/nios2/include/asm/pgtable.h b/arch/nios2/include/asm/pgtable.h index d389aa9ca57ce0..272707d48f1bb1 100644 --- a/arch/nios2/include/asm/pgtable.h +++ b/arch/nios2/include/asm/pgtable.h @@ -223,13 +223,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) return pmd_val(pmd); } -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx.\n", \ - __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", \ - __FILE__, __LINE__, pgd_val(e)) - /* * Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that * are !pte_none() && !pte_present(). diff --git a/arch/openrisc/include/asm/pgtable.h b/arch/openrisc/include/asm/pgtable.h index 6b89996d0b628e..13afcc0bd8631b 100644 --- a/arch/openrisc/include/asm/pgtable.h +++ b/arch/openrisc/include/asm/pgtable.h @@ -338,13 +338,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #define pte_pfn(x) ((unsigned long)(((x).pte)) >> PAGE_SHIFT) #define pfn_pte(pfn, prot) __pte((((pfn) << PAGE_SHIFT)) | pgprot_val(prot)) -#define pte_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pte %p(%08lx).\n", \ - __FILE__, __LINE__, &(e), pte_val(e)) -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %p(%08lx).\n", \ - __FILE__, __LINE__, &(e), pgd_val(e)) - extern pgd_t swapper_pg_dir[PTRS_PER_PGD]; /* defined in head.S */ struct vm_area_struct; diff --git a/arch/parisc/include/asm/pgtable.h b/arch/parisc/include/asm/pgtable.h index 467b8547ac8bfa..f6899375cb4393 100644 --- a/arch/parisc/include/asm/pgtable.h +++ b/arch/parisc/include/asm/pgtable.h @@ -75,15 +75,6 @@ extern void __update_cache(pte_t pte); #endif /* !__ASSEMBLER__ */ -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#if CONFIG_PGTABLE_LEVELS == 3 -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, (unsigned long)pmd_val(e)) -#endif -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, (unsigned long)pgd_val(e)) - /* This is the size of the initially mapped kernel memory */ #if defined(CONFIG_64BIT) || defined(CONFIG_KALLSYMS) #define KERNEL_INITIAL_ORDER 26 /* 1<<26 = 64MB */ diff --git a/arch/powerpc/include/asm/book3s/32/pgtable.h b/arch/powerpc/include/asm/book3s/32/pgtable.h index e18a4fa282a1b6..835e84caee13fa 100644 --- a/arch/powerpc/include/asm/book3s/32/pgtable.h +++ b/arch/powerpc/include/asm/book3s/32/pgtable.h @@ -203,8 +203,6 @@ void unmap_kernel_page(unsigned long va); /* Bits to mask out from a PGD to get to the PUD page */ #define PGD_MASKED_BITS 0 -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) /* * Bits in a linux-style PTE. These match the bits in the * (hardware-defined) PowerPC PTE as closely as possible. diff --git a/arch/powerpc/include/asm/book3s/64/pgtable.h b/arch/powerpc/include/asm/book3s/64/pgtable.h index f4db7d7fbd5c62..dff8790a047db5 100644 --- a/arch/powerpc/include/asm/book3s/64/pgtable.h +++ b/arch/powerpc/include/asm/book3s/64/pgtable.h @@ -991,13 +991,6 @@ static inline pmd_t *pud_pgtable(pud_t pud) return (pmd_t *)__va(pud_val(pud) & ~PUD_MASKED_BITS); } -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %08lx.\n", __FILE__, __LINE__, pud_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - static inline int map_kernel_page(unsigned long ea, unsigned long pa, pgprot_t prot) { if (radix_enabled()) { diff --git a/arch/powerpc/include/asm/nohash/32/pgtable.h b/arch/powerpc/include/asm/nohash/32/pgtable.h index 496ecc65ac255a..f17afde89fa13d 100644 --- a/arch/powerpc/include/asm/nohash/32/pgtable.h +++ b/arch/powerpc/include/asm/nohash/32/pgtable.h @@ -51,8 +51,6 @@ #define USER_PTRS_PER_PGD (TASK_SIZE / PGDIR_SIZE) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08llx.\n", __FILE__, __LINE__, (unsigned long long)pgd_val(e)) /* * This is the bottom of the PKMAP area with HIGHMEM or an arbitrary diff --git a/arch/powerpc/include/asm/nohash/64/pgtable-4k.h b/arch/powerpc/include/asm/nohash/64/pgtable-4k.h index fb6fa1d4e0749a..75cf3c331b9292 100644 --- a/arch/powerpc/include/asm/nohash/64/pgtable-4k.h +++ b/arch/powerpc/include/asm/nohash/64/pgtable-4k.h @@ -82,9 +82,6 @@ extern struct page *p4d_page(p4d_t p4d); #endif /* !__ASSEMBLER__ */ -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %08lx.\n", __FILE__, __LINE__, pud_val(e)) - /* * On all 4K setups, remap_4k_pfn() equates to remap_pfn_range() */ #define remap_4k_pfn(vma, addr, pfn, prot) \ diff --git a/arch/powerpc/include/asm/nohash/64/pgtable.h b/arch/powerpc/include/asm/nohash/64/pgtable.h index 661eb3820d1291..446dde8b6ead54 100644 --- a/arch/powerpc/include/asm/nohash/64/pgtable.h +++ b/arch/powerpc/include/asm/nohash/64/pgtable.h @@ -159,11 +159,6 @@ static inline void huge_ptep_set_wrprotect(struct mm_struct *mm, __young; \ }) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - /* * Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that * are !pte_none() && !pte_present(). diff --git a/arch/riscv/include/asm/page.h b/arch/riscv/include/asm/page.h index 709a36fb432343..b4bbae55e93111 100644 --- a/arch/riscv/include/asm/page.h +++ b/arch/riscv/include/asm/page.h @@ -76,12 +76,6 @@ typedef struct page *pgtable_t; #define __pgd(x) ((pgd_t) { (x) }) #define __pgprot(x) ((pgprot_t) { (x) }) -#ifdef CONFIG_64BIT -#define PTE_FMT "%016lx" -#else -#define PTE_FMT "%08lx" -#endif - #if defined(CONFIG_64BIT) && defined(CONFIG_MMU) /* * We override this value as its generic definition uses __pa too early in diff --git a/arch/riscv/include/asm/pgtable-64.h b/arch/riscv/include/asm/pgtable-64.h index 6e789fa58514c7..ae23182b572cdd 100644 --- a/arch/riscv/include/asm/pgtable-64.h +++ b/arch/riscv/include/asm/pgtable-64.h @@ -264,15 +264,6 @@ static inline unsigned long _pmd_pfn(pmd_t pmd) return __page_val_to_pfn(pmd_val(pmd)); } -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) - -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) - -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %016lx.\n", __FILE__, __LINE__, p4d_val(e)) - static inline void set_p4d(p4d_t *p4dp, p4d_t p4d) { if (pgtable_l4_enabled) diff --git a/arch/riscv/include/asm/pgtable.h b/arch/riscv/include/asm/pgtable.h index 40b1ed4f3ea893..4c8fc684550311 100644 --- a/arch/riscv/include/asm/pgtable.h +++ b/arch/riscv/include/asm/pgtable.h @@ -556,10 +556,6 @@ static inline pte_t pte_modify(pte_t pte, pgprot_t newprot) return __pte((pte_val(pte) & _PAGE_CHG_MASK) | newprot_val); } -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd " PTE_FMT ".\n", __FILE__, __LINE__, pgd_val(e)) - - /* Commit new configuration to MMU hardware */ static inline void update_mmu_cache_range(struct vm_fault *vmf, struct vm_area_struct *vma, unsigned long address, diff --git a/arch/s390/include/asm/pgtable.h b/arch/s390/include/asm/pgtable.h index e882663a58e776..2d5c2ab06de988 100644 --- a/arch/s390/include/asm/pgtable.h +++ b/arch/s390/include/asm/pgtable.h @@ -68,17 +68,6 @@ extern unsigned long zero_page_mask; /* TODO: s390 cannot support io_remap_pfn_range... */ -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %016lx.\n", __FILE__, __LINE__, p4d_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - /* * The vmalloc and module area will always be on the topmost area of the * kernel mapping. 512GB are reserved for vmalloc by default. diff --git a/arch/sh/include/asm/pgtable-3level.h b/arch/sh/include/asm/pgtable-3level.h index d1ce73f3bd85ef..3f4d747f30f40b 100644 --- a/arch/sh/include/asm/pgtable-3level.h +++ b/arch/sh/include/asm/pgtable-3level.h @@ -25,9 +25,6 @@ #define PTRS_PER_PMD ((1 << PGDIR_SHIFT) / PMD_SIZE) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %016llx.\n", __FILE__, __LINE__, pmd_val(e)) - typedef union { struct { unsigned long pmd_low; diff --git a/arch/sh/include/asm/pgtable_32.h b/arch/sh/include/asm/pgtable_32.h index c8eb9a7a4c4c78..cde1bf0c67342b 100644 --- a/arch/sh/include/asm/pgtable_32.h +++ b/arch/sh/include/asm/pgtable_32.h @@ -400,14 +400,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #define pmd_pfn(pmd) (__pa(pmd_val(pmd)) >> PAGE_SHIFT) #define pmd_page(pmd) (virt_to_page(pmd_val(pmd))) -#ifdef CONFIG_X2TLB -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %016llx.\n", __FILE__, __LINE__, pgd_val(e)) -#else -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) -#endif - /* * Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that * are !pte_none() && !pte_present(). diff --git a/arch/sparc/include/asm/pgtable_32.h b/arch/sparc/include/asm/pgtable_32.h index f89b1250661dff..5a5f54a090f5b2 100644 --- a/arch/sparc/include/asm/pgtable_32.h +++ b/arch/sparc/include/asm/pgtable_32.h @@ -40,9 +40,6 @@ void load_mmu(void); unsigned long calc_highpages(void); unsigned long __init bootmem_init(unsigned long *pages_avail); -#define pte_ERROR(e) __builtin_trap() -#define pmd_ERROR(e) __builtin_trap() -#define pgd_ERROR(e) __builtin_trap() #define PTRS_PER_PTE 64 #define PTRS_PER_PMD 64 diff --git a/arch/sparc/include/asm/pgtable_64.h b/arch/sparc/include/asm/pgtable_64.h index 0837ebbc5dce63..44d1333065a6ae 100644 --- a/arch/sparc/include/asm/pgtable_64.h +++ b/arch/sparc/include/asm/pgtable_64.h @@ -96,16 +96,6 @@ bool kern_addr_valid(unsigned long addr); #define PTRS_PER_PUD (1UL << PUD_BITS) #define PTRS_PER_PGD (1UL << PGDIR_BITS) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %p(%016lx) seen at (%pS)\n", \ - __FILE__, __LINE__, &(e), pmd_val(e), __builtin_return_address(0)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %p(%016lx) seen at (%pS)\n", \ - __FILE__, __LINE__, &(e), pud_val(e), __builtin_return_address(0)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %p(%016lx) seen at (%pS)\n", \ - __FILE__, __LINE__, &(e), pgd_val(e), __builtin_return_address(0)) - #endif /* !(__ASSEMBLER__) */ /* PTE bits which are the same in SUN4U and SUN4V format. */ diff --git a/arch/um/include/asm/pgtable-2level.h b/arch/um/include/asm/pgtable-2level.h index 14ec16f92ce408..fa625f5b5ef750 100644 --- a/arch/um/include/asm/pgtable-2level.h +++ b/arch/um/include/asm/pgtable-2level.h @@ -24,13 +24,6 @@ #define USER_PTRS_PER_PGD ((TASK_SIZE + (PGDIR_SIZE - 1)) / PGDIR_SIZE) #define PTRS_PER_PGD 1024 -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %p(%08lx).\n", __FILE__, __LINE__, &(e), \ - pte_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %p(%08lx).\n", __FILE__, __LINE__, &(e), \ - pgd_val(e)) - static inline int pgd_needsync(pgd_t pgd) { return 0; } static inline void pgd_mkuptodate(pgd_t pgd) { } diff --git a/arch/um/include/asm/pgtable-4level.h b/arch/um/include/asm/pgtable-4level.h index 7a271b7b83d2bd..ff82f99c80fa10 100644 --- a/arch/um/include/asm/pgtable-4level.h +++ b/arch/um/include/asm/pgtable-4level.h @@ -42,19 +42,6 @@ #define USER_PTRS_PER_PGD ((TASK_SIZE + (PGDIR_SIZE - 1)) / PGDIR_SIZE) -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pte_val(e)) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pmd_val(e)) -#define pud_ERROR(e) \ - printk("%s:%d: bad pud %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pud_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pgd_val(e)) - #define pud_none(x) (!(pud_val(x) & ~_PAGE_NEEDSYNC)) #define pud_bad(x) ((pud_val(x) & (~PAGE_MASK & ~_PAGE_USER)) != _KERNPG_TABLE) #define pud_present(x) (pud_val(x) & _PAGE_PRESENT) diff --git a/arch/x86/include/asm/pgtable-2level.h b/arch/x86/include/asm/pgtable-2level.h index e9482a11ac52d6..83427765cfbfd2 100644 --- a/arch/x86/include/asm/pgtable-2level.h +++ b/arch/x86/include/asm/pgtable-2level.h @@ -2,11 +2,6 @@ #ifndef _ASM_X86_PGTABLE_2LEVEL_H #define _ASM_X86_PGTABLE_2LEVEL_H -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx\n", __FILE__, __LINE__, (e).pte_low) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx\n", __FILE__, __LINE__, pgd_val(e)) - /* * Certain architectures need to do special things when PTEs * within a page table are directly modified. Thus, the following diff --git a/arch/x86/include/asm/pgtable-3level.h b/arch/x86/include/asm/pgtable-3level.h index dabafba957ea6f..d6729911e09a28 100644 --- a/arch/x86/include/asm/pgtable-3level.h +++ b/arch/x86/include/asm/pgtable-3level.h @@ -8,17 +8,6 @@ * * Copyright (C) 1999 Ingo Molnar */ - -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %p(%08lx%08lx)\n", \ - __FILE__, __LINE__, &(e), (e).pte_high, (e).pte_low) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %p(%016Lx)\n", \ - __FILE__, __LINE__, &(e), pmd_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %p(%016Lx)\n", \ - __FILE__, __LINE__, &(e), pgd_val(e)) - #define pxx_xchg64(_pxx, _ptr, _val) ({ \ _pxx##val_t *_p = (_pxx##val_t *)_ptr; \ _pxx##val_t _o = *_p; \ diff --git a/arch/x86/include/asm/pgtable_64.h b/arch/x86/include/asm/pgtable_64.h index ce45882ccd071b..c861f3832bed2e 100644 --- a/arch/x86/include/asm/pgtable_64.h +++ b/arch/x86/include/asm/pgtable_64.h @@ -29,24 +29,6 @@ extern pgd_t init_top_pgt[]; extern void paging_init(void); static inline void sync_initial_page_table(void) { } -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pte_val(e)) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pmd_val(e)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pud_val(e)) - -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), p4d_val(e)) - -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pgd_val(e)) - struct mm_struct; #define mm_p4d_folded mm_p4d_folded diff --git a/arch/xtensa/include/asm/pgtable.h b/arch/xtensa/include/asm/pgtable.h index f00a879dc298a5..60fb67a9972907 100644 --- a/arch/xtensa/include/asm/pgtable.h +++ b/arch/xtensa/include/asm/pgtable.h @@ -204,10 +204,6 @@ */ #ifndef __ASSEMBLER__ -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd entry %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #ifdef CONFIG_MMU extern pgd_t swapper_pg_dir[PAGE_SIZE/sizeof(pgd_t)]; diff --git a/include/asm-generic/pgtable-nop4d.h b/include/asm-generic/pgtable-nop4d.h index 89c21f84cffbe2..1cf739ee38aa15 100644 --- a/include/asm-generic/pgtable-nop4d.h +++ b/include/asm-generic/pgtable-nop4d.h @@ -22,7 +22,6 @@ static inline int pgd_none(pgd_t pgd) { return 0; } static inline int pgd_bad(pgd_t pgd) { return 0; } static inline int pgd_present(pgd_t pgd) { return 1; } static inline void pgd_clear(pgd_t *pgd) { } -#define p4d_ERROR(p4d) (pgd_ERROR((p4d).pgd)) #define pgd_populate(mm, pgd, p4d) do { } while (0) #define pgd_populate_safe(mm, pgd, p4d) do { } while (0) diff --git a/include/asm-generic/pgtable-nopmd.h b/include/asm-generic/pgtable-nopmd.h index 36b6490ed18081..ff4235cf84d772 100644 --- a/include/asm-generic/pgtable-nopmd.h +++ b/include/asm-generic/pgtable-nopmd.h @@ -33,7 +33,6 @@ static inline int pud_present(pud_t pud) { return 1; } static inline int pud_user(pud_t pud) { return 0; } static inline int pud_leaf(pud_t pud) { return 0; } static inline void pud_clear(pud_t *pud) { } -#define pmd_ERROR(pmd) (pud_ERROR((pmd).pud)) #define pud_populate(mm, pmd, pte) do { } while (0) diff --git a/include/asm-generic/pgtable-nopud.h b/include/asm-generic/pgtable-nopud.h index 356cbfbaab2476..eedee8e3ad68fd 100644 --- a/include/asm-generic/pgtable-nopud.h +++ b/include/asm-generic/pgtable-nopud.h @@ -29,7 +29,6 @@ static inline int p4d_none(p4d_t p4d) { return 0; } static inline int p4d_bad(p4d_t p4d) { return 0; } static inline int p4d_present(p4d_t p4d) { return 1; } static inline void p4d_clear(p4d_t *p4d) { } -#define pud_ERROR(pud) (p4d_ERROR((pud).p4d)) #define p4d_populate(mm, p4d, pud) do { } while (0) #define p4d_populate_safe(mm, p4d, pud) do { } while (0) From 030fe75dee087e44ddf010bc0ef77174f01dca63 Mon Sep 17 00:00:00 2001 From: Johannes Weiner Date: Sun, 30 Aug 2026 12:29:17 +0800 Subject: [PATCH 691/857] mm: add page_counter_margin() Patch series "mm: avoid large folio splits when swap is unavailable", v7. This is v7 of Barry's original RFC patch, "mm: Avoiding split large folios if swap has no space": https://lore.kernel.org/r/20260618221720.71768-1-baohua@kernel.org Barry's RFC showed the no-swap case with MADV_PAGEOUT on 16KB mTHP: the large-folio split counter increased by 1024 even though no swapout progress was possible. Skipping the split in that case kept the counter at 0. This series makes folio_alloc_swap() classify failures according to whether splitting a large folio might allow swapout to make progress. Callers can then avoid destroying the large folio when neither global swap availability nor the folio's memcg swap hierarchy has capacity for even a smaller folio. Patch #1 adds page_counter_margin(), a small helper that computes the minimum remaining chargeable space across a page_counter hierarchy. Patch #2 establishes the folio_alloc_swap() return-value contract: - -E2BIG: splitting may let smaller folios make progress - -ENOSPC: no global swap space is available - -ENOMEM: splitting is not expected to help, including when the folio's memcg swap hierarchy has no remaining capacity Patch #3 makes vmscan split a large folio only when folio_alloc_swap() returns -E2BIG. Other failures keep the existing activation path and avoid destroying the large folio when no smaller part can be backed by swap either. Patch #4 applies the same contract to shmem_writeout(), which currently splits a large folio on every folio_alloc_swap() failure. It now enters the split fallback only on -E2BIG; other failures redirty and reactivate the folio as before. Testing: With a 1GB anonymous mapping backed by 16KB mTHPs and memory.swap.max=0, the patch reduced the median latency of 30 process_madvise(MADV_PAGEOUT) runs from 743.8 ms to 181.7 ms, while the number of large-folio splits per run dropped from 65536 to 0. Neither kernel swapped out any pages. I also ran DaCapo h2 under swap pressure and found no statistically significant change in wall time or CPU time. The overall benefit appears minor and workload-dependent. This patch (of 4): mem_cgroup_get_nr_swap_pages() open-codes the remaining capacity across the memcg swap counter hierarchy. Add page_counter_margin() to return the minimum usable space from a page counter to the root, and use it in mem_cgroup_get_nr_swap_pages(). This is a pure refactoring with no intended behavior change. Link: https://lore.kernel.org/20260830042920.2280454-1-xueyuan.chen21@gmail.com Link: https://lore.kernel.org/20260830042920.2280454-2-xueyuan.chen21@gmail.com Signed-off-by: Johannes Weiner Signed-off-by: Xueyuan Chen Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Barry Song Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Cc: Nanzhe Zhao Cc: Youngjun Park Signed-off-by: Andrew Morton --- include/linux/page_counter.h | 1 + mm/memcontrol.c | 9 +++------ mm/page_counter.c | 20 ++++++++++++++++++++ 3 files changed, 24 insertions(+), 6 deletions(-) diff --git a/include/linux/page_counter.h b/include/linux/page_counter.h index d649b6bbbc871b..07b7cb12249c7c 100644 --- a/include/linux/page_counter.h +++ b/include/linux/page_counter.h @@ -68,6 +68,7 @@ static inline unsigned long page_counter_read(struct page_counter *counter) return atomic_long_read(&counter->usage); } +long page_counter_margin(struct page_counter *counter); void page_counter_cancel(struct page_counter *counter, unsigned long nr_pages); void page_counter_charge(struct page_counter *counter, unsigned long nr_pages); bool page_counter_try_charge(struct page_counter *counter, diff --git a/mm/memcontrol.c b/mm/memcontrol.c index d8c22070f24a5b..0ecb60ce245ea3 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5842,12 +5842,9 @@ long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) { long nr_swap_pages = get_nr_swap_pages(); - if (mem_cgroup_disabled() || do_memsw_account()) - return nr_swap_pages; - for (; !mem_cgroup_is_root(memcg); memcg = parent_mem_cgroup(memcg)) - nr_swap_pages = min_t(long, nr_swap_pages, - READ_ONCE(memcg->swap.max) - - page_counter_read(&memcg->swap)); + if (!mem_cgroup_disabled() && !do_memsw_account()) + nr_swap_pages = min(nr_swap_pages, page_counter_margin(&memcg->swap)); + return nr_swap_pages; } diff --git a/mm/page_counter.c b/mm/page_counter.c index 661e0f2a5127a5..450543f4b318b6 100644 --- a/mm/page_counter.c +++ b/mm/page_counter.c @@ -46,6 +46,26 @@ static void propagate_protected_usage(struct page_counter *c, } } +/** + * page_counter_margin - remaining usable space within hierarchical limits + * @counter: counter + * + * Return: The minimum value of max minus usage across @counter and all of + * its ancestors. The value may be negative during a concurrent charge. + */ +long page_counter_margin(struct page_counter *counter) +{ + long margin = PAGE_COUNTER_MAX; + + do { + long m = READ_ONCE(counter->max) - page_counter_read(counter); + + margin = min(margin, m); + } while ((counter = counter->parent)); + + return margin; +} + /** * page_counter_cancel - take pages out of the local counter * @counter: counter From 70de3cf215f39ca1f6a68d0ca8ceef0655357a34 Mon Sep 17 00:00:00 2001 From: Xueyuan Chen Date: Sun, 30 Aug 2026 12:29:18 +0800 Subject: [PATCH 692/857] mm: distinguish large folio swap allocation failures folio_alloc_swap() reports most failures with generic negative error codes. Reclaim callers consequently cannot tell whether splitting a large folio could make progress, or whether no swap space is available for even a single page. Classify failures using both the global free swap count and the remaining capacity in the folio's memcg swap hierarchy. Return -ENOSPC when global swap space is exhausted, -ENOMEM when splitting cannot overcome the failure, and -E2BIG for a large folio when allocating or charging a smaller folio might still succeed. Use this classification for all folio_alloc_swap() failure paths, including capability rejection, swap slot allocation failure, and memcg swap charge failure. Callers are updated separately to split large folios only on -E2BIG. Link: https://lore.kernel.org/20260830042920.2280454-3-xueyuan.chen21@gmail.com Signed-off-by: Xueyuan Chen Suggested-by: Kairui Song Suggested-by: Barry Song Suggested-by: Youngjun Park Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Johannes Weiner Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nanzhe Zhao Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Signed-off-by: Andrew Morton --- include/linux/swap.h | 6 ++++++ mm/memcontrol.c | 23 +++++++++++++++++++++++ mm/swapfile.c | 26 +++++++++++++++++++------- 3 files changed, 48 insertions(+), 7 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 5658a1634b85ea..7a43409879caed 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -508,6 +508,7 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_p __mem_cgroup_uncharge_swap(id, nr_pages); } +long mem_cgroup_get_folio_swap_margin(struct folio *folio); extern long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg); extern bool mem_cgroup_swap_full(struct folio *folio); #else @@ -521,6 +522,11 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, { } +static inline long mem_cgroup_get_folio_swap_margin(struct folio *folio) +{ + return PAGE_COUNTER_MAX; +} + static inline long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) { return get_nr_swap_pages(); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 0ecb60ce245ea3..5b9e0ebd42ae0f 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5848,6 +5848,29 @@ long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) return nr_swap_pages; } +/** + * mem_cgroup_get_folio_swap_margin - get a folio's memcg swap margin + * @folio: folio whose memcg margin is queried + * + * Return: Remaining chargeable pages in the folio's memcg hierarchy. + */ +long mem_cgroup_get_folio_swap_margin(struct folio *folio) +{ + struct mem_cgroup *memcg; + long margin; + + if (mem_cgroup_disabled() || do_memsw_account() || + !folio_memcg_charged(folio)) + return PAGE_COUNTER_MAX; + + rcu_read_lock(); + memcg = folio_memcg(folio); + margin = page_counter_margin(&memcg->swap); + rcu_read_unlock(); + + return margin; +} + bool mem_cgroup_swap_full(struct folio *folio) { struct mem_cgroup *memcg; diff --git a/mm/swapfile.c b/mm/swapfile.c index 408f6c72fb5a69..01e7b6b046b67d 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1735,7 +1735,9 @@ static int swap_dup_entries_cluster(struct swap_info_struct *si, * swap cache. * * Context: Caller needs to hold the folio lock. - * Return: Whether the folio was added to the swap cache. + * Return: %0 on success, %-E2BIG if splitting the folio might allow swapout, + * %-ENOSPC if no global swap space is available, or %-ENOMEM if splitting + * would not help. */ int folio_alloc_swap(struct folio *folio) { @@ -1747,11 +1749,11 @@ int folio_alloc_swap(struct folio *folio) if (order) { /* - * Reject large allocation when THP_SWAP is disabled, - * the caller should split the folio and try again. + * Reject large allocation when THP_SWAP is disabled. Check below + * whether splitting and retrying can make progress. */ if (!IS_ENABLED(CONFIG_THP_SWAP)) - return -EAGAIN; + goto failed; /* * Allocation size should never exceed cluster size @@ -1759,7 +1761,7 @@ int folio_alloc_swap(struct folio *folio) */ if (size > SWAPFILE_CLUSTER) { VM_WARN_ON_ONCE(1); - return -EINVAL; + goto failed; } } @@ -1775,13 +1777,23 @@ int folio_alloc_swap(struct folio *folio) } /* Need to call this even if allocation failed, for MEMCG_SWAP_FAIL. */ - if (unlikely(mem_cgroup_try_charge_swap(folio))) + if (unlikely(mem_cgroup_try_charge_swap(folio))) { swap_cache_del_folio(folio); + goto failed; + } if (unlikely(!folio_test_swapcache(folio))) - return -ENOMEM; + goto failed; return 0; + +failed: + if (get_nr_swap_pages() <= 0) + return -ENOSPC; + if (mem_cgroup_get_folio_swap_margin(folio) <= 0) + return -ENOMEM; + + return order ? -E2BIG : -ENOMEM; } /** From b0c1ced4db1d3456b1640a7b1e629205713a5ed9 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Sun, 30 Aug 2026 12:29:19 +0800 Subject: [PATCH 693/857] mm/vmscan: avoid pointless large folio splits without swap When swap is disabled, exhausted, or unavailable due to memcg swap limits, splitting a large anonymous folio cannot make swapout progress. The fallback only destroys the large folio and inflates split statistics. Use -E2BIG from folio_alloc_swap() as the explicit signal that splitting the folio might allow swapout of smaller pieces. For other allocation failures, keep the existing activation path and avoid the split. This preserves the split fallback for fragmented or partially available swap, while avoiding it when there is no backing space for any part of the folio. Link: https://lore.kernel.org/20260830042920.2280454-4-xueyuan.chen21@gmail.com Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Xueyuan Chen Reported-by: Nanzhe Zhao Acked-by: David Hildenbrand (Arm) Reviewed-by: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Cc: Youngjun Park Signed-off-by: Andrew Morton --- mm/vmscan.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 413efe44d1f695..4d51b0d8e28768 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1259,6 +1259,8 @@ static unsigned int shrink_folio_list(struct list_head *folio_list, */ if (folio_test_anon(folio) && folio_test_swapbacked(folio) && !folio_test_swapcache(folio)) { + int ret; + if (!(sc->gfp_mask & __GFP_IO)) goto keep_locked; if (folio_maybe_dma_pinned(folio)) @@ -1277,11 +1279,14 @@ static unsigned int shrink_folio_list(struct list_head *folio_list, split_folio_to_list(folio, folio_list)) goto activate_locked; } - if (folio_alloc_swap(folio)) { + ret = folio_alloc_swap(folio); + if (ret) { int __maybe_unused order = folio_order(folio); if (!folio_test_large(folio)) goto activate_locked_split; + if (ret != -E2BIG) + goto activate_locked; /* Fallback to swap normal pages */ if (split_folio_to_list(folio, folio_list)) goto activate_locked; From 127f8a47305a4e370e98da3b1b0d54e717c903a0 Mon Sep 17 00:00:00 2001 From: Xueyuan Chen Date: Sun, 30 Aug 2026 12:29:20 +0800 Subject: [PATCH 694/857] mm/shmem: split large folios only on -E2BIG shmem_writeout() currently splits a large folio on every folio_alloc_swap() failure. With the refined return-value contract, only -E2BIG indicates that splitting might allow smaller folios to be swapped out. Enter the split fallback only for -E2BIG. For -ENOSPC and -ENOMEM, redirty and reactivate the folio as before. Link: https://lore.kernel.org/20260830042920.2280454-5-xueyuan.chen21@gmail.com Signed-off-by: Xueyuan Chen Suggested-by: Baolin Wang Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Reviewed-by: Barry Song Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nanzhe Zhao Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Cc: Youngjun Park Signed-off-by: Andrew Morton --- mm/shmem.c | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index eeb9a78c125a55..255d69ebceba09 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1813,7 +1813,7 @@ int shmem_writeout(struct swap_io_ctx *ctx, struct folio *folio, struct shmem_inode_info *info = SHMEM_I(inode); struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); pgoff_t index; - int nr_pages; + int nr_pages, ret; bool split = false; if ((info->flags & SHMEM_F_LOCKED) || sbinfo->noswap) @@ -1894,7 +1894,8 @@ int shmem_writeout(struct swap_io_ctx *ctx, struct folio *folio, folio_mark_uptodate(folio); } - if (!folio_alloc_swap(folio)) { + ret = folio_alloc_swap(folio); + if (!ret) { bool first_swapped = shmem_recalc_inode(inode, 0, nr_pages); int error; @@ -1947,7 +1948,7 @@ int shmem_writeout(struct swap_io_ctx *ctx, struct folio *folio, swap_cache_del_folio(folio); goto redirty; } - if (nr_pages > 1) + if (nr_pages > 1 && ret == -E2BIG) goto try_split; redirty: folio_mark_dirty(folio); From 0b2b7b3ba8d4193b871c6f73d1946a66c237b7f3 Mon Sep 17 00:00:00 2001 From: Aristeu Rozanski Date: Thu, 27 Aug 2026 21:55:43 -0400 Subject: [PATCH 695/857] mm: gup: move pmd_protnone() into gup_fast_pmd_leaf() Patch series "mm: gup: cleanup gup_fast call chain", v3. These two patches implement the refactor in gup_fast call chain David Hildenbrand mentioned in [1]. This patch (of 2): Make pmd handling match pud handling by calling pmd_protnone() inside gup_fast_pmd_leaf(). Link: https://lore.kernel.org/20260828015542.125576330@ruivo.org Link: https://lore.kernel.org/20260828015542.245315718@ruivo.org Link: https://lore.kernel.org/all/85e760cf-b994-40db-8d13-221feee55c60@redhat.com/T/#u [1] Signed-off-by: Aristeu Rozanski Suggested-by: David Hildenbrand Link: https://lore.kernel.org/all/85e760cf-b994-40db-8d13-221feee55c60@redhat.com/T/#u Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Signed-off-by: Andrew Morton --- mm/gup.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index eb898ea1ee22e5..bd981ab3975143 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -2929,6 +2929,10 @@ static int gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, unsigned long addr, struct folio *folio; int refs; + /* See gup_fast_pte_range() */ + if (pmd_protnone(orig)) + return 0; + if (!pmd_access_permitted(orig, flags & FOLL_WRITE)) return 0; @@ -3024,10 +3028,6 @@ static int gup_fast_pmd_range(pud_t *pudp, pud_t pud, unsigned long addr, return 0; if (unlikely(pmd_leaf(pmd))) { - /* See gup_fast_pte_range() */ - if (pmd_protnone(pmd)) - return 0; - if (!gup_fast_pmd_leaf(pmd, pmdp, addr, next, flags, pages, nr)) return 0; From fd00b018b3b900cc457002e883d164b7b1c00a73 Mon Sep 17 00:00:00 2001 From: Aristeu Rozanski Date: Thu, 27 Aug 2026 21:55:44 -0400 Subject: [PATCH 696/857] mm: gup: cleanup the gup_fast_*() call chain Refactor gup_fast functions so each step of the way returns the number of pages pinned. Because the previous step of the chain knows what the number it should be, less indicates an error. This way there's no need to pass *nr along. Link: https://lore.kernel.org/20260828015542.334186653@ruivo.org Signed-off-by: Aristeu Rozanski Suggested-by: David Hildenbrand Link: https://lore.kernel.org/all/85e760cf-b994-40db-8d13-221feee55c60@redhat.com/T/#u Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Signed-off-by: Andrew Morton --- mm/gup.c | 179 +++++++++++++++++++++++++++++-------------------------- 1 file changed, 94 insertions(+), 85 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index bd981ab3975143..a4036c02e2137f 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -2826,11 +2826,11 @@ static bool gup_fast_folio_allowed(struct folio *folio, unsigned int flags) * also check pmd here to make sure pmd doesn't change (corresponds to * pmdp_collapse_flush() in the THP collapse code path). */ -static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { - int ret = 0; + unsigned long nr_pages = 0; pte_t *ptep, *ptem; ptem = ptep = pte_offset_map(&pmd, addr); @@ -2892,15 +2892,13 @@ static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, goto pte_unmap; } folio_set_referenced(folio); - pages[*nr] = page; - (*nr)++; + pages[nr_pages] = page; + nr_pages++; } while (ptep++, addr += PAGE_SIZE, addr != end); - ret = 1; - pte_unmap: pte_unmap(ptem); - return ret; + return nr_pages; } #else @@ -2913,21 +2911,21 @@ static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, * get_user_pages_fast_only implementation that can pin pages. Thus it's still * useful to have gup_fast_pmd_leaf even if we can't operate on ptes. */ -static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { return 0; } #endif /* CONFIG_ARCH_HAS_PTE_SPECIAL */ -static int gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { struct page *page; struct folio *folio; - int refs; + unsigned long nr_pages, i; /* See gup_fast_pte_range() */ if (pmd_protnone(orig)) @@ -2939,42 +2937,40 @@ static int gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, unsigned long addr, if (pmd_special(orig)) return 0; - refs = (end - addr) >> PAGE_SHIFT; + nr_pages = (end - addr) >> PAGE_SHIFT; page = pmd_page(orig) + ((addr & ~PMD_MASK) >> PAGE_SHIFT); - folio = try_grab_folio_fast(page, refs, flags); + folio = try_grab_folio_fast(page, nr_pages, flags); if (!folio) return 0; if (unlikely(pmd_val(orig) != pmd_val(pmdp_get_lockless(pmdp)))) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!gup_fast_folio_allowed(folio, flags)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!pmd_write(orig) && gup_must_unshare(NULL, flags, &folio->page)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } - pages += *nr; - *nr += refs; - for (; refs; refs--) + for (i = 0; i < nr_pages; i++) *(pages++) = page++; folio_set_referenced(folio); - return 1; + return nr_pages; } -static int gup_fast_pud_leaf(pud_t orig, pud_t *pudp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pud_leaf(pud_t orig, pud_t *pudp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { struct page *page; struct folio *folio; - int refs; + unsigned long nr_pages, i; if (!pud_access_permitted(orig, flags & FOLL_WRITE)) return 0; @@ -2982,41 +2978,39 @@ static int gup_fast_pud_leaf(pud_t orig, pud_t *pudp, unsigned long addr, if (pud_special(orig)) return 0; - refs = (end - addr) >> PAGE_SHIFT; + nr_pages = (end - addr) >> PAGE_SHIFT; page = pud_page(orig) + ((addr & ~PUD_MASK) >> PAGE_SHIFT); - folio = try_grab_folio_fast(page, refs, flags); + folio = try_grab_folio_fast(page, nr_pages, flags); if (!folio) return 0; if (unlikely(pud_val(orig) != pud_val(pudp_get(pudp)))) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!gup_fast_folio_allowed(folio, flags)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!pud_write(orig) && gup_must_unshare(NULL, flags, &folio->page)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } - pages += *nr; - *nr += refs; - for (; refs; refs--) + for (i = 0; i < nr_pages; i++) *(pages++) = page++; folio_set_referenced(folio); - return 1; + return nr_pages; } -static int gup_fast_pmd_range(pud_t *pudp, pud_t pud, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pmd_range(pud_t *pudp, pud_t pud, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; pmd_t *pmdp; pmdp = pmd_offset_lockless(pudp, pud, addr); @@ -3025,26 +3019,30 @@ static int gup_fast_pmd_range(pud_t *pudp, pud_t pud, unsigned long addr, next = pmd_addr_end(addr, end); if (!pmd_present(pmd)) - return 0; + break; if (unlikely(pmd_leaf(pmd))) { - if (!gup_fast_pmd_leaf(pmd, pmdp, addr, next, flags, - pages, nr)) - return 0; - - } else if (!gup_fast_pte_range(pmd, pmdp, addr, next, flags, - pages, nr)) - return 0; + chunk_nr_pages = gup_fast_pmd_leaf(pmd, pmdp, addr, + next, flags, + &pages[nr_pages]); + + } else + chunk_nr_pages = gup_fast_pte_range(pmd, pmdp, addr, + next, flags, + &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (pmdp++, addr = next, addr != end); - return 1; + return nr_pages; } -static int gup_fast_pud_range(p4d_t *p4dp, p4d_t p4d, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pud_range(p4d_t *p4dp, p4d_t p4d, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; pud_t *pudp; pudp = pud_offset_lockless(p4dp, p4d, addr); @@ -3053,24 +3051,27 @@ static int gup_fast_pud_range(p4d_t *p4dp, p4d_t p4d, unsigned long addr, next = pud_addr_end(addr, end); if (unlikely(!pud_present(pud))) - return 0; - if (unlikely(pud_leaf(pud))) { - if (!gup_fast_pud_leaf(pud, pudp, addr, next, flags, - pages, nr)) - return 0; - } else if (!gup_fast_pmd_range(pudp, pud, addr, next, flags, - pages, nr)) - return 0; + break; + if (unlikely(pud_leaf(pud))) + chunk_nr_pages = gup_fast_pud_leaf(pud, pudp, addr, + next, flags, + &pages[nr_pages]); + else + chunk_nr_pages = gup_fast_pmd_range(pudp, pud, addr, + next, flags, + &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (pudp++, addr = next, addr != end); - return 1; + return nr_pages; } -static int gup_fast_p4d_range(pgd_t *pgdp, pgd_t pgd, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_p4d_range(pgd_t *pgdp, pgd_t pgd, unsigned long addr, + unsigned long end, unsigned int flags, struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; p4d_t *p4dp; p4dp = p4d_offset_lockless(pgdp, pgd, addr); @@ -3079,20 +3080,23 @@ static int gup_fast_p4d_range(pgd_t *pgdp, pgd_t pgd, unsigned long addr, next = p4d_addr_end(addr, end); if (!p4d_present(p4d)) - return 0; + break; BUILD_BUG_ON(p4d_leaf(p4d)); - if (!gup_fast_pud_range(p4dp, p4d, addr, next, flags, - pages, nr)) - return 0; + chunk_nr_pages = gup_fast_pud_range(p4dp, p4d, addr, next, + flags, &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (p4dp++, addr = next, addr != end); - return 1; + return nr_pages; } -static void gup_fast_pgd_range(unsigned long addr, unsigned long end, - unsigned int flags, struct page **pages, int *nr) +static unsigned long gup_fast_pgd_range(unsigned long addr, + unsigned long end, unsigned int flags, + struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; pgd_t *pgdp; pgdp = pgd_offset(current->mm, addr); @@ -3101,17 +3105,23 @@ static void gup_fast_pgd_range(unsigned long addr, unsigned long end, next = pgd_addr_end(addr, end); if (pgd_none(pgd)) - return; + break; BUILD_BUG_ON(pgd_leaf(pgd)); - if (!gup_fast_p4d_range(pgdp, pgd, addr, next, flags, - pages, nr)) - return; + chunk_nr_pages = gup_fast_p4d_range(pgdp, pgd, addr, next, + flags, &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (pgdp++, addr = next, addr != end); + + return nr_pages; } #else -static inline void gup_fast_pgd_range(unsigned long addr, unsigned long end, - unsigned int flags, struct page **pages, int *nr) +static inline unsigned long gup_fast_pgd_range(unsigned long addr, + unsigned long end, unsigned int flags, + struct page **pages) { + return 0; } #endif /* CONFIG_HAVE_GUP_FAST */ @@ -3129,8 +3139,7 @@ static bool gup_fast_permitted(unsigned long start, unsigned long end) static unsigned long gup_fast(unsigned long start, unsigned long end, unsigned int gup_flags, struct page **pages) { - unsigned long flags; - int nr_pinned = 0; + unsigned long flags, nr_pinned; unsigned seq; if (!IS_ENABLED(CONFIG_HAVE_GUP_FAST) || @@ -3154,7 +3163,7 @@ static unsigned long gup_fast(unsigned long start, unsigned long end, * that come from callers of tlb_remove_table_sync_one(). */ local_irq_save(flags); - gup_fast_pgd_range(start, end, gup_flags, pages, &nr_pinned); + nr_pinned = gup_fast_pgd_range(start, end, gup_flags, pages); local_irq_restore(flags); /* From 9fe8c5e2546c94803bfa0e83866dbfdb960dc932 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:55:15 +0800 Subject: [PATCH 697/857] mm/page_table_check: add explicit pmd_none check in pte_clear_range In __page_table_check_pte_clear_range(), the condition to determine whether to iterate over PTEs only checked pmd_bad() and pmd_leaf(). This relies on the implicit assumption that pmd_none() is always a subset of pmd_bad() on all architectures supporting PAGE_TABLE_CHECK. While this assumption currently holds for x86_64, arm64, s390, riscv, and powerpc, it is an architecture-dependent behavior that may not hold for future architectures. Add an explicit pmd_none() check to make the intent clear and avoid calling pte_offset_map() on an empty PMD, which could lead to undefined behavior. Link: https://lore.kernel.org/20260819025516.2967199-1-ye.liu@linux.dev Signed-off-by: Ye Liu Cc: Albert Ou Cc: Alexandre Ghiti Cc: Palmer Dabbelt Cc: Pasha Tatashin Signed-off-by: Andrew Morton --- mm/page_table_check.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/page_table_check.c b/mm/page_table_check.c index 6ffc536359cd06..ed3e1a76f26643 100644 --- a/mm/page_table_check.c +++ b/mm/page_table_check.c @@ -278,7 +278,7 @@ void __page_table_check_pte_clear_range(struct mm_struct *mm, if (&init_mm == mm) return; - if (!pmd_bad(pmd) && !pmd_leaf(pmd)) { + if (!pmd_none(pmd) && !pmd_bad(pmd) && !pmd_leaf(pmd)) { pte_t *ptep = pte_offset_map(&pmd, addr); unsigned long i; From c91e6aab95fb6833b3a82b69912d81b71a4375b7 Mon Sep 17 00:00:00 2001 From: Zhiling Zou Date: Tue, 21 Jul 2026 23:56:38 +0800 Subject: [PATCH 698/857] mm/page_table_check: skip zero pages page_table_check_set() accounts pages by whether they are PageAnon(). The shared zero page is a special page, not an ordinary file-backed page. Read faults on private anonymous mappings can install many read-only PTEs that point at the zero page, but page_table_check currently accounts them in file_map_count. That lets an unprivileged process populate enough zero-page mappings to overflow file_map_count and trip the BUG_ON() in page_table_check_set(). Skip zero pages in page_table_check accounting. They do not need the anonymous/file mapping conflict checks that page_table_check performs for ordinary pages, and this keeps the existing counter size and page_ext layout unchanged. Link: https://lore.kernel.org/1f8848512d2e3ded944f8d595c29faee8fdaeab0.1784645969.git.roxy520tt@gmail.com Fixes: df4e817b7108 ("mm: page table check") Signed-off-by: Zhiling Zou Signed-off-by: Ren Wei Cc: Ye Liu Reported-by: Vega Assisted-by: Codex:gpt-5.4 Cc: Signed-off-by: Andrew Morton --- mm/page_table_check.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/page_table_check.c b/mm/page_table_check.c index ed3e1a76f26643..143a918c0bde22 100644 --- a/mm/page_table_check.c +++ b/mm/page_table_check.c @@ -67,7 +67,7 @@ static void page_table_check_clear(unsigned long pfn, unsigned long pgcnt) struct page *page; bool anon; - if (!pfn_valid(pfn)) + if (!pfn_valid(pfn) || is_zero_pfn(pfn) || is_huge_zero_pfn(pfn)) return; page = pfn_to_page(pfn); @@ -102,7 +102,7 @@ static void page_table_check_set(unsigned long pfn, unsigned long pgcnt, struct page *page; bool anon; - if (!pfn_valid(pfn)) + if (!pfn_valid(pfn) || is_zero_pfn(pfn) || is_huge_zero_pfn(pfn)) return; page = pfn_to_page(pfn); From 5af97fcf53220bdd3331139de945f3389b91252e Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:42 -0700 Subject: [PATCH 699/857] mm/damon/core: skip applying scheme if region split for quota fails Patch series "mm/damon: fix DAMOS bugs in core, paddr and vaddr". Fix misc bugs of DAMOS. Patch 1 makes DAMOS less stress memory allocator under extreme situation. Patches 2 and 3 fix wrong folios walking in DAMON_PADDR. Patches 4 and 5 fix wrong folios walking in DAMON_VADDR. Patches 6-8 handle extreme and unlikely memory situations that can cause divide by zero and underflow. All bugs are discovered by Sashiko. This patch (of 8): damos_apply_scheme() splits a region and apply the action to the subregion if it is needed for not violating the quota. The split operation (damon_split_region_at()) could fail for allocation failure. In the case, the quota could be violated. From the user's perspective, DAMOS becomes more aggressive than expected under the extreme situation. Handle the failure. The user impact is not critical. The failure of damon_split_region_at() is unlikely since it is arguably too small to fail. Also DAMOS being aggressive is limited to the single region. Users can set min_nr_regions to set the maximum size of each region. If it is reasonably set, the transient overhead shouldn't be critical. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-1-sj@kernel.org Link: https://lore.kernel.org/20260901131850.98037-2-sj@kernel.org Link: https://lore.kernel.org/20260718171523.87547-1-sj@kernel.org [1] Fixes: 2b8a248d5873 ("mm/damon/schemes: implement size quota for schemes application speed control") Signed-off-by: SJ Park Cc: Usama Arif Cc: # 5.16.x Signed-off-by: Andrew Morton --- mm/damon/core.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 3f89cfdf5f0221..a6fdb3068262da 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2604,7 +2604,8 @@ static void damos_apply_scheme(struct damon_ctx *c, struct damon_target *t, c->min_region_sz); if (!sz) goto update_stat; - damon_split_region_at(t, r, sz); + if (damon_split_region_at(t, r, sz)) + goto update_stat; } if (damos_core_filter_out(c, t, r, s)) return; From 92cde011385fcff239f04ff18e14526163d64ffc Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:43 -0700 Subject: [PATCH 700/857] mm/damon/paddr: respect folio end for DAMOS_STAT The function for applying DAMOS_STAT in DAMON physical address space operation set (paddr), namely damon_pa_stat(), applies DAMOS filters to folios of the given region. For that, it gets folios of addresses in the region. It starts from the region start address and advances the address by the size of the folio of the address until it goes out of the region. If the start address is in the middle of a large folio, and if the next folios are small, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that the DAMOS_STAT-based page level monitoring results become inaccurate. Since the page level monitoring is supposed to provide relatively high precision, this is definitely a problem. It is arguably not critical since it is only monitoring quality degradation. Link: https://lore.kernel.org/20260901131850.98037-3-sj@kernel.org Fixes: bdbe1d7bc325 ("mm/damon/paddr: increment pa_stat damon address range by folio size") Signed-off-by: SJ Park Cc: Usama Arif Cc: # 6.14.x Signed-off-by: Andrew Morton --- mm/damon/paddr.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 5c6c3a597fd0bf..2ab7b3842701ed 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -379,7 +379,7 @@ static unsigned long damon_pa_stat(struct damon_region *r, if (!damos_pa_filter_out(s, folio)) *sz_filter_passed += folio_size(folio) / addr_unit; - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } s->last_applied = folio; From c8ad43eb299bf9a18b1ded2fb8992ae5769abf90 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:44 -0700 Subject: [PATCH 701/857] mm/damon/paddr: respect folio end for DAMOS actions except STAT A few functions for applying DAMOS actions including pageout, lru_[de]prio and migrate_{hot,cold} in DAMON physical address space operation set (paddr) collect folios of the given region by getting the folios of region-internal addresses. Then, those functions apply the action to the collected folios at once. The collection starts from the region start address and advances the address by the size of the folio of the address until it goes out of the region. If the start address is in the middle of a large folio, and if the next folios are small, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that DAMOS action is applied to less than expected amount of memory. Given the best effort nature of DAMON, it is no big problem, but it is clearly a bug that is better to be fixed. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-4-sj@kernel.org Link: https://lore.kernel.org/20260517234112.89245-1-sj@kernel.org [1] Fixes: 3a06696305e7 ("mm/damon/ops: have damon_get_folio return folio even for tail pages") Signed-off-by: SJ Park Cc: Usama Arif Cc: # 6.15.x Signed-off-by: Andrew Morton --- mm/damon/paddr.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 2ab7b3842701ed..9ddd1ec8202b7f 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -264,7 +264,7 @@ static unsigned long damon_pa_pageout(struct damon_region *r, else list_add(&folio->lru, &folio_list); put_folio: - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } if (install_young_filter) @@ -302,7 +302,7 @@ static inline unsigned long damon_pa_de_activate( folio_deactivate(folio); applied += folio_nr_pages(folio); put_folio: - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } s->last_applied = folio; @@ -350,7 +350,7 @@ static unsigned long damon_pa_migrate(struct damon_region *r, folio_is_file_lru(folio)); list_add(&folio->lru, &folio_list); put_folio: - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } applied = damon_migrate_pages(&folio_list, s->target_nid); From 6cab689613552fd4a9b3c05849d7b836c199c426 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:45 -0700 Subject: [PATCH 702/857] mm/damon/vaddr: respect folio end for DAMOS_STAT For applying DAMOS_STAT action to a region, DAMON virtual address space operation set (vaddr) calls walk_page_range[_vma]() for the region. The pmd walk entry function, namely damon_va_stat_pmd_entry(), applies DAMOS filters to folios of addresses of the region in the pmd. It starts from the walking address and advances the address by the size of the folio of the address until it goes out of the pmd or the region. Let's suppose it is for the first pmd of the region, and the region start address is in the middle of a large folio. Also, the next folios are small. Then, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that the DAMOS_STAT-based page level monitoring results become inaccurate. Since the page level monitoring is supposed to provide relatively high precision, this is definitely a problem. It is arguably not critical since it is only monitoring quality degradation. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-5-sj@kernel.org Link: https://lore.kernel.org/20260514015053.149396-1-sj@kernel.org [1] Fixes: 63f39737d1e3 ("mm/damon/vaddr: support stat-purpose DAMOS filters") Signed-off-by: SJ Park Cc: Usama Arif Cc: # 6.18.x Signed-off-by: Andrew Morton --- mm/damon/vaddr.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 0648400b2d65b4..4b0b5edf679527 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -831,6 +831,8 @@ static int damos_va_stat_pmd_entry(pmd_t *pmd, unsigned long addr, return 0; for (; addr < next; pte += nr, addr += nr * PAGE_SIZE) { + unsigned long page_idx; + nr = 1; ptent = ptep_get(pte); @@ -844,7 +846,8 @@ static int damos_va_stat_pmd_entry(pmd_t *pmd, unsigned long addr, if (!damos_va_filter_out(s, folio, vma, addr, pte, NULL)) *sz_filter_passed += folio_size(folio); - nr = folio_nr_pages(folio); + page_idx = folio_page_idx(folio, pte_page(ptent)); + nr = folio_nr_pages(folio) - page_idx; s->last_applied = folio; } pte_unmap_unlock(start_pte, ptl); From dcb67f7425d51d12324b0de1ec676619499893a4 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:46 -0700 Subject: [PATCH 703/857] mm/damon/vaddr: respect folio end for DAMOS_MIGRATE_{HOT,COLD} For applying DAMOS_MIGRATE_{HOT,COLD} actions to a region, DAMON virtual address space operation set (vaddr) calls walk_page_range[_vma]() for the region. The pmd walk entry function, namely damon_va_migrate_pmd_entry(), collects folios of addresses of the region in the pmd. It starts from the walking address and advances the address by the size of the folio of the address until it goes out of the pmd or the region. Let's suppose it is for the first pmd of the region, and the region start address is in the middle of a large folio. Also, the next folios are small. Then, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that DAMOS action is applied to less than expected amount of memory. Given the best effort nature of DAMON, it is no big problem, but it is clearly a bug that is better to be fixed. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-6-sj@kernel.org Link: https://lore.kernel.org/20260514015053.149396-1-sj@kernel.org [1] Fixes: 09efc56a3b1c ("mm/damon/vaddr: consistently use only pmd_entry for damos_migrate") Signed-off-by: SJ Park Cc: Usama Arif Cc: # 6.19.x Signed-off-by: Andrew Morton --- mm/damon/vaddr.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 4b0b5edf679527..c8c32b2ae0402e 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -669,6 +669,8 @@ static int damos_va_migrate_pmd_entry(pmd_t *pmd, unsigned long addr, return 0; for (; addr < next; pte += nr, addr += nr * PAGE_SIZE) { + unsigned long page_idx; + nr = 1; ptent = ptep_get(pte); @@ -681,7 +683,8 @@ static int damos_va_migrate_pmd_entry(pmd_t *pmd, unsigned long addr, continue; damos_va_migrate_dests_add(folio, walk->vma, addr, dests, migration_lists); - nr = folio_nr_pages(folio); + page_idx = folio_page_idx(folio, pte_page(ptent)); + nr = folio_nr_pages(folio) - page_idx; } pte_unmap_unlock(start_pte, ptl); return 0; From 8255b853639681d452ed4185fa64a0f1d0884db9 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:47 -0700 Subject: [PATCH 704/857] mm/damon/core: handle extreme memory state in damon_get_node_mem_bp() In an extreme and unlikely situation, si_meminfo_node() might let the caller show zero total ram. That could cause a divide by zero in damon_get_node_mem_bp(). It could also show free memory larger than the total memory. This could cause underflow and make DAMOS temporarily make unexpected behavior. Thanks to safety guards in the auto-tuning feedback loop, that should not be a real problem, though. Fix the problems by respectively returning 100% and 0% for used and free memory queries in the corner cases. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-7-sj@kernel.org Link: https://lore.kernel.org/20260328133216.9697-1-sj@kernel.org [1] Fixes: 0e1c773b501f ("mm/damon/core: introduce damos quota goal metrics for memory node utilization") Signed-off-by: SJ Park Cc: Usama Arif Cc: # 6.16.x Signed-off-by: Andrew Morton --- mm/damon/core.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index a6fdb3068262da..ad3657d356fbca 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2807,6 +2807,13 @@ static __kernel_ulong_t damos_get_node_mem_bp( } si_meminfo_node(&i, goal->nid); + if (!i.totalram || i.totalram < i.freeram) { + if (goal->metric == DAMOS_QUOTA_NODE_MEM_USED_BP) + return 10000; + else /* DAMOS_QUOTA_NODE_MEM_FREE_BP */ + return 0; + } + if (goal->metric == DAMOS_QUOTA_NODE_MEM_USED_BP) numerator = i.totalram - i.freeram; else /* DAMOS_QUOTA_NODE_MEM_FREE_BP */ From f732d8c75b57020da6a6a561bdd49fb5eb46fbbc Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:48 -0700 Subject: [PATCH 705/857] mm/damon/core: handle extreme memory state in get_node_memcg_used_bp() In extreme unlikely situations, total memory might be zero. In less extreme but still very unlikely situations, lruvec_page_state() calls might let the caller show used memory larger than total memory. In the two cases, damos_get_node_memcg_used_bp() could cause division by zero, or return underflowed value, respectively. Handle the cases by respectively returning 100% and 0% for used and free memory queries in the corner cases. This issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-8-sj@kernel.org Link: https://lore.kernel.org/20260329154813.47382-1-sj@kernel.org [1] Fixes: b74a120bcf50 ("mm/damon/core: implement DAMOS_QUOTA_NODE_MEMCG_USED_BP") Signed-off-by: SJ Park Cc: Usama Arif Cc: # 6.19.x Signed-off-by: Andrew Morton --- mm/damon/core.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index ad3657d356fbca..3060edf5e4fa6c 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2854,6 +2854,12 @@ static unsigned long damos_get_node_memcg_used_bp( mem_cgroup_put(memcg); si_meminfo_node(&i, goal->nid); + if (!i.totalram || i.totalram < used_pages) { + if (goal->metric == DAMOS_QUOTA_NODE_MEMCG_USED_BP) + return 10000; + else /* DAMOS_QUOTA_NODE_MEMCG_FREE_BP */ + return 0; + } if (goal->metric == DAMOS_QUOTA_NODE_MEMCG_USED_BP) numerator = used_pages; else /* DAMOS_QUOTA_NODE_MEMCG_FREE_BP */ From 17d8da471c43715b94ee89f009d0fd11de99a1c7 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:49 -0700 Subject: [PATCH 706/857] mm/damon/core: handle extreme memory state in get_in_active_mem_bp() damos_get_in_active_mem_bp() uses the sum of the active and inactive memory amount as a denominator. In an extreme and unlikely environment, active and inactive memory might be zero. In this case, hence, it results in a divide by zero problem. Avoid it by changing the denominator to one if it is zero, before it is being used. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-9-sj@kernel.org Link: https://lore.kernel.org/20260721034756.147011-1-sj@kernel.org [1] Fixes: 4835e2871321 ("mm/damon/core: introduce [in]active memory ratio damos quota goal metric") Signed-off-by: SJ Park Cc: Usama Arif Cc: # 7.0.x Signed-off-by: Andrew Morton --- mm/damon/core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 3060edf5e4fa6c..0df785e72438f4 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3006,7 +3006,7 @@ static unsigned int damos_get_in_active_mem_bp(bool active_ratio) global_node_page_state(NR_LRU_BASE + LRU_ACTIVE_FILE); inactive = global_node_page_state(NR_LRU_BASE + LRU_INACTIVE_ANON) + global_node_page_state(NR_LRU_BASE + LRU_INACTIVE_FILE); - total = active + inactive; + total = max(active + inactive, 1); if (active_ratio) return mult_frac(active, 10000, total); return mult_frac(inactive, 10000, total); From 8637ddd35f9343fbd42d34238dc2a95ea457ab20 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:48 -0700 Subject: [PATCH 707/857] mm/damon/core: introduce DAMON_FILTER_TYPE_PGIDLE_UNSET Patch series "mm/damon: introduce data access-as-a-data attribute", v1.1. TL;DR: extend DAMON's data attributes monitoring system to support page table accessed bit and PG_idle based access monitoring. DAMON was initially introduced as a data access monitor. Users found data access pattern becomes more useful when it is combined with other data attributes such as belonging cgroups and backing page types. For such cases, DAMON has extended to support such data attributes monitoring in addition to the original data access monitoring. The DAMON probes system was introduced for this purpose. Users set probes for filtering data attributes of their interest. For use cases where the primary interests are the attributes but the access pattern, probe weights system has been introduced. When it is used, DAMON applies its adaptive regions adjustment based on the monitored data attributes. However, DAMON stops access monitoring when the probe weights are used. DAMON cannot optimally help users who have interests in both data attributes and access patterns. Data access can also be thought of as another data attribute, though. Extend the probe system to support data access as a data attribute. Introduce a new probe filter type, pgidle_unset. It shows if the page is not set as idle. Specifically, it shows the page table accessed bit and the PG_idle flag. It can inform if the region is ever accessed. But it cannot say when it is accessed. To answer the second question, introduce a new probe feature, prep actions. Using the features, Users can specify what preparation actions should be made to each region for each probe. DAMON executes the preparation actions for each sampling interval, like it is doing the preparation for access check in the access monitoring mode. To help 'pgidle_unset' probe action use case, 'set_pgidle' preparation action is introduced together. The action does exactly what the access monitoring was doing: clearing the page table accessed bits and setting the PG_idle flags. Tests ===== I compared the access pattern monitoring results from the classic way and the probe based way. As expected, the probe based way shows the results similar to that of the classic way. More detailed test methods and results are below. First, do the access monitoring using the DAMON user-space tool [1] in the classic way. The system is idle. It shows no access as expected. $ sudo ./damo/damo start $ sudo ./damo/damo report access heatmap: 00000000000000000000000000000000000000008999999811111110000000000000000000000000 # min/max temperatures: -640,000,000, -100,000,000, column size: 99.800 MiB intervals: sample 5 ms aggr 100 ms (max access hz 200) 0 addr 4.000 KiB size 3.898 GiB access 0 hz age 6.400 s 1 addr 3.898 GiB size 787.301 MiB access 0 hz age 1 s 2 addr 4.667 GiB size 773.457 MiB access 0 hz age 5.800 s 3 addr 5.423 GiB size 2.374 GiB access 0 hz age 6.400 s memory bw estimate: 0 B per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms Start an artificial memory access generator (masim) [2] in another window. $ ./masim/masim.py run --config_file ./masim/configs/zigzag.cfg Show the monitoring results. As expected, it captures accesses. $ sudo ./damo/damo report access heatmap: 00000000000000000000000000000000000000011111111111111118888888833378988888888888 # min/max temperatures: -1,470,000,000, -12,482,536, column size: 99.800 MiB intervals: sample 5 ms aggr 100 ms (max access hz 200) 0 addr 4.000 KiB size 1.542 GiB access 0 hz age 14.700 s 1 addr 1.542 GiB size 788.324 MiB access 0 hz age 14.600 s 2 addr 2.311 GiB size 772.844 MiB access 0 hz age 14.200 s [...] 50 addr 6.839 GiB size 1.742 MiB access 0 hz age 1.300 s 51 addr 6.840 GiB size 8.000 KiB access 100 hz age 0 ns 52 addr 6.840 GiB size 1.496 MiB access 160 hz age 0 ns [...] 97 addr 7.162 GiB size 1.199 MiB access 40 hz age 400 ms 98 addr 7.163 GiB size 1.199 MiB access 20 hz age 400 ms 99 addr 7.164 GiB size 2.004 MiB access 160 hz age 0 ns 100 addr 7.166 GiB size 646.004 MiB access 0 hz age 400 ms memory bw estimate: 26.148 GiB per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms After the artificial memory access generator (masim) is terminated, restart DAMON with the probe-based access monitoring. As expected, it shows no access since the system is idle again. $ sudo ./damo/damo stop $ sudo ./damo/damo start --probe_prep set_pgidle \ --probe_filter allow pgidle_unset --probe_weight 1 $ sudo ./damo/damo report attrs heatmap: 00000000000000000000000000000000000000008999999711111100000000000000000000000000 # min/max temperatures: -600,000,000, -430,000,000, column size: 99.800 MiB probe prep: set_pgidle, filter: allow pgidle_unset (weight: 1) intervals: sample 5 ms aggr 100 ms (max probe hits 20) # addr size age probe_hits 0 4.000 KiB 3.898 GiB 6 s 0 1 5.285 GiB 2.512 GiB 6 s 0 2 4.659 GiB 641.816 MiB 5.700 s 0 3 3.898 GiB 778.375 MiB 4.300 s 0 memory bw estimate: 0 B per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms Start the artificial memory access generator [2] again. $ ./masim/masim.py run --config_file ./masim/configs/zigzag.cfg Show the monitoring results. As expected, it captures accesses similar to the classic monitoring mode. $ sudo ./damo/damo report attrs heatmap: 00000000000000000000000000000000000000011111110000000177777777777777878798887777 # min/max temperatures: -1,330,000,000, 84,847,514, column size: 99.800 MiB probe prep: set_pgidle, filter: allow pgidle_unset (weight: 1) intervals: sample 5 ms aggr 100 ms (max probe hits 20) # addr size age probe_hits 0 4.000 KiB 1.508 GiB 13.300 s 0 1 1.508 GiB 794.684 MiB 13.200 s 0 2 2.284 GiB 764.555 MiB 13 s 0 [...] 50 6.711 GiB 4.625 MiB 0 ns 17 51 6.716 GiB 2.504 MiB 0 ns 17 52 6.732 GiB 796.000 KiB 0 ns 17 [...] 90 7.162 GiB 512.000 KiB 2.200 s 2 91 7.061 GiB 700.000 KiB 2.300 s 1 92 6.935 GiB 316.000 KiB 2.300 s 4 memory bw estimate: 0 B per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms Patches Sequence ================ First four patches (patches 1-4) introduce the new probe filter type for knowing if a region is accessed. Patch 1 defines the new type in the core. Patch 2 implements the execution of the new filter in the physical address space DAMON operation set. Patch 3 implements a user interface on DAMON sysfs interface. Patch 4 updates the documentation for the new filter type. Following 13 patches (patches 5-17) introduce the probe preparation actions feature. Patch 5 defines the data structure for specifying the preparation actions. Patch 6 completes setup of the API parameter for the prep. Patch 7 extends the DAMON operation set callback list to connect the parameter with the underlying operation set. Patch 8 implements the execution of the prep in the physical address space DAMON operation set. Following five patches (patches 9-13) extends DAMON sysfs interface for the new prep feature. Patch 14 adds simple selftest for basic file operations of the new sysfs files. Final three patches (patches 15-17) respectively update design, usage and ABI documents for the new feature and its interface. This patch (of 17): Introduce a new DAMON filter type, pgidle_unset. It will match pages that have their PG_Idle flag unset, or the page table accessed bit set. In other words, it says if the page is accessed. Link: https://lore.kernel.org/20260901132506.99243-1-sj@kernel.org Link: https://lore.kernel.org/20260901132506.99243-2-sj@kernel.org Link: https://github.com/damonitor/damo [1] Link: https://github.com/sjp38/masim [2] Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/damon.h | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 7b1b6050a8286f..7a42cbe791845d 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -745,12 +745,14 @@ struct damon_intervals_goal { /** * enum damon_filter_type - Type of &struct damon_filter * - * @DAMON_FILTER_TYPE_ANON: Anonymous pages. - * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. + * @DAMON_FILTER_TYPE_ANON: Anonymous pages. + * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. + * @DAMON_FILTER_TYPE_PGIDLE_UNSET: Pgidle is unset. */ enum damon_filter_type { DAMON_FILTER_TYPE_ANON, DAMON_FILTER_TYPE_MEMCG, + DAMON_FILTER_TYPE_PGIDLE_UNSET, }; /** From 13a4ce0edcb6b0681f18334bd01f50ca2ee27842 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:49 -0700 Subject: [PATCH 708/857] mm/damon/paddr: support PGIDLE_UNSET probe filter type Implement support of DAMON_FILTER_TYPE_PGIDLE_UNSET in the physical address space DAMON operations set. It reuses damon_folio_young(), which was being used for access monitoring. Link: https://lore.kernel.org/20260901132506.99243-3-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/paddr.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 9ddd1ec8202b7f..6f756f84938948 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -132,6 +132,12 @@ static bool damon_pa_filter_match(struct damon_filter *filter, matched = filter->memcg_id == mem_cgroup_id(memcg); rcu_read_unlock(); break; + case DAMON_FILTER_TYPE_PGIDLE_UNSET: + if (!folio) + matched = false; + else + matched = damon_folio_young(folio); + break; default: break; } From 497090f1b55e5cea32ec86760139ad4189034485 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:50 -0700 Subject: [PATCH 709/857] mm/damon/sysfs: support pgidle_unset probe filter type Extend DAMON sysfs interface to allow users to set DAMON_FILTER_TYPE_PGIDLE_UNSET by writing 'pgidle_unset' to the probe filter type file. Link: https://lore.kernel.org/20260901132506.99243-4-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/sysfs.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 3c81b4c91ac0dd..c1ff739dab27a8 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -781,6 +781,10 @@ damon_sysfs_filter_type_names[] = { .type = DAMON_FILTER_TYPE_MEMCG, .name = "memcg", }, + { + .type = DAMON_FILTER_TYPE_PGIDLE_UNSET, + .name = "pgidle_unset", + }, }; static ssize_t type_show(struct kobject *kobj, From 2f31b32bfddb07a3f7e2a6a7087a1f5ea53278b8 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:51 -0700 Subject: [PATCH 710/857] Docs/mm/damon/design: document pgidle_unset probe filter type Update DAMON design document for the newly added pgidle_unset probe filter type. Also use a list for the types, as it becomes not very easy to read the whole types in a simple sentence. Link: https://lore.kernel.org/20260901132506.99243-5-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/mm/damon/design.rst | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 63cbb7b536da20..947d91ae24a392 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -293,8 +293,12 @@ registration is made by specifying a probe per attribute. Each of the probe specifies a rule to determine if a given memory region has the related attribute. The rule is constructed with multiple filters. The filters work same to :ref:`DAMOS filters ` except the supported -filter types. Currently only ``anon`` and ``memcg`` filter types are supported -for data attributes monitoring. +filter types. Currently below filter types are supported. + +- ``anon``: Same to that for DAMOS filters. +- ``memcg``: Same to that for DAMOS filters. +- ``pgidle_unset``: Matches if the page for the memory is marked as not + access-idle. If such probes are registered, DAMON executes the probes for each region's sampling memory when it does the access :ref:`sampling From add4f2614382895aee9d6b5cd155cc3172807e2a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:52 -0700 Subject: [PATCH 711/857] mm/damon/core: introduce damon_prep struct Some DAMON probe filter types require preparatory actions. For example, pgilde_unset probe filter can say if the page was accessed but when. To answer the second question, the PG_Idle flag should be set at a specific time. It can make life much easier if DAMON can do such preparatory actions. Introduce a new data type called damon_prep. It specifies each of the preparation actions for each probe. DAMON will execute the action for each region per sampling interval, like it clears page table accessed bits and unsets PG_Idle flag for access monitoring. Also introduce DAMON_PREP_SET_PGIDLE as the initial prep action. As the name says, it will do exactly what DAMON was doing as the preparation action for the access monitoring. Link: https://lore.kernel.org/20260901132506.99243-6-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/damon.h | 32 ++++++++++++++++++++++++++++++++ mm/damon/core.c | 26 ++++++++++++++++++++++++++ 2 files changed, 58 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 7a42cbe791845d..1780c14942e634 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -742,6 +742,27 @@ struct damon_intervals_goal { unsigned long max_sample_us; }; +/** + * enum damon_prep_action - DAMON probing preparation action. + * + * @DAMON_PREP_SET_PGIDLE: Set the probing memory as idle page. + */ +enum damon_prep_action { + DAMON_PREP_SET_PGIDLE, +}; + +/** + * struct damon_prep - DAMON probing preparation request. + * + * @action: Action to do to the probing memory for the preparation. + */ +struct damon_prep { + enum damon_prep_action action; +/* private: */ + /* siblings list. */ + struct list_head list; +}; + /** * enum damon_filter_type - Type of &struct damon_filter * @@ -783,6 +804,8 @@ struct damon_filter { struct damon_probe { unsigned int weight; /* private: */ + /* Preparation actions to apply to each probing memory. */ + struct list_head preps; /* Filters for assessing if a given region is for this probe. */ struct list_head filters; /* Siblings list. */ @@ -962,6 +985,12 @@ static inline unsigned long damon_sz_region(struct damon_region *r) return r->ar.end - r->ar.start; } +#define damon_for_each_prep(p, probe) \ + list_for_each_entry(p, &(probe)->preps, list) + +#define damon_for_each_prep_safe(p, next, probe) \ + list_for_each_entry_safe(p, next, &(probe)->preps, list) + #define damon_for_each_filter(f, p) \ list_for_each_entry(f, &(p)->filters, list) @@ -1015,6 +1044,9 @@ static inline unsigned long damon_sz_region(struct damon_region *r) #ifdef CONFIG_DAMON +struct damon_prep *damon_new_prep(enum damon_prep_action action); +void damon_add_prep(struct damon_probe *p, struct damon_prep *prep); + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow); void damon_add_filter(struct damon_probe *probe, struct damon_filter *f); diff --git a/mm/damon/core.c b/mm/damon/core.c index 0df785e72438f4..c74ddf84aa5520 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -112,6 +112,28 @@ int damon_select_ops(struct damon_ctx *ctx, enum damon_ops_id id) return err; } +struct damon_prep *damon_new_prep(enum damon_prep_action action) +{ + struct damon_prep *prep; + + prep = kmalloc_obj(*prep); + if (!prep) + return NULL; + prep->action = action; + INIT_LIST_HEAD(&prep->list); + return prep; +} + +void damon_add_prep(struct damon_probe *p, struct damon_prep *prep) +{ + list_add_tail(&prep->list, &p->preps); +} + +static void damon_free_prep(struct damon_prep *p) +{ + kfree(p); +} + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow) { @@ -168,6 +190,7 @@ struct damon_probe *damon_new_probe(void) if (!p) return NULL; p->weight = 0; + INIT_LIST_HEAD(&p->preps); INIT_LIST_HEAD(&p->filters); INIT_LIST_HEAD(&p->list); return p; @@ -185,8 +208,11 @@ static void damon_del_probe(struct damon_probe *p) static void damon_free_probe(struct damon_probe *p) { + struct damon_prep *prep, *prep_next; struct damon_filter *f, *next; + damon_for_each_prep_safe(prep, prep_next, p) + damon_free_prep(prep); damon_for_each_filter_safe(f, next, p) damon_free_filter(f); kfree(p); From c8905c79ea481f829addaf4f092e277ac174dcc1 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:53 -0700 Subject: [PATCH 712/857] mm/damon/core: commit preps damon_commit_probes() is ignoring damon_prep. Commit the prep actions, too. Link: https://lore.kernel.org/20260901132506.99243-7-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/core.c | 59 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 59 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index c74ddf84aa5520..b69fa34cf5f162 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -129,11 +129,34 @@ void damon_add_prep(struct damon_probe *p, struct damon_prep *prep) list_add_tail(&prep->list, &p->preps); } +static void damon_del_prep(struct damon_prep *p) +{ + list_del(&p->list); +} + static void damon_free_prep(struct damon_prep *p) { kfree(p); } +static void damon_destroy_prep(struct damon_prep *p) +{ + damon_del_prep(p); + damon_free_prep(p); +} + +static struct damon_prep *damon_nth_prep(int n, struct damon_probe *p) +{ + struct damon_prep *prep; + int i = 0; + + damon_for_each_prep(prep, p) { + if (i++ == n) + return prep; + } + return NULL; +} + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow) { @@ -1699,6 +1722,36 @@ static int damon_commit_targets( return err; } +static void damon_commit_prep(struct damon_prep *dst, struct damon_prep *src) +{ + dst->action = src->action; +} + +static int damon_commit_preps(struct damon_probe *dst, struct damon_probe *src) +{ + struct damon_prep *dst_prep, *next, *src_prep, *new_prep; + int i = 0, j = 0; + + damon_for_each_prep_safe(dst_prep, next, dst) { + src_prep = damon_nth_prep(i++, src); + if (src_prep) + damon_commit_prep(dst_prep, src_prep); + else + damon_destroy_prep(dst_prep); + } + + damon_for_each_prep_safe(src_prep, next, src) { + if (j++ < i) + continue; + + new_prep = damon_new_prep(src_prep->action); + if (!new_prep) + return -ENOMEM; + damon_add_prep(dst, new_prep); + } + return 0; +} + static void damon_commit_filter(struct damon_filter *dst, struct damon_filter *src) { @@ -1757,6 +1810,9 @@ static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src) src_probe = damon_nth_probe(i++, src); if (src_probe) { dst_probe->weight = src_probe->weight; + err = damon_commit_preps(dst_probe, src_probe); + if (err) + return err; err = damon_commit_filters(dst_probe, src_probe); if (err) return err; @@ -1774,6 +1830,9 @@ static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src) return -ENOMEM; damon_add_probe(dst, new_probe); new_probe->weight = src_probe->weight; + err = damon_commit_preps(new_probe, src_probe); + if (err) + return err; err = damon_commit_filters(new_probe, src_probe); if (err) return err; From 529a054755717750f7ae2f423aa6d3ed25259394 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:54 -0700 Subject: [PATCH 713/857] mm/damon/core: introduce damon_operations->prep_probes() damon_prep needs to be executed by the underlying DAMON operation set. Extend the operation set callback list for the execution of damon_prep actions. If the underlying operation set implements the callback, DAMON core executes it in the monitoring preparation time. Link: https://lore.kernel.org/20260901132506.99243-8-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/damon.h | 5 +++++ mm/damon/core.c | 20 +++++++++++++++++++- 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 1780c14942e634..871d26adf6ae57 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -630,6 +630,7 @@ enum damon_ops_id { * @update: Update operations-related data structures. * @prepare_access_checks: Prepare next access check of target regions. * @check_accesses: Check the accesses to target regions. + * @prep_probes: Prepare applying probes for each region. * @apply_probes: Apply probes for each region. * @get_scheme_score: Get the score of a region for a scheme. * @apply_scheme: Apply a DAMON-based operation scheme. @@ -657,6 +658,9 @@ enum damon_ops_id { * last preparation and update the number of observed accesses of each region. * It should also return max number of observed accesses that made as a result * of its update. The value will be used for regions adjustment threshold. + * @prep_probes should execute required &struct damon_prep for next &struct + * damon_probe applications to each region. It should also set + * &damon_region->sampling_addr of each region if ``set_samples`` is true. * @apply_probes should apply the data attribute probes to each region and * accordingly update the probe hits counter of the region. It should also * set &damon_region->sampling_addr of each region if ``set_samples`` is true. @@ -679,6 +683,7 @@ struct damon_operations { void (*update)(struct damon_ctx *context); void (*prepare_access_checks)(struct damon_ctx *context); unsigned int (*check_accesses)(struct damon_ctx *context); + void (*prep_probes)(struct damon_ctx *context, bool set_samples); unsigned int (*apply_probes)(struct damon_ctx *context, bool set_samples, bool return_max_wsum); int (*get_scheme_score)(struct damon_ctx *context, diff --git a/mm/damon/core.c b/mm/damon/core.c index b69fa34cf5f162..d8073e735bbecb 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -157,6 +157,18 @@ static struct damon_prep *damon_nth_prep(int n, struct damon_probe *p) return NULL; } +static bool damon_has_prep(struct damon_ctx *c) +{ + struct damon_prep *prep; + struct damon_probe *probe; + + damon_for_each_probe(probe, c) { + damon_for_each_prep(prep, probe) + return true; + } + return false; +} + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow) { @@ -3894,14 +3906,19 @@ static int kdamond_fn(void *data) unsigned long next_ops_update_sis = ctx->next_ops_update_sis; unsigned long sample_interval = ctx->attrs.sample_interval; bool access_check_disabled = damon_has_probe_weights(ctx); + bool do_prep; unsigned int max_merge_score = 0, max_wsum; bool get_max_wsum; if (kdamond_wait_activation(ctx)) break; + do_prep = ctx->ops.prep_probes && damon_has_prep(ctx); + if (!access_check_disabled && ctx->ops.prepare_access_checks) ctx->ops.prepare_access_checks(ctx); + if (do_prep) + ctx->ops.prep_probes(ctx, access_check_disabled); kdamond_usleep(sample_interval); ctx->passed_sample_intervals++; @@ -3916,7 +3933,8 @@ static int kdamond_fn(void *data) else get_max_wsum = false; max_wsum = ctx->ops.apply_probes(ctx, - access_check_disabled, get_max_wsum); + access_check_disabled && !do_prep, + get_max_wsum); if (get_max_wsum) max_merge_score = max_wsum; } From 2176209b61c58fef51fe85d0b20bebf8abbae4be Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:55 -0700 Subject: [PATCH 714/857] mm/damon/paddr: support damon_prep Implement damon_operations->prep_probes() callback. Support the only existing prep action, DAMON_PREP_SET_PGIDLE in a way similar to what it was doing for the access check preparation: unset page table accessed bits and set PG_Idle flag. Reuse the function for the access check preparation. Link: https://lore.kernel.org/20260901132506.99243-9-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/paddr.c | 35 +++++++++++++++++++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 6f756f84938948..c1e7d7a4f40df3 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -105,6 +105,40 @@ static unsigned int damon_pa_check_accesses(struct damon_ctx *ctx) return max_nr_accesses; } +static void damon_pa_prep_probes_region(struct damon_region *r, + struct damon_probe *probe, struct damon_ctx *ctx) +{ + struct damon_prep *p; + + damon_for_each_prep(p, probe) { + switch (p->action) { + case DAMON_PREP_SET_PGIDLE: + damon_pa_mkold(damon_pa_phys_addr(r->sampling_addr, + ctx->addr_unit)); + break; + default: + break; + } + } +} + +static void damon_pa_prep_probes(struct damon_ctx *ctx, bool set_samples) +{ + struct damon_target *t; + struct damon_region *r; + struct damon_probe *p; + + damon_for_each_target(t, ctx) { + damon_for_each_region(r, t) { + if (set_samples) + r->sampling_addr = damon_rand(ctx, r->ar.start, + r->ar.end); + damon_for_each_probe(p, ctx) + damon_pa_prep_probes_region(r, p, ctx); + } + } +} + static bool damon_pa_filter_match(struct damon_filter *filter, struct folio *folio) { @@ -448,6 +482,7 @@ static int __init damon_pa_initcall(void) .update = NULL, .prepare_access_checks = damon_pa_prepare_access_checks, .check_accesses = damon_pa_check_accesses, + .prep_probes = damon_pa_prep_probes, .apply_probes = damon_pa_apply_probes, .target_valid = NULL, .apply_scheme = damon_pa_apply_scheme, From 859eccd30a40b313c6e91672910f249f507c72a5 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:56 -0700 Subject: [PATCH 715/857] mm/damon/sysfs: implement preps directory Implement a sysfs directory named 'preps' under the probe directory. It will be evolved to be used for specifying probe preps. Link: https://lore.kernel.org/20260901132506.99243-10-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/sysfs.c | 65 +++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 59 insertions(+), 6 deletions(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index c1ff739dab27a8..35fb308039b612 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -749,6 +749,35 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { .default_groups = damon_sysfs_intervals_groups, }; +/* + * preps directory + */ + +struct damon_sysfs_preps { + struct kobject kobj; +}; + +static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) +{ + return kzalloc_obj(struct damon_sysfs_preps); +} + +static void damon_sysfs_preps_release(struct kobject *kobj) +{ + kfree(container_of(kobj, struct damon_sysfs_preps, kobj)); +} + +static struct attribute *damon_sysfs_preps_attrs[] = { + NULL, +}; +ATTRIBUTE_GROUPS(damon_sysfs_preps); + +static const struct kobj_type damon_sysfs_preps_ktype = { + .release = damon_sysfs_preps_release, + .sysfs_ops = &kobj_sysfs_ops, + .default_groups = damon_sysfs_preps_groups, +}; + /* * filter directory */ @@ -1069,6 +1098,7 @@ static const struct kobj_type damon_sysfs_filters_ktype = { struct damon_sysfs_probe { struct kobject kobj; unsigned int weight; + struct damon_sysfs_preps *preps; struct damon_sysfs_filters *filters; }; @@ -1079,25 +1109,48 @@ static struct damon_sysfs_probe *damon_sysfs_probe_alloc(void) static int damon_sysfs_probe_add_dirs(struct damon_sysfs_probe *probe) { + struct damon_sysfs_preps *preps; struct damon_sysfs_filters *filters; int err; - filters = damon_sysfs_filters_alloc(); - if (!filters) + preps = damon_sysfs_preps_alloc(); + if (!preps) return -ENOMEM; + probe->preps = preps; + + err = kobject_init_and_add(&preps->kobj, &damon_sysfs_preps_ktype, + &probe->kobj, "preps"); + if (err) + goto put_preps_out; + + filters = damon_sysfs_filters_alloc(); + if (!filters) { + err = -ENOMEM; + goto del_preps_out; + } probe->filters = filters; err = kobject_init_and_add(&filters->kobj, &damon_sysfs_filters_ktype, &probe->kobj, "filters"); - if (err) { - kobject_put(&filters->kobj); - probe->filters = NULL; - } + if (err) + goto put_filters_out; + return err; + +put_filters_out: + kobject_put(&filters->kobj); + probe->filters = NULL; +del_preps_out: + kobject_del(&preps->kobj); +put_preps_out: + kobject_put(&preps->kobj); + probe->preps = NULL; return err; } static void damon_sysfs_probe_rm_dirs(struct damon_sysfs_probe *probe) { + if (probe->preps) + kobject_put(&probe->preps->kobj); if (probe->filters) { damon_sysfs_filters_rm_dirs(probe->filters); kobject_put(&probe->filters->kobj); From d2ef98d62a1ebd8c9602e7b65d7311536bc75940 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:57 -0700 Subject: [PATCH 716/857] mm/damon/sysfs: implement preps/nr_preps file Implement nr_preps file under the preps directory. It will be evolved to be used for generating sub directories that will represent each probe preparation action. Link: https://lore.kernel.org/20260901132506.99243-11-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/sysfs.c | 53 +++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 52 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 35fb308039b612..b9722ebffd6f17 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -755,6 +755,7 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { struct damon_sysfs_preps { struct kobject kobj; + int nr; }; static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) @@ -762,12 +763,60 @@ static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) return kzalloc_obj(struct damon_sysfs_preps); } +static void damon_sysfs_preps_rm_dirs(struct damon_sysfs_preps *preps) +{ + preps->nr = 0; +} + +static int damon_sysfs_preps_add_dirs(struct damon_sysfs_preps *preps, + int nr_preps) +{ + preps->nr = nr_preps; + return 0; +} + +static ssize_t nr_preps_show(struct kobject *kobj, struct kobj_attribute *attr, + char *buf) +{ + struct damon_sysfs_preps *preps = container_of(kobj, + struct damon_sysfs_preps, kobj); + + return sysfs_emit(buf, "%d\n", preps->nr); +} + +static ssize_t nr_preps_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_preps *preps; + int nr, err = kstrtoint(buf, 0, &nr); + + if (err) + return err; + if (nr < 0) + return -EINVAL; + + preps = container_of(kobj, struct damon_sysfs_preps, kobj); + + if (!mutex_trylock(&damon_sysfs_lock)) + return -EBUSY; + err = damon_sysfs_preps_add_dirs(preps, nr); + mutex_unlock(&damon_sysfs_lock); + if (err) + return err; + + return count; +} + static void damon_sysfs_preps_release(struct kobject *kobj) { kfree(container_of(kobj, struct damon_sysfs_preps, kobj)); } +static struct kobj_attribute damon_sysfs_preps_nr_attr = + __ATTR_RW_MODE(nr_preps, 0600); + static struct attribute *damon_sysfs_preps_attrs[] = { + &damon_sysfs_preps_nr_attr.attr, NULL, }; ATTRIBUTE_GROUPS(damon_sysfs_preps); @@ -1149,8 +1198,10 @@ static int damon_sysfs_probe_add_dirs(struct damon_sysfs_probe *probe) static void damon_sysfs_probe_rm_dirs(struct damon_sysfs_probe *probe) { - if (probe->preps) + if (probe->preps) { + damon_sysfs_preps_rm_dirs(probe->preps); kobject_put(&probe->preps->kobj); + } if (probe->filters) { damon_sysfs_filters_rm_dirs(probe->filters); kobject_put(&probe->filters->kobj); From 0b5b1fd1c5a0f9ac8bd3cfa394328db7a8ff0396 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:58 -0700 Subject: [PATCH 717/857] mm/damon/sysfs: create directories for nr_preps writes Implement nr_preps write action to actually create subdirectories of the number. Each of the directory will be evolved to represent each probe preparation action. Link: https://lore.kernel.org/20260901132506.99243-12-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/sysfs.c | 80 +++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 79 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index b9722ebffd6f17..4ff473b6895355 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -749,12 +749,50 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { .default_groups = damon_sysfs_intervals_groups, }; +/* + * prep directory + */ + +struct damon_sysfs_prep { + struct kobject kobj; +}; + +static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void) +{ + struct damon_sysfs_prep *prep; + + prep = kzalloc_obj(struct damon_sysfs_prep); + if (!prep) + return prep; + return prep; +} + +static void damon_sysfs_prep_release(struct kobject *kobj) +{ + struct damon_sysfs_prep *prep = container_of(kobj, + struct damon_sysfs_prep, kobj); + + kfree(prep); +} + +static struct attribute *damon_sysfs_prep_attrs[] = { + NULL, +}; +ATTRIBUTE_GROUPS(damon_sysfs_prep); + +static const struct kobj_type damon_sysfs_prep_ktype = { + .release = damon_sysfs_prep_release, + .sysfs_ops = &kobj_sysfs_ops, + .default_groups = damon_sysfs_prep_groups, +}; + /* * preps directory */ struct damon_sysfs_preps { struct kobject kobj; + struct damon_sysfs_prep **preps_arr; int nr; }; @@ -765,13 +803,53 @@ static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) static void damon_sysfs_preps_rm_dirs(struct damon_sysfs_preps *preps) { + struct damon_sysfs_prep **preps_arr = preps->preps_arr; + int i; + + for (i = 0; i < preps->nr; i++) { + kobject_del(&preps_arr[i]->kobj); + kobject_put(&preps_arr[i]->kobj); + } preps->nr = 0; + kfree(preps_arr); + preps->preps_arr = NULL; } static int damon_sysfs_preps_add_dirs(struct damon_sysfs_preps *preps, int nr_preps) { - preps->nr = nr_preps; + struct damon_sysfs_prep **preps_arr, *prep; + int err, i; + + damon_sysfs_preps_rm_dirs(preps); + if (!nr_preps) + return 0; + + preps_arr = kmalloc_objs(*preps_arr, nr_preps, + GFP_KERNEL | __GFP_NOWARN); + if (!preps_arr) + return -ENOMEM; + preps->preps_arr = preps_arr; + + for (i = 0; i < nr_preps; i++) { + prep = damon_sysfs_prep_alloc(); + if (!prep) { + damon_sysfs_preps_rm_dirs(preps); + return -ENOMEM; + } + + err = kobject_init_and_add(&prep->kobj, + &damon_sysfs_prep_ktype, &preps->kobj, "%d", + i); + if (err) { + kobject_put(&prep->kobj); + damon_sysfs_preps_rm_dirs(preps); + return err; + } + + preps_arr[i] = prep; + preps->nr++; + } return 0; } From dfb3e4940c0c11da0d35524af9a0061b19497ba3 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:59 -0700 Subject: [PATCH 718/857] mm/damon/sysfs: implement prep_action file Add a file named prep_action under the prep directory. It represents the corresponding probe preparation action. Link: https://lore.kernel.org/20260901132506.99243-13-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/sysfs.c | 57 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 57 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 4ff473b6895355..2118089dd9d8dd 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -755,6 +755,7 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { struct damon_sysfs_prep { struct kobject kobj; + enum damon_prep_action action; }; static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void) @@ -764,9 +765,61 @@ static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void) prep = kzalloc_obj(struct damon_sysfs_prep); if (!prep) return prep; + prep->action = DAMON_PREP_SET_PGIDLE; return prep; } +struct damon_sysfs_prep_action_name { + const enum damon_prep_action action; + const char *name; +}; + +static const struct damon_sysfs_prep_action_name +damon_sysfs_prep_action_names[] = { + { + .action = DAMON_PREP_SET_PGIDLE, + .name = "set_pgidle", + }, +}; + +static ssize_t prep_action_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + struct damon_sysfs_prep *prep = container_of(kobj, + struct damon_sysfs_prep, kobj); + int i; + + for (i = 0; i < ARRAY_SIZE(damon_sysfs_prep_action_names); i++) { + const struct damon_sysfs_prep_action_name *action_name; + + action_name = &damon_sysfs_prep_action_names[i]; + if (action_name->action == prep->action) + return sysfs_emit(buf, "%s\n", action_name->name); + } + return -EINVAL; +} + +static ssize_t prep_action_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_prep *prep = container_of(kobj, + struct damon_sysfs_prep, kobj); + ssize_t ret = -EINVAL; + int i; + + for (i = 0; i < ARRAY_SIZE(damon_sysfs_prep_action_names); i++) { + const struct damon_sysfs_prep_action_name *action_name; + + action_name = &damon_sysfs_prep_action_names[i]; + if (sysfs_streq(buf, action_name->name)) { + prep->action = action_name->action; + ret = count; + break; + } + } + return ret; +} + static void damon_sysfs_prep_release(struct kobject *kobj) { struct damon_sysfs_prep *prep = container_of(kobj, @@ -775,7 +828,11 @@ static void damon_sysfs_prep_release(struct kobject *kobj) kfree(prep); } +static struct kobj_attribute damon_sysfs_prep_prep_action_attr = + __ATTR_RW_MODE(prep_action, 0600); + static struct attribute *damon_sysfs_prep_attrs[] = { + &damon_sysfs_prep_prep_action_attr.attr, NULL, }; ATTRIBUTE_GROUPS(damon_sysfs_prep); From 66483461eb3cb884b83a6d5e38f20946c06ad988 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:00 -0700 Subject: [PATCH 719/857] mm/damon/sysfs: pass preps to DAMON core DAMON sysfs interface provides the files for setting DAMON probe preps. But the underlying code is not really passing the user-set values to DAMON core. Pass those. Link: https://lore.kernel.org/20260901132506.99243-14-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- mm/damon/sysfs.c | 28 ++++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 2118089dd9d8dd..7f340b6f1921b9 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -812,7 +812,10 @@ static ssize_t prep_action_store(struct kobject *kobj, action_name = &damon_sysfs_prep_action_names[i]; if (sysfs_streq(buf, action_name->name)) { + if (!mutex_trylock(&damon_sysfs_lock)) + return -EBUSY; prep->action = action_name->action; + mutex_unlock(&damon_sysfs_lock); ret = count; break; } @@ -2175,6 +2178,23 @@ static int damon_sysfs_set_attrs(struct damon_ctx *ctx, return damon_set_attrs(ctx, &attrs); } +static int damon_sysfs_set_preps(struct damon_probe *probe, + struct damon_sysfs_preps *sys_preps) +{ + int i; + + for (i = 0; i < sys_preps->nr; i++) { + struct damon_sysfs_prep *sys_prep = sys_preps->preps_arr[i]; + struct damon_prep *prep; + + prep = damon_new_prep(sys_prep->action); + if (!prep) + return -ENOMEM; + damon_add_prep(probe, prep); + } + return 0; +} + static int damon_sysfs_set_filters(struct damon_probe *probe, struct damon_sysfs_filters *sys_filters) { @@ -2210,7 +2230,15 @@ static int damon_sysfs_set_probe(struct damon_probe *probe, struct damon_sysfs_probe *sys_probe) { struct damon_sysfs_filters *sys_filters; + struct damon_sysfs_preps *sys_preps; + int err; + sys_preps = sys_probe->preps; + if (sys_preps) { + err = damon_sysfs_set_preps(probe, sys_preps); + if (err) + return err; + } sys_filters = sys_probe->filters; if (!sys_filters) return 0; From ec0d6f5e539b9789a2ce21ccbbd88f6d7817438e Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:01 -0700 Subject: [PATCH 720/857] selftests/damon/sysfs.sh: test probe prep sysfs files Add basic file operations test for newly introduced DAMON probe prep sysfs directories and files. Link: https://lore.kernel.org/20260901132506.99243-15-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- tools/testing/selftests/damon/sysfs.sh | 27 ++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs.sh b/tools/testing/selftests/damon/sysfs.sh index f7fb94b84e716d..ddebde6edabe4e 100755 --- a/tools/testing/selftests/damon/sysfs.sh +++ b/tools/testing/selftests/damon/sysfs.sh @@ -361,6 +361,32 @@ test_intervals() test_intervals_goal "$intervals_dir/intervals_goal" } +test_damon_prep() +{ + damon_prep_dir=$1 + ensure_file "$damon_prep_dir/prep_action" "exist" "600" + ensure_write_succ "$damon_prep_dir/prep_action" "set_pgidle" \ + "valid input" + ensure_write_fail "$damon_prep_dir/prep_action" "foo" "invalid input" +} + +test_damon_preps() +{ + preps_dir=$1 + ensure_dir "$preps_dir" "exist" + ensure_file "$preps_dir/nr_preps" "exist" "600" + ensure_write_succ "$preps_dir/nr_preps" "1" "valid input" + test_damon_prep "$preps_dir/0" + + ensure_write_succ "$preps_dir/nr_preps" "2" "valid input" + test_damon_prep "$preps_dir/0" + test_damon_prep "$preps_dir/1" + + ensure_write_succ "$preps_dir/nr_preps" "0" "valid input" + ensure_dir "$preps_dir/0" "not_exist" + ensure_dir "$preps_dir/1" "not_exist" +} + test_damon_filter() { damon_filter_dir=$1 @@ -392,6 +418,7 @@ test_probe() { probe_dir=$1 ensure_dir "$probe_dir" "exist" + test_damon_preps "$probe_dir/preps" test_damon_filters "$probe_dir/filters" } From 287911a86033df672b20ffa300d77a13db48dad0 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:02 -0700 Subject: [PATCH 721/857] Docs/mm/damon/design: document probe preps Update DAMON design document for the newly added DAMON probe preps feature. Link: https://lore.kernel.org/20260901132506.99243-16-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/mm/damon/design.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 947d91ae24a392..d036340dae8afb 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -309,6 +309,13 @@ Users can therefore know how much of a given DAMON region has a specific data attribute by reading the per-region per-probe probe hits counter after each aggregation interval. +Users can optionally register probing preparation actions per probe. If such +actions are registered, DAMON applies the actions to each region's sampling +memory before starting the next sampling interval. Currently only one action, +``set_pgidle`` is supported. The action marks the page for the probing target +memory as access-idle. This can be useful to be used together with +``pgidle_unset`` probe filter. + This is a sampling based mechanism. Hence, it is lightweight but the output may include some measurement errors. The output should be used with good understanding of statistics. From e294c11d788a8aa0bb8da0b5cab4e4fd0ff2af11 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:03 -0700 Subject: [PATCH 722/857] Docs/admin-guide/mm/damon/usage: document probe preps sysfs files Update DAMON usage document for the newly added DAMON probe preps sysfs files. Link: https://lore.kernel.org/20260901132506.99243-17-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/admin-guide/mm/damon/usage.rst | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index da5f9afd08aef8..023c6334024f8e 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -74,6 +74,9 @@ comma (","). │ │ │ │ │ │ nr_regions/min,max │ │ │ │ │ │ :ref:`probes `/nr_probes │ │ │ │ │ │ │ 0/weight + │ │ │ │ │ │ │ │ preps/nr_preps + │ │ │ │ │ │ │ │ │ 0/prep_action + │ │ │ │ │ │ │ │ │ ... │ │ │ │ │ │ │ │ filters/nr_filters │ │ │ │ │ │ │ │ │ 0/type,matching,allow,path │ │ │ │ │ │ │ │ │ ... @@ -283,9 +286,18 @@ In the beginning, this directory has only one file, ``nr_probes``. Writing a number (``N``) to the file creates the number of child directories named ``0`` to ``N-1``. Each directory represents each monitoring probe. -In each probe directory, one directory, ``filters`` exists. The directory -contains files for installing filters for the probe, that is used to determine -the data attribute for the probe. +In each probe directory, two directories, ``preps`` and ``filters`` exist. The +directories contain files for installing probing preparation actions and +filters for the probe, that are used to determine the data attribute for the +probe. + +In the beginning, ``preps`` directory has only one file, ``nr_preps``. +Writing a number (``N``) to the file creates the number of child directories +named ``0`` to ``N-1``. Each directory represents each preparation action. +Each directory has one file, ``prep_action``. The preparation action can be +selected by writing the name of the action to the ``prep_action`` file. Refer +to the :ref:`design doc ` for the list of +supported actions. Each probe directory also contains ``weight`` file. Reading from and writing to the file gets and sets the :ref:`attributes-only monitoring From 72bd4f158ba5d2b670aac493805d4f539383cc9a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:04 -0700 Subject: [PATCH 723/857] Docs/ABI/damon: document probe prep sysfs files Update DAMON ABI document for the newly added DAMON probe prep sysfs files. Link: https://lore.kernel.org/20260901132506.99243-18-sj@kernel.org Signed-off-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/ABI/testing/sysfs-kernel-mm-damon | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/Documentation/ABI/testing/sysfs-kernel-mm-damon b/Documentation/ABI/testing/sysfs-kernel-mm-damon index e675a57145e36d..f8d2601e829047 100644 --- a/Documentation/ABI/testing/sysfs-kernel-mm-damon +++ b/Documentation/ABI/testing/sysfs-kernel-mm-damon @@ -173,6 +173,19 @@ Contact: SJ Park Description: Writing to and reading from this file sets and gets the per-probe attribute weight. +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/preps/nr_preps +Date: Jun 2026 +Contact: SJ Park +Description: Writing a number 'N' to this file creates the number of + directories for each DAMON probing preparation action named '0' + to 'N-1' under the preps/ directory. + +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/preps//prep_action +Date: Jun 2026 +Contact: SJ Park +Description: Writing to and reading from this file sets and gets the probing + preparation action. + What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/filters/nr_filters Date: May 2026 Contact: SJ Park From a3208afc56177e805aca32630cec874631da11c1 Mon Sep 17 00:00:00 2001 From: Qinyun Tan Date: Tue, 1 Sep 2026 19:51:04 +0800 Subject: [PATCH 724/857] mm/list_lru: don't copy stale shrinker id from non-memcg-aware shrinkers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit With cgroup.memory=nokmem, shrinker_memcg_alloc() fails with -ENOSYS for shrinkers without SHRINKER_NONSLAB, and shrinker_alloc() falls back to a non-memcg-aware shrinker. On this fallback path, shrinker->id is never assigned and keeps 0 from kzalloc(), which is a valid id belonging to whichever memcg-aware shrinker registers first. __list_lru_init() copies shrinker->id unconditionally, so every list_lru backed by such a fallback shrinker (thp-deferred_split, zswap-shrinker, workingset shadow nodes, superblock lrus, ...) ends up with lru->shrinker_id == 0 instead of -1. Under nokmem the list_lru collapses to the shared per-node lists, but __list_lru_add() still calls set_shrinker_bit() against the memcg of the added object. Most list_lru users are unaffected because their objects resolve to a NULL memcg without kmem accounting, but the THP deferred split queue holds user folios, which are charged regardless of nokmem. Since no memcg-aware shrinker can register under nokmem, shrinker_nr_max stays 0 and every memcg's shrinker_info has map_nr_max == 0, so the first folio added by khugepaged triggers on every boot: WARNING: mm/shrinker.c:212 at set_shrinker_bit+0x99/0xa0 On systems where a SHRINKER_NONSLAB shrinker (btrfs, xfs) did register and expand the maps, there is no warning; instead bit 0 is set spuriously for an unrelated shrinker. shrinker->id is only meaningful while SHRINKER_MEMCG_AWARE is set, and all readers inside mm/shrinker.c already check the flag before using the id. Make __list_lru_init() do the same and fall back to -1, so set_shrinker_bit() is never reached with a bogus id. The stale shrinker->id itself is left as is; cleaning that up is a separate topic. Verified on a machine booting with cgroup.memory=nokmem and CONFIG_TRANSPARENT_HUGEPAGE=y: the warning fires once per boot from khugepaged, disappears when nokmem is removed from the command line, and no longer triggers with this fix applied and nokmem set. Link: https://lore.kernel.org/20260901115104.2944996-1-qinyuntan@linux.alibaba.com Fixes: 03375203e1da ("mm: do not allocate shrinker info with cgroup.memory=nokmem") Signed-off-by: Qinyun Tan Acked-by: Muchun Song Reviewed-by: Baolin Wang Cc: Michal Hocko Cc: Roman Gushchin Cc: Johannes Weiner Cc: Shakeel Butt Cc: Dave Chinner Cc: David Hildenbrand Cc: Lance Yang Cc: Michal Koutný Cc: Xunlei Pang Cc: Signed-off-by: Andrew Morton --- mm/list_lru.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/mm/list_lru.c b/mm/list_lru.c index a4522ca93ebcb9..8a6dd0a489e12c 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -666,7 +666,12 @@ int __list_lru_init(struct list_lru *lru, bool memcg_aware, struct shrinker *shr int i; #ifdef CONFIG_MEMCG - if (shrinker) + /* + * If the shrinker fell back to being non-memcg-aware (e.g. with + * cgroup.memory=nokmem), its id was never assigned and holds a + * stale 0. Don't let set_shrinker_bit() act on it. + */ + if (shrinker && (shrinker->flags & SHRINKER_MEMCG_AWARE)) lru->shrinker_id = shrinker->id; else lru->shrinker_id = -1; From ff9bb4d569cfed1eca2919ec4629eb086592f482 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 1 Sep 2026 15:38:04 -0400 Subject: [PATCH 725/857] mm: remove unused mark_page_reserved() Patch series "mm: remove three unused helpers from mm.h", v2. I happened to notice these were unused. Two of them are relatively recently unused, and one has been unused for a few years. Remove them. This patch (of 2): mark_page_reserved() lost its last caller in commit 6215d9f4470f ("arch, mm: consolidate empty_zero_page"). Remove it. Link: https://lore.kernel.org/20260901-mm-remove-unused-helpers-v2-0-f6474e169c23@columbia.edu Link: https://lore.kernel.org/20260901-mm-remove-unused-helpers-v2-1-f6474e169c23@columbia.edu Signed-off-by: Tal Zussman Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/mm.h | 6 ------ 1 file changed, 6 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 1b28e6fc8d5dd1..e3736c42c4dbc0 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4080,12 +4080,6 @@ static inline void free_reserved_page(struct page *page) free_reserved_pages(page, 0); } -static inline void mark_page_reserved(struct page *page) -{ - SetPageReserved(page); - adjust_managed_page_count(page, -1); -} - static inline void free_reserved_ptdesc(struct ptdesc *pt) { free_reserved_page(ptdesc_page(pt)); From 931dc4e7a0d3380707e0c5cda736bb0acd657654 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 1 Sep 2026 15:38:05 -0400 Subject: [PATCH 726/857] mm: remove unused totalram_pages_inc() and totalram_pages_dec() totalram_pages_inc() and totalram_pages_dec() have had no callers since commit 7fbc5e26123e ("memblock: extract page freeing from free_reserved_area() into a helper") and commit 287b89773d81 ("powerpc/pseries/cmm: Use adjust_managed_page_count() insted of totalram_pages_*"), respectively. Remove them. Drop the totalram_pages_inc() stub from tools mm.h too. Link: https://lore.kernel.org/20260901-mm-remove-unused-helpers-v2-2-f6474e169c23@columbia.edu Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Signed-off-by: Tal Zussman Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- include/linux/mm.h | 10 ---------- tools/include/linux/mm.h | 4 ---- 2 files changed, 14 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index e3736c42c4dbc0..c105a3758915b2 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -57,16 +57,6 @@ static inline unsigned long totalram_pages(void) return (unsigned long)atomic_long_read(&_totalram_pages); } -static inline void totalram_pages_inc(void) -{ - atomic_long_inc(&_totalram_pages); -} - -static inline void totalram_pages_dec(void) -{ - atomic_long_dec(&_totalram_pages); -} - static inline void totalram_pages_add(long count) { atomic_long_add(count, &_totalram_pages); diff --git a/tools/include/linux/mm.h b/tools/include/linux/mm.h index 84b5954f66c3d5..d586a510e6e18b 100644 --- a/tools/include/linux/mm.h +++ b/tools/include/linux/mm.h @@ -33,10 +33,6 @@ static inline phys_addr_t virt_to_phys(volatile void *address) return (phys_addr_t)address; } -static inline void totalram_pages_inc(void) -{ -} - static inline void totalram_pages_add(long count) { } From 9ef0b42b20b21cb88a2a0426ac39f88fb0222b13 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:01 +0900 Subject: [PATCH 727/857] percpu: remove redundant assignments to bits Patch series "percpu: remove code with no effect". While reading the percpu initialization path, I found some minor cleanups for unnecessary initializations and an obsolete return statement. No functional change. This patch (of 4): pcpu_chunk_refresh_hint() and pcpu_find_block_fit() set bits to 0 and later call pcpu_next_md_free_region() or pcpu_next_fit_region(), which unconditionally set *bits to 0. Nothing uses it in between, so the assignments have no effect. Remove redundant assignments. No functional change. Link: https://lore.kernel.org/20260901165307.1026248-1-ekffu200098@gmail.com Link: https://lore.kernel.org/20260901165307.1026248-2-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Cc: Dennis Zhou Cc: Tejun Heo Cc: Christoph Lameter Signed-off-by: Andrew Morton --- mm/percpu.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/percpu.c b/mm/percpu.c index 47a903fe3b5124..b094c617147cb5 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -758,7 +758,6 @@ static void pcpu_chunk_refresh_hint(struct pcpu_chunk *chunk, bool full_scan) chunk_md->contig_hint = 0; } - bits = 0; pcpu_for_each_md_free_region(chunk, bit_off, bits) pcpu_block_update(chunk_md, bit_off, bit_off + bits); } @@ -1122,7 +1121,6 @@ static int pcpu_find_block_fit(struct pcpu_chunk *chunk, int alloc_bits, return -1; bit_off = pcpu_next_hint(chunk_md, alloc_bits); - bits = 0; pcpu_for_each_fit_region(chunk, alloc_bits, align, bit_off, bits) { if (!pop_only || pcpu_is_populated(chunk, bit_off, bits, &next_off)) From a9c90f4f4844998831f4336166355690edf651cf Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:02 +0900 Subject: [PATCH 728/857] percpu: remove unnecessary initialization in pcpu_build_alloc_info() pcpu_build_alloc_info() initializes nr_groups to 1 and unconditionally sets it to the number of groups. Nothing uses it in between, so the initialization has no effect. Remove unnecessary initialization. No functional change. Link: https://lore.kernel.org/20260901165307.1026248-3-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Cc: Christoph Lameter Cc: Dennis Zhou Cc: Tejun Heo Signed-off-by: Andrew Morton --- mm/percpu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/percpu.c b/mm/percpu.c index b094c617147cb5..0ec9b2adfc208e 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -2817,7 +2817,7 @@ static struct pcpu_alloc_info * __init __flatten pcpu_build_alloc_info( static int group_cnt[NR_CPUS] __initdata; static struct cpumask mask __initdata; const size_t static_size = __per_cpu_end - __per_cpu_start; - int nr_groups = 1, nr_units = 0; + int nr_groups, nr_units = 0; size_t size_sum, min_unit_size, alloc_size; int upa, max_upa, best_upa; /* units_per_alloc */ int last_allocs, group, unit; From 7e73e7bdefd575a222887ecf749016852676a596 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:03 +0900 Subject: [PATCH 729/857] percpu: remove unnecessary cpumask_clear() in pcpu_build_alloc_info() pcpu_build_alloc_info() clears mask and unconditionally sets it to cpu_possible_mask. Nothing uses it in between, so cpumask_clear() has no effect. Remove unnecessary cpumask_clear(). No functional change. Link: https://lore.kernel.org/20260901165307.1026248-4-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Cc: Christoph Lameter Cc: Dennis Zhou Cc: Tejun Heo Signed-off-by: Andrew Morton --- mm/percpu.c | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/percpu.c b/mm/percpu.c index 0ec9b2adfc208e..fe9fe5a0c3d597 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -2828,7 +2828,6 @@ static struct pcpu_alloc_info * __init __flatten pcpu_build_alloc_info( /* this function may be called multiple times */ memset(group_map, 0, sizeof(group_map)); memset(group_cnt, 0, sizeof(group_cnt)); - cpumask_clear(&mask); /* calculate size_sum and ensure dyn_size is enough for early alloc */ size_sum = PFN_ALIGN(static_size + reserved_size + From 45c8b710db8f597121759d7995974a6a0faf347d Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:04 +0900 Subject: [PATCH 730/857] percpu: remove unnecessary return in pcpu_populate_pte() Since commit c6f239796b55 ("mm/memblock: add memblock_alloc_or_panic interface"), pcpu_populate_pte() no longer has an error label after the return statement, so the return statement has no effect. Remove unnecessary return. No functional change. Link: https://lore.kernel.org/20260901165307.1026248-5-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Cc: Christoph Lameter Cc: Dennis Zhou Cc: Tejun Heo Signed-off-by: Andrew Morton --- mm/percpu.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/percpu.c b/mm/percpu.c index fe9fe5a0c3d597..3eff382e565ad5 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -3178,8 +3178,6 @@ void __init __weak pcpu_populate_pte(unsigned long addr) new = memblock_alloc_or_panic(PTE_TABLE_SIZE, PTE_TABLE_SIZE); pmd_populate_kernel(&init_mm, pmd, new); } - - return; } /** From 7577a90cbb4d43bdf7d9e82b2347afdd393fabce Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Tue, 26 May 2026 15:14:09 -0700 Subject: [PATCH 731/857] drivers/media/v4l2-core/v4l2-vp9.c: reduce inlining csky allmodconfig, gcc-15.2.0: drivers/media/v4l2-core/v4l2-vp9.c: In function 'v4l2_vp9_adapt_noncoef_probs': drivers/media/v4l2-core/v4l2-vp9.c:1834:1: error: the frame size of 1436 bytes is larger than 1280 bytes [-Werror=frame-larger-than=] The amount of inlining in there is simply nuts. This patch semi-randomly uninlines various things and fixes the above. Ad the .text size reduction is tremendous: ts:/usr/src/25> size drivers/media/v4l2-core/v4l2-vp9.o text data bss dec hex filename 22450 36 0 22486 57d6 drivers/media/v4l2-core/v4l2-vp9.o-before 16144 36 0 16180 3f34 drivers/media/v4l2-core/v4l2-vp9.o-after Reviewed-by: Daniel Almeida Cc: Mauro Carvalho Chehab Signed-off-by: Andrew Morton --- drivers/media/v4l2-core/v4l2-vp9.c | 30 +++++++++++++++--------------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/drivers/media/v4l2-core/v4l2-vp9.c b/drivers/media/v4l2-core/v4l2-vp9.c index 859589f1fd35f5..e965ffbd9b8aa2 100644 --- a/drivers/media/v4l2-core/v4l2-vp9.c +++ b/drivers/media/v4l2-core/v4l2-vp9.c @@ -1582,25 +1582,25 @@ static inline u8 noncoef_merge_prob(u8 pre_prob, u32 ct0, u32 ct1) * merge_prob(p[9], c[9], [10]) */ -static inline void merge_probs_variant_a(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) +static noinline_for_stack void merge_probs_variant_a(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) { p[1] = merge_prob(p[1], c[0], c[1] + c[2], count_sat, update_factor); p[2] = merge_prob(p[2], c[1], c[2], count_sat, update_factor); } -static inline void merge_probs_variant_b(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) +static noinline_for_stack void merge_probs_variant_b(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) { p[0] = merge_prob(p[0], c[0], c[1], count_sat, update_factor); } -static inline void merge_probs_variant_c(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_c(u8 *p, const u32 *c) { p[0] = noncoef_merge_prob(p[0], c[2], c[1] + c[0] + c[3]); p[1] = noncoef_merge_prob(p[1], c[0], c[1] + c[3]); p[2] = noncoef_merge_prob(p[2], c[1], c[3]); } -static void merge_probs_variant_d(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_d(u8 *p, const u32 *c) { u32 sum = 0, s2; @@ -1624,20 +1624,20 @@ static void merge_probs_variant_d(u8 *p, const u32 *c) p[8] = noncoef_merge_prob(p[8], c[6], c[7]); } -static inline void merge_probs_variant_e(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_e(u8 *p, const u32 *c) { p[0] = noncoef_merge_prob(p[0], c[0], c[1] + c[2] + c[3]); p[1] = noncoef_merge_prob(p[1], c[1], c[2] + c[3]); p[2] = noncoef_merge_prob(p[2], c[2], c[3]); } -static inline void merge_probs_variant_f(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_f(u8 *p, const u32 *c) { p[0] = noncoef_merge_prob(p[0], c[0], c[1] + c[2]); p[1] = noncoef_merge_prob(p[1], c[1], c[2]); } -static void merge_probs_variant_g(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_g(u8 *p, const u32 *c) { u32 sum; @@ -1659,12 +1659,12 @@ static void merge_probs_variant_g(u8 *p, const u32 *c) } /* 8.4.3 Coefficient probability adaptation process */ -static inline void adapt_probs_variant_a_coef(u8 *p, const u32 *c, u32 update_factor) +static noinline_for_stack void adapt_probs_variant_a_coef(u8 *p, const u32 *c, u32 update_factor) { merge_probs_variant_a(p, c, 24, update_factor); } -static inline void adapt_probs_variant_b_coef(u8 *p, const u32 *c, u32 update_factor) +static noinline_for_stack void adapt_probs_variant_b_coef(u8 *p, const u32 *c, u32 update_factor) { merge_probs_variant_b(p, c, 24, update_factor); } @@ -1724,33 +1724,33 @@ static inline void adapt_probs_variant_b(u8 *p, const u32 *c) merge_probs_variant_b(p, c, 20, 128); } -static inline void adapt_probs_variant_c(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_c(u8 *p, const u32 *c) { merge_probs_variant_c(p, c); } -static inline void adapt_probs_variant_d(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_d(u8 *p, const u32 *c) { merge_probs_variant_d(p, c); } -static inline void adapt_probs_variant_e(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_e(u8 *p, const u32 *c) { merge_probs_variant_e(p, c); } -static inline void adapt_probs_variant_f(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_f(u8 *p, const u32 *c) { merge_probs_variant_f(p, c); } -static inline void adapt_probs_variant_g(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_g(u8 *p, const u32 *c) { merge_probs_variant_g(p, c); } /* 8.4.4 Non coefficient probability adaptation process, adapt_prob() */ -static inline u8 adapt_prob(u8 prob, const u32 counts[2]) +static noinline_for_stack u8 adapt_prob(u8 prob, const u32 counts[2]) { return noncoef_merge_prob(prob, counts[0], counts[1]); } From 1bf0e5788faf7a0fd590b848340283f4b3d1a862 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Tue, 4 Aug 2026 09:34:00 +0000 Subject: [PATCH 732/857] taskstats: copy signal->stats under siglock in taskstats_exit taskstats_exit() copies tsk->signal->stats into the exit reply without taking any lock. Every other writer of this struct holds sighand->siglock before touching it, and this copy does not. The copy happens on the last thread of a thread group that exits. group_dead being 1 only says that every thread has dropped signal->live, it does not say how far the other threads got in do_exit(). One of them can still be inside fill_tgid_exit() adding its counters to the struct while the last thread copies it out, so the copy can read the struct in the middle of an update. The commit that added the copy assumed no locking was needed because the group was dead: /* No locking needed for tsk->signal->stats since group is dead */ but at that point the other threads have not necessarily finished their exit path. cpu0 (thread A, not last) cpu1 (thread B, last) =========================== ============================== atomic_dec(&signal->live) atomic_dec(&signal->live) -> 0 group_dead = 0 group_dead = 1 ... taskstats_exit(tsk, 1) taskstats_exit(tsk, 0) fill_tgid_exit(tsk) [siglock] fill_tgid_exit(tsk) memcpy(stats, signal->stats) spin_lock(siglock) reads ac_utime (new) stats->ac_utime += x reads ac_stime (old) stats->ac_stime += y torn snapshot -> netlink spin_unlock(siglock) The listeners receive a partially updated tgid snapshot, with some fields from before the concurrent update and some from after. There is no crash or splat, which is likely why this went unnoticed since 2006. A userspace model of the same shape, writer under a lock and a lockless memcpy reader, produces millions of torn reads in a few seconds. Take siglock around the copy like every other access does. sighand is still alive here because taskstats_exit() runs before exit_notify(), and fill_tgid_exit() already takes this same lock earlier in this function. Link: https://lore.kernel.org/20260804093400.3922-1-include@grrlz.net Fixes: ad4ecbcba728 ("[PATCH] delay accounting taskstats interface send tgid once") Signed-off-by: Bradley Morgan Acked-by: Oleg Nesterov Cc: Balbir Singh Signed-off-by: Andrew Morton --- kernel/taskstats.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/kernel/taskstats.c b/kernel/taskstats.c index f31df72f0e9df1..9a48827e22bce3 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -590,6 +590,7 @@ void taskstats_exit(struct task_struct *tsk, int group_dead) struct sk_buff *rep_skb; size_t size; int is_thread_group; + unsigned long flags; if (!family_registered) return; @@ -635,7 +636,10 @@ void taskstats_exit(struct task_struct *tsk, int group_dead) if (!stats) goto err; + /* This was racy before, copy the stats under siglock. */ + spin_lock_irqsave(&tsk->sighand->siglock, flags); memcpy(stats, tsk->signal->stats, sizeof(*stats)); + spin_unlock_irqrestore(&tsk->sighand->siglock, flags); stats->version = TASKSTATS_VERSION; send: From f20f22cbe688dcdadb2d2b8b20a42080f40aa1df Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Mon, 10 Aug 2026 18:11:59 +0200 Subject: [PATCH 733/857] checkpatch: skip CamelCase cache for --no-tree without root Running outside tree (--no-tree) without git root (--root DIR) is not doable because we have no include/ directory which could be cached. But 3445686af721 expected that we are always in Linux tree (w/a git). But --no-tree does not require --root. Therefore skip whole caching in that case. This fixes perl and find errors when running checkpatch.pl *with* --no-tree --strict and *without* --root: No structs that should be const will be found - file 'scripts/const_structs.checkpatch': No such file or directory Use of uninitialized value $root in concatenation (.) or string at scripts/checkpatch.pl line 1213. find: `/include': No such file or directory Link: https://lore.kernel.org/20260810161159.1044160-1-pvorel@suse.cz Fixes: 3445686af721 ("checkpatch: ignore existing CamelCase uses from include/...") Signed-off-by: Petr Vorel Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn Signed-off-by: Andrew Morton --- scripts/checkpatch.pl | 2 ++ 1 file changed, 2 insertions(+) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index 8a7787d228a63d..f424dafce5bce7 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -1206,6 +1206,8 @@ sub seed_camelcase_includes { my $git_last_include_commit = `${git_command} log --no-merges --pretty=format:"%h%n" -1 -- include`; chomp $git_last_include_commit; $camelcase_cache = ".checkpatch-camelcase.git.$git_last_include_commit"; + } elsif (not defined $root) { + return; } else { my $last_mod_date = 0; $files = `find $root/include -name "*.h"`; From cd2943868dae888227665f04465b4a91d878c50c Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Thu, 20 Aug 2026 09:19:12 -0400 Subject: [PATCH 734/857] USB: gadgetfs: do not WARN about excessively large memory allocations GadgetFS passes an excessively large user input len to kmalloc and kmalloc gives a WARN (see below for details). Suppress it by passing __GFP_NOWARN to kmalloc used by both ep_write_iter() and ep_read_iter(). Follow the same method as commit 4f2629ea67e72 ("USB: usbfs: Don't WARN about excessively large memory allocations"). kmalloc is used to allocate physically contiguous memory for kernel allocations. For requests larger than KMALLOC_MAX_CACHE_SIZE, kmalloc uses the page allocator and can only support up to KMALLOC_MAX_SIZE. For request sizes bigger than KMALLOC_MAX_SIZE, the page allocator can emit a WARN because kmalloc allocates an order greater than MAX_PAGE_ORDER. Link: https://lore.kernel.org/DKTTMAS94IMH.2C6ERY0ZIVWVZ@nvidia.com Fixes: b3c466ce5129 ("page allocator: do not sanity check order in the fast path") Signed-off-by: Zi Yan Reported-by: syzbot+805630f1453e490427fa@syzkaller.appspotmail.com Closes: https://lore.kernel.org/all/6a820ebc.9ebadd4d.20b15e.001b.GAE@google.com/ Tested-by: syzbot+805630f1453e490427fa@syzkaller.appspotmail.com Acked-by: Alan Stern Acked-by: Vlastimil Babka (SUSE) Cc: Greg Kroah-Hartman Cc: Signed-off-by: Andrew Morton --- drivers/usb/gadget/legacy/inode.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/usb/gadget/legacy/inode.c b/drivers/usb/gadget/legacy/inode.c index db961aaa3740d6..c48f0fbcd0d856 100644 --- a/drivers/usb/gadget/legacy/inode.c +++ b/drivers/usb/gadget/legacy/inode.c @@ -613,7 +613,7 @@ ep_read_iter(struct kiocb *iocb, struct iov_iter *to) return -EBADMSG; } - buf = kmalloc(len, GFP_KERNEL); + buf = kmalloc(len, GFP_KERNEL | __GFP_NOWARN); if (unlikely(!buf)) { mutex_unlock(&epdata->lock); return -ENOMEM; @@ -675,7 +675,7 @@ ep_write_iter(struct kiocb *iocb, struct iov_iter *from) return -EBADMSG; } - buf = kmalloc(len, GFP_KERNEL); + buf = kmalloc(len, GFP_KERNEL | __GFP_NOWARN); if (unlikely(!buf)) { mutex_unlock(&epdata->lock); return -ENOMEM; From e14599274d6c22ceb682314f714b10a3dee06395 Mon Sep 17 00:00:00 2001 From: DAI RENJIE Date: Fri, 21 Aug 2026 14:08:17 +0000 Subject: [PATCH 735/857] resource: fix lost wakeup when waiting for a muxed region A task waiting for a muxed region can sleep forever in TASK_UNINTERRUPTIBLE even though the region it waits for is already free. __request_region_locked() queues itself on muxed_resource_wait and drops resource_lock before setting TASK_UNINTERRUPTIBLE, while __release_region() wakes the queue after dropping the same lock. A wakeup landing in between finds TASK_RUNNING, does not match TASK_NORMAL and is discarded; callers hold a muxed region only across a bounded transaction, so no further release is coming. The task is unkillable and its caller never returns. The window is one store wide, but an interrupt is enough to hold the waiter in it, and the machine this was seen on runs PREEMPT_DYNAMIC in its voluntary default. Since v6.11 spd5118 exports the DDR5 sensors of AMD boards through i2c-piix4, which takes a muxed region per SMBus transaction; a third of the in-tree users of request_muxed_region() are hwmon drivers, so reading a world-readable attribute is all an unprivileged user needs to drive the contention. The blocked task sleeps holding the i2c adapter bus lock, and 27 more piled up behind it. Reproduced by building a kernel with the two orderings selectable at runtime and a 2ms delay inside the window. Switching only that knob, a two-thread barriered reproducer loses the wakeup 200 times out of 200 before the fix and 0 out of 200 after it; without the delay it goes 20000 times through the wait path and loses none. Fix it by setting the task state before dropping resource_lock, as prepare_to_wait() does: the releasing side needs resource_lock to unlink the resource, so it cannot reach the wakeup before the state is published. Link: https://lore.kernel.org/20260821-b4-resource-muxed-lost-wakeup-v1-1-37eb6473a76c@gmail.com Fixes: 8b6d043b7ee2 ("resource: shared I/O region support") Signed-off-by: DAI RENJIE Reviewed-by: Bradley Morgan Assisted-by: Claude:claude-opus-5 Reviewed-by: Andrew Morton Cc: Mark Brown Cc: Kees Cook Cc: Bjorn Helgaas Cc: Signed-off-by: Andrew Morton --- kernel/resource.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/resource.c b/kernel/resource.c index e60539a55541df..54d7199695fbe9 100644 --- a/kernel/resource.c +++ b/kernel/resource.c @@ -1350,8 +1350,8 @@ static int __request_region_locked(struct resource *res, struct resource *parent } if (conflict->flags & flags & IORESOURCE_MUXED) { add_wait_queue(&muxed_resource_wait, &wait); - write_unlock(&resource_lock); set_current_state(TASK_UNINTERRUPTIBLE); + write_unlock(&resource_lock); schedule(); remove_wait_queue(&muxed_resource_wait, &wait); write_lock(&resource_lock); From c0037bb347e1e11cb6227d62d0183761da5f4a62 Mon Sep 17 00:00:00 2001 From: Michael Liang Date: Fri, 21 Aug 2026 12:15:27 -0600 Subject: [PATCH 736/857] fault-inject: fix dentry leak fault_create_debugfs_attr() has always taken an extra dentry reference on the created directory (attr->dname = dget(dir)) so that fail_dump() could print the name via %pd from any context. Nothing anywhere in the tree ever calls dput() on attr->dname. For callers with a matching teardown, that unmatched reference causes one dentry plus its attached inode to leak per fault_create_debugfs_attr / debugfs_remove_recursive cycle. simple_recursive_removal() drops debugfs's own +1 ref on the child dentry, but the dget()'d ref keeps its refcount at 1: the dentry ends up unhashed but pinned, and its inode is never freed. Boot-once callers (mm/failslab, block/blk-core, etc.) leak exactly once at init and never destroy the tree, so the impact there is bounded. But per-lifecycle callers (drivers/nvme, drivers/infiniband/hw/hfi1, drivers/mmc, drivers/iommu/iommufd, drivers/media, drivers/misc, drivers/gpu/drm/msm, drivers/crypto, net/sunrpc) leak on every create/destroy cycle. We observed this in production: an NVMe/RDMA host repeatedly reconnecting to a target that rejected the CRTO Property Get went through ~50 nvme controller create/destroy cycles per second, and dentry and inode_cache grew by ~13k pinned objects per 240 s -- unrecoverable through drop_caches. Byte math matched a per-cycle 1-dentry / 1-inode leak from the "fault_inject" directory dentry. Fix this by not holding any external reference in fault_attr. Embed the directory name as a fixed-size char array (FAULT_ATTR_DNAME_LEN, 64 bytes) inside struct fault_attr, copied by strscpy() at fault_create_debugfs_attr() time. fail_dump() prints it via %s. Advantages of an embedded array over kstrdup() + kfree() paired with a new destroy API: - Zero API footprint. No new export and no caller changes required: callers already own their fault_attr's memory and free it when they are done, and now that suffices. - No allocation on the create path. - fault_create_debugfs_attr() cannot fail from the name-copy step. - No lifetime coupling between attr->dname and debugfs; the string is valid for exactly as long as the containing struct. The 64-byte length accommodates every in-tree caller with generous headroom (the longest current name is "fail_dma_array_full", 19 chars). The user-visible fail_dump() format changes from "name %pd" to "name %s", but the printed content is identical -- %pd on the created directory renders the same string that was passed in as @name. drivers/infiniband/hw/hfi1/fault.c drops a now-invalid "attr.dname = NULL" statement; the surrounding kzalloc() already zero-initialises the array. Link: https://lore.kernel.org/20260821181527.3271414-1-mliang@purestorage.com Fixes: 6adc4a22f20b ("fault-inject: add ratelimit option") Signed-off-by: Michael Liang Reviewed-by: Andrew Morton Cc: Akinbou Mita Cc: Dennis Dalessandro Cc: Jason Gunthorpe Cc: Leon Romanovsky Cc: Vlastimil Babka Cc: Signed-off-by: Andrew Morton --- drivers/infiniband/hw/hfi1/fault.c | 1 - include/linux/fault-inject.h | 10 ++++++++-- lib/fault-inject.c | 7 +++++-- 3 files changed, 13 insertions(+), 5 deletions(-) diff --git a/drivers/infiniband/hw/hfi1/fault.c b/drivers/infiniband/hw/hfi1/fault.c index 4ab72ef03ba11b..941a0b96590b60 100644 --- a/drivers/infiniband/hw/hfi1/fault.c +++ b/drivers/infiniband/hw/hfi1/fault.c @@ -216,7 +216,6 @@ int hfi1_fault_init_debugfs(struct hfi1_ibdev *ibd) ibd->fault->attr.interval = 1; ibd->fault->attr.require_end = ULONG_MAX; ibd->fault->attr.stacktrace_depth = 32; - ibd->fault->attr.dname = NULL; ibd->fault->attr.verbose = 0; ibd->fault->enable = false; ibd->fault->opcode = false; diff --git a/include/linux/fault-inject.h b/include/linux/fault-inject.h index 58fd14c8227080..5c74748a53f38f 100644 --- a/include/linux/fault-inject.h +++ b/include/linux/fault-inject.h @@ -18,6 +18,13 @@ enum fault_flags { #include #include +/* + * Length of the debugfs directory name embedded in struct fault_attr. + * Chosen to accommodate every in-tree caller of fault_create_debugfs_attr() + * (the longest is "fail_dma_array_full", 19 chars) with generous headroom. + */ +#define FAULT_ATTR_DNAME_LEN 64 + /* * For explanation of the elements of this struct, see * Documentation/fault-injection/fault-injection.rst @@ -37,7 +44,7 @@ struct fault_attr { unsigned long count; struct ratelimit_state ratelimit_state; - struct dentry *dname; + char dname[FAULT_ATTR_DNAME_LEN]; }; #define FAULT_ATTR_INITIALIZER { \ @@ -47,7 +54,6 @@ struct fault_attr { .stacktrace_depth = 32, \ .ratelimit_state = RATELIMIT_STATE_INIT_DISABLED, \ .verbose = 2, \ - .dname = NULL, \ } #define DECLARE_FAULT_ATTR(name) struct fault_attr name = FAULT_ATTR_INITIALIZER diff --git a/lib/fault-inject.c b/lib/fault-inject.c index 999053fa133e3f..02916ef2761c60 100644 --- a/lib/fault-inject.c +++ b/lib/fault-inject.c @@ -5,6 +5,7 @@ #include #include #include +#include #include #include #include @@ -64,7 +65,7 @@ static void fail_dump(struct fault_attr *attr) { if (attr->verbose > 0 && __ratelimit(&attr->ratelimit_state)) { printk(KERN_NOTICE "FAULT_INJECTION: forcing a failure.\n" - "name %pd, interval %lu, probability %lu, " + "name %s, interval %lu, probability %lu, " "space %d, times %d\n", attr->dname, attr->interval, attr->probability, atomic_read(&attr->space), @@ -261,7 +262,9 @@ struct dentry *fault_create_debugfs_attr(const char *name, debugfs_create_xul("reject-end", mode, dir, &attr->reject_end); #endif /* CONFIG_FAULT_INJECTION_STACKTRACE_FILTER */ - attr->dname = dget(dir); + if (strscpy(attr->dname, name, sizeof(attr->dname)) == -E2BIG) + pr_warn("FAULT_INJECTION: name '%s' truncated to '%s'\n", + name, attr->dname); return dir; } EXPORT_SYMBOL_GPL(fault_create_debugfs_attr); From 6e37a06fc78a1ed7618112ba13e7f6e3fddbcc2f Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Tue, 25 Aug 2026 15:11:57 +0000 Subject: [PATCH 737/857] mailmap: update email address for Bradley Morgan I switched to brads@mainlining.org for kernel work, so map the old include@grrlz.net address over to keep shortlog and blame from splitting commits between the two. Link: https://lore.kernel.org/20260825151157.4533-1-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index 22f016bc4f7f8c..ab90b70764c4eb 100644 --- a/.mailmap +++ b/.mailmap @@ -170,6 +170,7 @@ Boris Brezillon Boris Brezillon Boris Brezillon Boris Brezillon +Bradley Morgan Brendan Higgins Brendan Jackman Brian Avery From 089da2c295657405734241ca6a80e3507fe3e907 Mon Sep 17 00:00:00 2001 From: OGAWA Hirofumi Date: Tue, 25 Aug 2026 21:11:32 +0900 Subject: [PATCH 738/857] fat: fix fat_ent_write() for reverting the value commit 64d9183203ee ("fat: restore original value when fat_ent_write failed") try to revert the fatent value to old value when got the error on mirror FAT. However it didn't work if the error is when writing the fatent bh. In that case, the bh is cleared the uptodate flag, so reuse bh is invalid. Fix this by reverting the fatent only if got the error on mirror FAT. Link: https://lore.kernel.org/87ik4yz9fv.fsf_-_@mail.parknet.co.jp Fixes: 64d9183203ee ("fat: restore original value when fat_ent_write failed") Signed-off-by: OGAWA Hirofumi Reported-by: syzbot+e64c6472a3d96a75172a@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=e64c6472a3d96a75172a Reported-by: syzbot+26461e903494e689c24f@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=26461e903494e689c24f Cc: Yemu Lu Cc: Ren Wei Cc: Yuan Tan Cc: Yifan Wu Cc: Juefei Pu Cc: Xin Liu Cc: Signed-off-by: Andrew Morton --- fs/fat/fat.h | 2 +- fs/fat/fatent.c | 21 ++++++++++++++++++--- fs/fat/file.c | 3 ++- fs/fat/misc.c | 6 ++---- 4 files changed, 23 insertions(+), 9 deletions(-) diff --git a/fs/fat/fat.h b/fs/fat/fat.h index 61338413d9f3e4..fbd207c55859e7 100644 --- a/fs/fat/fat.h +++ b/fs/fat/fat.h @@ -392,7 +392,7 @@ extern void fat_ent_access_init(struct super_block *sb); extern int fat_ent_read(struct inode *inode, struct fat_entry *fatent, int entry); extern int fat_ent_write(struct inode *inode, struct fat_entry *fatent, - int new, int wait); + int new, int old, int wait); extern int fat_alloc_clusters(struct inode *inode, int *cluster, int nr_cluster); extern int fat_free_clusters(struct inode *inode, int cluster); diff --git a/fs/fat/fatent.c b/fs/fat/fatent.c index f0801d99dd62ae..df23fc85f31307 100644 --- a/fs/fat/fatent.c +++ b/fs/fat/fatent.c @@ -413,7 +413,7 @@ static int fat_mirror_bhs(struct super_block *sb, struct buffer_head **bhs, } int fat_ent_write(struct inode *inode, struct fat_entry *fatent, - int new, int wait) + int new, int old, int wait) { struct super_block *sb = inode->i_sb; const struct fatent_operations *ops = MSDOS_SB(sb)->fatent_ops; @@ -422,10 +422,25 @@ int fat_ent_write(struct inode *inode, struct fat_entry *fatent, ops->ent_put(fatent, new); if (wait) { err = fat_sync_bhs(fatent->bhs, fatent->nr_bhs); - if (err) + if (err) { + /* + * bhs are not uptodate after I/O error. So we + * can't simply re-dirty to revert. And it + * would not have value to write again on I/O + * error. + */ return err; + } } - return fat_mirror_bhs(sb, fatent->bhs, fatent->nr_bhs); + + err = fat_mirror_bhs(sb, fatent->bhs, fatent->nr_bhs); + if (err) { + /* Try to revert if got the error on mirror FAT */ + ops->ent_put(fatent, old); + if (wait) + fat_sync_bhs(fatent->bhs, fatent->nr_bhs); + } + return err; } static inline int fat_ent_next(struct msdos_sb_info *sbi, diff --git a/fs/fat/file.c b/fs/fat/file.c index 1c835ca5f21a51..6c475c53334c6c 100644 --- a/fs/fat/file.c +++ b/fs/fat/file.c @@ -363,7 +363,8 @@ static int fat_free(struct inode *inode, int skip) __func__, MSDOS_I(inode)->i_pos); ret = -EIO; } else if (ret > 0) { - err = fat_ent_write(inode, &fatent, FAT_ENT_EOF, wait); + err = fat_ent_write(inode, &fatent, FAT_ENT_EOF, ret, + wait); if (err) ret = err; } diff --git a/fs/fat/misc.c b/fs/fat/misc.c index e79762cf19754d..c44296756eae61 100644 --- a/fs/fat/misc.c +++ b/fs/fat/misc.c @@ -133,11 +133,9 @@ int fat_chain_add(struct inode *inode, int new_dclus, int nr_cluster) ret = fat_ent_read(inode, &fatent, last); if (ret >= 0) { int wait = inode_needs_sync(inode); - int old = ret; - ret = fat_ent_write(inode, &fatent, new_dclus, wait); - if (ret < 0) - fat_ent_write(inode, &fatent, old, wait); + ret = fat_ent_write(inode, &fatent, new_dclus, ret, + wait); fatent_brelse(&fatent); } if (ret < 0) From 5ad5c7cf66d9dbf0d000c2c549a9ab6340abe29d Mon Sep 17 00:00:00 2001 From: Thorsten Blum Date: Wed, 26 Aug 2026 12:09:03 +0200 Subject: [PATCH 739/857] init: fix early boot crash with bare hostname parameter When a bare hostname parameter is specified on the kernel command line without the '=' separator, early parameter parsing passes NULL to early_hostname(), which dereferences it in strscpy() and can crash the system during early boot. Reject NULL values in early_hostname() and return -EINVAL instead. Link: https://lore.kernel.org/20260826100904.296151-2-blum@kernel.org Fixes: 5a704629f2c1 ("init: add "hostname" kernel parameter") Signed-off-by: Thorsten Blum Cc: Dan Moulding Cc: Signed-off-by: Andrew Morton --- init/version.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/init/version.c b/init/version.c index 94c96f6fbfe6a2..0bd5c45aabc463 100644 --- a/init/version.c +++ b/init/version.c @@ -23,6 +23,9 @@ static int __init early_hostname(char *arg) size_t maxlen = bufsize - 1; ssize_t arglen; + if (!arg) + return -EINVAL; + arglen = strscpy(init_uts_ns.name.nodename, arg, bufsize); if (arglen < 0) { pr_warn("hostname parameter exceeds %zd characters and will be truncated", From 3856c04507723309d6c9330faca2fbb7a5d88c8f Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Wed, 26 Aug 2026 19:26:59 +0800 Subject: [PATCH 740/857] ocfs2: fix deadlock in inline-data truncate transactions Updating an inode xattr can cause an ABBA deadlock with inline file truncation: ocfs2_truncate_file() down_write(&oi->ip_alloc_sem) ocfs2_truncate_inline() ocfs2_start_trans() ocfs2_xattr_set() ocfs2_start_trans() ocfs2_xattr_ibody_set() down_write(&oi->ip_alloc_sem) The xattr set path starts the merged transaction before the inode-body xattr helper acquires ip_alloc_sem, reversing the ip_alloc_sem -> transaction order used by the allocation and truncate paths. The transaction merge in commit 85db90e77806 ("ocfs2/xattr: Merge xattr set transaction.") introduced this ordering. Fix it by acquiring ip_alloc_sem once in ocfs2_xattr_set(), before xattr preparation, allocation reservations and ocfs2_start_trans(), and removing the per-helper acquisition from ocfs2_xattr_ibody_find(), ocfs2_xattr_ibody_set() and ocfs2_xattr_create_index_block(). These helpers now assert via lockdep that the caller holds ip_alloc_sem. ocfs2_xattr_set_handle(), which only sets initial ACL or security xattrs on unpublished inodes inside the create transaction, takes ip_alloc_sem under a dedicated lockdep subclass so that the assertions hold without creating a transaction -> ip_alloc_sem cycle against the ip_alloc_sem -> transaction order. The inode is unpublished, so the acquisition can never contend. This keeps the established ip_alloc_sem -> transaction order and makes the locking unconditional, so lockdep can verify a single plain ordering instead of conditional acquisitions. Link: https://lore.kernel.org/20260826112659.246574-1-joseph.qi@linux.alibaba.com Fixes: 85db90e77806 ("ocfs2/xattr: Merge xattr set transaction.") Signed-off-by: Joseph Qi Cc: ZhengYuan Huang Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao Signed-off-by: Andrew Morton --- fs/ocfs2/xattr.c | 79 ++++++++++++++++++++++++++++++++---------------- 1 file changed, 53 insertions(+), 26 deletions(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 35bcbb0ff607b2..0062cbeb1e8be5 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -2912,6 +2912,9 @@ static int ocfs2_xattr_has_space_inline(struct inode *inode, * * Find extended attribute in inode block and * fill search info into struct ocfs2_xattr_search. + * + * The inline free-space check races with truncate and allocation, so + * callers must hold ip_alloc_sem for writing. */ static int ocfs2_xattr_ibody_find(struct inode *inode, int name_index, @@ -2923,13 +2926,13 @@ static int ocfs2_xattr_ibody_find(struct inode *inode, int ret; int has_space = 0; + lockdep_assert_held_write(&oi->ip_alloc_sem); + if (inode->i_sb->s_blocksize == OCFS2_MIN_BLOCKSIZE) return 0; if (!(oi->ip_dyn_features & OCFS2_INLINE_XATTR_FL)) { - down_read(&oi->ip_alloc_sem); has_space = ocfs2_xattr_has_space_inline(inode, di); - up_read(&oi->ip_alloc_sem); if (!has_space) return 0; } @@ -3010,6 +3013,7 @@ static int ocfs2_xattr_ibody_init(struct inode *inode, * * Set, replace or remove an extended attribute into inode block. * + * Callers must hold ip_alloc_sem for writing. */ static int ocfs2_xattr_ibody_set(struct inode *inode, struct ocfs2_xattr_info *xi, @@ -3020,16 +3024,17 @@ static int ocfs2_xattr_ibody_set(struct inode *inode, struct ocfs2_inode_info *oi = OCFS2_I(inode); struct ocfs2_xa_loc loc; + lockdep_assert_held_write(&oi->ip_alloc_sem); + if (inode->i_sb->s_blocksize == OCFS2_MIN_BLOCKSIZE) return -ENOSPC; - down_write(&oi->ip_alloc_sem); if (!(oi->ip_dyn_features & OCFS2_INLINE_XATTR_FL)) { ret = ocfs2_xattr_ibody_init(inode, xs->inode_bh, ctxt); if (ret) { if (ret != -ENOSPC) mlog_errno(ret); - goto out; + return ret; } } @@ -3039,13 +3044,10 @@ static int ocfs2_xattr_ibody_set(struct inode *inode, if (ret) { if (ret != -ENOSPC) mlog_errno(ret); - goto out; + return ret; } xs->here = loc.xl_entry; -out: - up_write(&oi->ip_alloc_sem); - return ret; } @@ -3689,6 +3691,18 @@ static int __ocfs2_xattr_set_handle(struct inode *inode, return ret; } +/* + * ip_alloc_sem subclass for inodes being initialized before publication. + * ocfs2_xattr_set_handle() runs inside the create transaction, so taking + * ip_alloc_sem there adds a transaction -> ip_alloc_sem order that would + * form a lockdep cycle with the ip_alloc_sem -> transaction order used + * elsewhere, if not for this separate subclass. The inode is unpublished + * so the acquisition can never contend. + */ +enum { + OCFS2_IP_ALLOC_SEM_UNPUBLISHED = 1, +}; + /* * This helper is only for setting initial ACL or security xattrs on an inode * that is still unpublished, unhashed, and unattached to a dentry. @@ -3750,6 +3764,13 @@ int ocfs2_xattr_set_handle(handle_t *handle, xis.inode_bh = xbs.inode_bh = di_bh; di = (struct ocfs2_dinode *)di_bh->b_data; + /* + * The inode is unpublished and cannot contend, but take the + * semaphore anyway so the helpers' lockdep assertions hold. + */ + down_write_nested(&OCFS2_I(inode)->ip_alloc_sem, + OCFS2_IP_ALLOC_SEM_UNPUBLISHED); + ret = ocfs2_xattr_ibody_find(inode, name_index, name, &xis); if (ret) goto cleanup; @@ -3762,6 +3783,7 @@ int ocfs2_xattr_set_handle(handle_t *handle, ret = __ocfs2_xattr_set_handle(inode, di, &xi, &xis, &xbs, &ctxt); cleanup: + up_write(&OCFS2_I(inode)->ip_alloc_sem); brelse(xbs.xattr_bh); ocfs2_xattr_bucket_free(xbs.bucket); @@ -3830,30 +3852,38 @@ int ocfs2_xattr_set(struct inode *inode, di = (struct ocfs2_dinode *)di_bh->b_data; down_write(&OCFS2_I(inode)->ip_xattr_sem); + /* + * The allocation and truncate paths take ip_alloc_sem before + * starting a transaction, so take it here before xattr + * preparation, allocation reservations and ocfs2_start_trans() + * to keep that order. The xattr helpers below no longer take + * it themselves. + */ + down_write(&OCFS2_I(inode)->ip_alloc_sem); /* * Scan inode and external block to find the same name * extended attribute and collect search information. */ ret = ocfs2_xattr_ibody_find(inode, name_index, name, &xis); if (ret) - goto cleanup; + goto out_free_ac; if (xis.not_found) { ret = ocfs2_xattr_block_find(inode, name_index, name, &xbs); if (ret) - goto cleanup; + goto out_free_ac; } if (xis.not_found && xbs.not_found) { ret = -ENODATA; if (flags & XATTR_REPLACE) - goto cleanup; + goto out_free_ac; ret = 0; if (!value) - goto cleanup; + goto out_free_ac; } else { ret = -EEXIST; if (flags & XATTR_CREATE) - goto cleanup; + goto out_free_ac; } /* Check whether the value is refcounted and do some preparation. */ @@ -3864,7 +3894,7 @@ int ocfs2_xattr_set(struct inode *inode, &ref_meta, &ref_credits); if (ret) { mlog_errno(ret); - goto cleanup; + goto out_free_ac; } } @@ -3875,7 +3905,7 @@ int ocfs2_xattr_set(struct inode *inode, if (ret < 0) { inode_unlock(tl_inode); mlog_errno(ret); - goto cleanup; + goto out_free_ac; } } inode_unlock(tl_inode); @@ -3884,7 +3914,7 @@ int ocfs2_xattr_set(struct inode *inode, &xbs, &ctxt, ref_meta, &credits); if (ret) { mlog_errno(ret); - goto cleanup; + goto out_free_ac; } /* we need to update inode's ctime field, so add credit for it. */ @@ -3902,6 +3932,7 @@ int ocfs2_xattr_set(struct inode *inode, ocfs2_commit_trans(osb, ctxt.handle); out_free_ac: + up_write(&OCFS2_I(inode)->ip_alloc_sem); if (ctxt.data_ac) ocfs2_free_alloc_context(ctxt.data_ac); if (ctxt.meta_ac) @@ -3910,7 +3941,6 @@ int ocfs2_xattr_set(struct inode *inode, ocfs2_schedule_truncate_log_flush(osb, 1); ocfs2_run_deallocs(osb, &ctxt.dealloc); -cleanup: if (ref_tree) ocfs2_unlock_refcount_tree(osb, ref_tree, 1); up_write(&OCFS2_I(inode)->ip_xattr_sem); @@ -4511,6 +4541,10 @@ static void ocfs2_xattr_update_xattr_search(struct inode *inode, xs->here = &xs->header->xh_entries[i]; } +/* + * Caller must hold ip_alloc_sem for writing, since a new xattr block + * is allocated and the xattr block header is rewritten. + */ static int ocfs2_xattr_create_index_block(struct inode *inode, struct ocfs2_xattr_search *xs, struct ocfs2_xattr_set_ctxt *ctxt) @@ -4526,19 +4560,14 @@ static int ocfs2_xattr_create_index_block(struct inode *inode, struct ocfs2_xattr_tree_root *xr; u16 xb_flags = le16_to_cpu(xb->xb_flags); + lockdep_assert_held_write(&oi->ip_alloc_sem); + trace_ocfs2_xattr_create_index_block_begin( (unsigned long long)xb_bh->b_blocknr); BUG_ON(xb_flags & OCFS2_XATTR_INDEXED); BUG_ON(!xs->bucket); - /* - * XXX: - * We can use this lock for now, and maybe move to a dedicated mutex - * if performance becomes a problem later. - */ - down_write(&oi->ip_alloc_sem); - ret = ocfs2_journal_access_xb(handle, INODE_CACHE(inode), xb_bh, OCFS2_JOURNAL_ACCESS_WRITE); if (ret) { @@ -4600,8 +4629,6 @@ static int ocfs2_xattr_create_index_block(struct inode *inode, ocfs2_journal_dirty(handle, xb_bh); out: - up_write(&oi->ip_alloc_sem); - return ret; } From 0b1c36593e0bebf54d1910e60c10c5cf4b01f1c4 Mon Sep 17 00:00:00 2001 From: ZhengYuan Huang Date: Thu, 6 Aug 2026 16:50:12 +0800 Subject: [PATCH 741/857] ocfs2: reject inconsistent local xattr entries [BUG] A corrupt OCFS2 xattr entry can set OCFS2_XATTR_ENTRY_LOCAL while keeping xe_value_size larger than OCFS2_XATTR_INLINE_SIZE. When that entry reaches namevalue_size_xe(), the filesystem hits its BUG_ON: kernel BUG at fs/ocfs2/xattr.c:231! Oops: invalid opcode: 0000 [#1] SMP KASAN NOPTI RIP: 0010:namevalue_size_xe fs/ocfs2/xattr.c:231 [inline] RIP: 0010:ocfs2_xa_block_wipe_namevalue+0x2e4/0x330 fs/ocfs2/xattr.c:1638 Call Trace: ocfs2_xa_wipe_namevalue fs/ocfs2/xattr.c:1470 [inline] ocfs2_xa_remove_entry+0xae/0x1d0 fs/ocfs2/xattr.c:1941 ocfs2_xa_remove fs/ocfs2/xattr.c:2043 [inline] ocfs2_xa_set+0x11a8/0x30a0 fs/ocfs2/xattr.c:2247 ocfs2_xattr_ibody_set+0x302/0xc50 fs/ocfs2/xattr.c:2795 __ocfs2_xattr_set_handle+0x7e6/0xdb0 fs/ocfs2/xattr.c:3416 ocfs2_xattr_set+0x1447/0x2610 fs/ocfs2/xattr.c:3650 ocfs2_xattr_security_set+0x37/0x50 fs/ocfs2/xattr.c:7241 __vfs_removexattr+0x14d/0x1d0 fs/xattr.c:518 cap_inode_killpriv+0x29/0x50 security/commoncap.c:355 security_inode_killpriv+0x105/0x220 security/security.c:2724 setattr_prepare+0x147/0x8a0 fs/attr.c:219 ocfs2_setattr+0x504/0x1fd0 fs/ocfs2/file.c:1148 notify_change+0x4b5/0x1030 fs/attr.c:546 do_truncate+0x1d2/0x230 fs/open.c:68 handle_truncate fs/namei.c:3596 [inline] do_open fs/namei.c:3979 [inline] path_openat+0x260f/0x2ce0 fs/namei.c:4134 do_filp_open+0x1f6/0x430 fs/namei.c:4161 do_sys_openat2+0x117/0x1c0 fs/open.c:1437 ... [CAUSE] namevalue_size_xe() assumes that local entries contain an inline value no larger than OCFS2_XATTR_INLINE_SIZE. Existing xattr metadata validation only checks whether the value fits the storage region, and cached entries can reach lookup and bucket maintenance paths without a semantic check. A corrupt entry can therefore be passed to namevalue_size_xe(). [FIX] Validate the local/value-size invariant in the existing flat and bucket metadata validators and before accepting matched entries or traversing bucket entries in paths that call namevalue_size_xe(). Return an OCFS2 corruption error instead of firing the assertion. Link: https://lore.kernel.org/20260806085012.2650042-1-gality369@gmail.com Signed-off-by: ZhengYuan Huang Reviewed-by: Joseph Qi Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao Signed-off-by: Andrew Morton --- fs/ocfs2/xattr.c | 53 +++++++++++++++++++++++++++++++++++++++++------- 1 file changed, 46 insertions(+), 7 deletions(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 0062cbeb1e8be5..143d6f75f9c923 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -237,6 +237,21 @@ static int namevalue_size_xe(struct ocfs2_xattr_entry *xe) return namevalue_size(xe->xe_name_len, value_len); } +static int ocfs2_validate_xattr_entry(struct super_block *sb, u64 blkno, + struct ocfs2_xattr_entry *xe) +{ + u64 value_len = le64_to_cpu(xe->xe_value_size); + + if (value_len > OCFS2_XATTR_INLINE_SIZE && + ocfs2_xattr_is_local(xe)) + return ocfs2_error(sb, + "Invalid local xattr in block %llu: value size %llu\n", + (unsigned long long)blkno, + (unsigned long long)value_len); + + return 0; +} + static int ocfs2_xattr_bucket_get_name_value(struct super_block *sb, struct ocfs2_xattr_header *xh, @@ -992,7 +1007,7 @@ static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, size_t entries_limit = region_size; size_t nv_limit = region_size; size_t max_entries; - int i; + int i, ret; if (region_size < sizeof(*xh)) return ocfs2_error(sb, @@ -1012,6 +1027,11 @@ static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, struct ocfs2_xattr_entry *xe = &xh->xh_entries[i]; size_t name_offset = le16_to_cpu(xe->xe_name_offset); size_t value_offset; + u64 value_len = le64_to_cpu(xe->xe_value_size); + + ret = ocfs2_validate_xattr_entry(sb, blkno, xe); + if (ret) + return ret; if (name_offset > nv_limit || xe->xe_name_len > nv_limit - name_offset) @@ -1026,8 +1046,7 @@ static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, (unsigned long long)blkno, i); if (ocfs2_xattr_is_local(xe)) { - if (le64_to_cpu(xe->xe_value_size) > - nv_limit - value_offset) + if (value_len > nv_limit - value_offset) return ocfs2_error(sb, "Invalid xattr in block %llu: entry %d value is out of bounds\n", (unsigned long long)blkno, @@ -1112,7 +1131,7 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, size_t entries_limit = sb->s_blocksize; size_t nv_limit = sb->s_blocksize; size_t max_entries; - int i; + int i, ret; if (region_size < sizeof(*xh)) return ocfs2_error(sb, @@ -1140,6 +1159,11 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, size_t block_off = name_offset >> sb->s_blocksize_bits; size_t block_offset = name_offset % nv_limit; size_t value_offset; + u64 value_len = le64_to_cpu(xe->xe_value_size); + + ret = ocfs2_validate_xattr_entry(sb, blkno, xe); + if (ret) + return ret; if (name_offset >= region_size || block_off >= bucket->bu_blocks) return ocfs2_error(sb, @@ -1158,8 +1182,7 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, (unsigned long long)blkno, i); if (ocfs2_xattr_is_local(xe)) { - if (le64_to_cpu(xe->xe_value_size) > - nv_limit - value_offset) + if (value_len > nv_limit - value_offset) return ocfs2_error(sb, "Invalid xattr bucket %llu: entry %d value is out of bounds\n", (unsigned long long)blkno, @@ -1307,7 +1330,7 @@ static int ocfs2_xattr_find_entry(struct inode *inode, int name_index, { struct ocfs2_xattr_entry *entry; size_t name_len; - int i, name_offset, cmp = 1; + int i, name_offset, cmp = 1, ret; if (name == NULL) return -EINVAL; @@ -1330,6 +1353,12 @@ static int ocfs2_xattr_find_entry(struct inode *inode, int name_index, return -EFSCORRUPTED; } cmp = memcmp(name, (xs->base + name_offset), name_len); + if (!cmp) { + ret = ocfs2_validate_xattr_entry(inode->i_sb, + OCFS2_I(inode)->ip_blkno, entry); + if (ret) + return ret; + } } if (cmp == 0) break; @@ -4071,6 +4100,10 @@ static int ocfs2_find_xe_in_bucket(struct inode *inode, xe_name = bucket_block(bucket, block_off) + new_offset; if (!memcmp(name, xe_name, name_len)) { + ret = ocfs2_validate_xattr_entry(inode->i_sb, + OCFS2_I(inode)->ip_blkno, xe); + if (ret) + break; *xe_index = i; *found = 1; ret = 0; @@ -4708,6 +4741,9 @@ static int ocfs2_defrag_xattr_bucket(struct inode *inode, xe = xh->xh_entries; end = OCFS2_XATTR_BUCKET_SIZE; for (i = 0; i < le16_to_cpu(xh->xh_count); i++, xe++) { + ret = ocfs2_validate_xattr_entry(inode->i_sb, blkno, xe); + if (ret) + goto out; offset = le16_to_cpu(xe->xe_name_offset); len = namevalue_size_xe(xe); @@ -4990,6 +5026,9 @@ static int ocfs2_divide_xattr_bucket(struct inode *inode, name_value_len = 0; for (i = 0; i < start; i++) { xe = &xh->xh_entries[i]; + ret = ocfs2_validate_xattr_entry(inode->i_sb, blk, xe); + if (ret) + goto out; name_value_len += namevalue_size_xe(xe); if (le16_to_cpu(xe->xe_name_offset) < name_offset) name_offset = le16_to_cpu(xe->xe_name_offset); From 91a20190a27f1698cc54005d843ac8e977120048 Mon Sep 17 00:00:00 2001 From: Geert Uytterhoeven Date: Mon, 24 Aug 2026 17:14:01 +0200 Subject: [PATCH 742/857] raid/kunit: enable RAID6 PQ and XOR benchmarks if KUNIT_ALL_TESTS=m Enabling the (possibly long-running benchmarks) by default may cause a big delay in boot time in case of built-in tests. However, they can still safely be enabled by default if all tests are modular, as they would only run when requested explicitly by the system administrator. Link: https://lore.kernel.org/64c6e0191bd8ccef0074ffbb09bd0584680d710b.1787584360.git.geert@linux-m68k.org Signed-off-by: Geert Uytterhoeven Reviewed-by: Christoph Hellwig Acked-by: Ard Biesheuvel Cc: Eric Biggers Signed-off-by: Andrew Morton --- lib/raid/Kconfig | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/raid/Kconfig b/lib/raid/Kconfig index 01f007b2522cfc..563a178aa930c4 100644 --- a/lib/raid/Kconfig +++ b/lib/raid/Kconfig @@ -32,6 +32,7 @@ config XOR_KUNIT_TEST config XOR_BENCHMARK bool "Benchmark for xor_gen" depends on XOR_KUNIT_TEST + default y if KUNIT_ALL_TESTS=m help Include benchmarks in the KUnit test suite for xor_gen. @@ -63,6 +64,7 @@ config RAID6_PQ_KUNIT_TEST config RAID6_PQ_KUNIT_BENCHMARK bool "Benchmark for RAID6 PQ" depends on RAID6_PQ_KUNIT_TEST + default y if KUNIT_ALL_TESTS=m help Include benchmarks in the KUnit test suite for raid P/Q generation. From a978ccbfced8cbcdbb6d749aa28642ec4969e200 Mon Sep 17 00:00:00 2001 From: Daehyeon Ko <4ncienth@gmail.com> Date: Mon, 24 Aug 2026 13:23:59 +0900 Subject: [PATCH 743/857] ipc/mqueue: release notification resources during inode eviction mqueue_flush_file() removes an mq_notify() registration only when the closing task belongs to the thread group stored in notify_owner. A task in a separate thread group created with CLONE_FILES can register SIGEV_THREAD notification and exit without closing the shared file table. If another thread group then unlinks and last-closes the queue, ->flush() skips the registration and inode eviction loses the only pointers to its resources. The orphaned registration permanently retains the notification skb, its netlink socket, a pid reference and a user namespace reference. An unprivileged process can repeat the sequence with new queues and sockets. No inode users remain during eviction. Remove any stale registration there after dropping info->lock, since netlink_sendskb() may release the final socket reference. This bug creates an unkillable kernel resource leak by failing to free netlink socket, PID, and user namespace references when a POSIX message queue is evicted. An unprivileged process can exploit this leak repeatedly to cause kernel memory exhaustion and lead to a DoS. Link: https://lore.kernel.org/20260824042359.925145-1-4ncienth@gmail.com Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Assisted-by: LLM Signed-off-by: Daehyeon Ko <4ncienth@gmail.com> Cc: Davidlohr Bueso Cc: Al Viro Cc: Christian Brauner Cc: Jan Kara Cc: Manfred Spraul Cc: NeilBrown Cc: Signed-off-by: Andrew Morton --- ipc/mqueue.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/ipc/mqueue.c b/ipc/mqueue.c index d1dd36a651b0d0..d1a1965c98118c 100644 --- a/ipc/mqueue.c +++ b/ipc/mqueue.c @@ -528,6 +528,13 @@ static void mqueue_evict_inode(struct inode *inode) list_add_tail(&msg->m_list, &tmp_msg); kfree(info->node_cache); spin_unlock(&info->lock); + /* + * A shared file table can let the notification owner exit without + * running ->flush(). No users of the inode remain during eviction, so + * tear down any stale notification after dropping info->lock because + * netlink_sendskb() may release the final socket reference. + */ + remove_notification(info); list_for_each_entry_safe(msg, nmsg, &tmp_msg, m_list) { list_del(&msg->m_list); From 8145894b66574123e589168d4c89b8e63ba81054 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Thu, 27 Aug 2026 06:17:08 +0200 Subject: [PATCH 744/857] klist: avoid accesses after waking klist_remove() klist_remove() waits until a node is unreferenced so that its caller can free the containing object. klist_release() currently publishes waiter->woken and wakes the waiter before its final accesses to the waiter and node. klist_remove() can then return, allowing its stack waiter and the containing object to be freed or reused while klist_release() is still running. In particular, bus_remove_driver() can free drv->p while __device_attach() walks the same bus klist with bus_for_each_drv(). On an arm64 Cortex-A72 system, an unpatched 7.2.0-rc3 kernel with CONFIG_PREEMPT_RT=y and CONFIG_KASAN=y reproduced the bug through the in-tree I2C/at24 path. KASAN reported a use-after-free in klist_dec_and_del() reached from klist_next()/bus_for_each_drv() while at24 was being unregistered. Clear n_klist and take a task reference before publishing woken. Use release/acquire accesses for that publication and wake the referenced task. The task reference keeps the waiter task alive if it returns and exits before wake_up_process(). Link: https://lore.kernel.org/20260827041708.31682-1-kmehltretter@gmail.com Fixes: 8b0c250be489 ("[PATCH] add klist_node_attached() to determine if a node is on a list or not.") Fixes: 210272a28465 ("driver core: Remove completion from struct klist_node") Assisted-by: LLM Signed-off-by: Karl Mehltretter Cc: Danilo Krummrich Cc: Greg Kroah-Hartman Cc: Matthew Wilcox (Oracle) Cc: "Rafael J. Wysocki" Cc: Signed-off-by: Andrew Morton --- lib/klist.c | 16 ++++++++++++---- 1 file changed, 12 insertions(+), 4 deletions(-) diff --git a/lib/klist.c b/lib/klist.c index 332a4fbf18ff08..f133740b1c2cb7 100644 --- a/lib/klist.c +++ b/lib/klist.c @@ -36,6 +36,7 @@ #include #include #include +#include /* * Use the lowest bit of n_klist to mark deleted nodes and exclude @@ -187,18 +188,24 @@ static void klist_release(struct kref *kref) WARN_ON(!knode_dead(n)); list_del(&n->n_node); + knode_set_klist(n, NULL); spin_lock(&klist_remove_lock); list_for_each_entry_safe(waiter, tmp, &klist_remove_waiters, list) { + struct task_struct *p; + if (waiter->node != n) continue; + p = waiter->process; + get_task_struct(p); list_del(&waiter->list); - waiter->woken = 1; + /* Publish only after the final waiter and n accesses */ + smp_store_release(&waiter->woken, 1); mb(); - wake_up_process(waiter->process); + wake_up_process(p); + put_task_struct(p); } spin_unlock(&klist_remove_lock); - knode_set_klist(n, NULL); } static int klist_dec_and_del(struct klist_node *n) @@ -250,7 +257,8 @@ void klist_remove(struct klist_node *n) for (;;) { set_current_state(TASK_UNINTERRUPTIBLE); - if (waiter.woken) + /* Pairs with the release store in klist_release() */ + if (smp_load_acquire(&waiter.woken)) break; schedule(); } From 881e42df44ee90433db50dee6dee5620f231c293 Mon Sep 17 00:00:00 2001 From: Vishal Badole Date: Wed, 26 Aug 2026 22:45:37 +0530 Subject: [PATCH 745/857] lib/group_cpus: snapshot cluster masks to keep grouping hotplug invariant group_cpus_evenly() builds the managed-IRQ affinity spread used by multi-queue devices such as NVMe. That spread is meant to be a property of the static CPU topology: it walks cpu_present_mask and then cpu_possible_mask so every hardware queue owns a fixed set of CPUs, including CPUs that are offline at the time. A driver depends on that partition staying stable across re-computation - the CPUs a queue is given at probe must still describe the same queue after the device is later reset and its affinity recomputed. On an AMD system that stability breaks across an s2idle cycle. With CPUs 3-11 offlined and only CPUs 0-2 left online, the machine is suspended to s2idle and resumed. The NVMe controller uses the simple-suspend quirk, so resume fully re-initialises it and recomputes the affinity spread. The system then hangs for roughly two minutes and stays sluggish afterwards, the controller only making progress through its command-timeout poll: nvme nvme0: I/O tag 898 (3382) QID 9 timeout, completion polled nvme nvme0: I/O tag 398 (618e) QID 11 timeout, completion polled QID 9 and QID 11 are the queues whose CPUs were offline when the spread was recomputed. "completion polled" means the commands did finish in hardware, but their interrupts were never delivered to a CPU that was watching the queue, so nothing reaped them until the timeout fired. It happens because commit 89802ca36c96 ("lib/group_cpus: make group CPU cluster aware") derives the cluster groups from topology_cluster_cpumask(), which lists only the cluster siblings that are online when it is called. The resulting partition therefore depends on the transient online mask rather than on the topology alone. Recomputed on resume while the non-boot CPUs are still offline, it no longer matches the boot-time partition, and a queue is left with an affinity that does not cover the CPU it is meant to serve once that CPU comes back online. The dependence is on the online mask, not on any AMD-specific behaviour, so the same stall is reproducible on Intel platforms as well. Make the cluster grouping depend on the complete cluster topology rather than on whichever CPUs happen to be online. Snapshot the cluster masks once while every present CPU is online and reuse that view for every later spread. Every spread then groups from the same masks, so the partition computed when the controller is reset matches the one computed at probe and each queue's IRQ still covers the CPUs it serves. If the snapshot was never taken, the cluster path is skipped and the plain present/possible spread is used. Link: https://lore.kernel.org/20260826171537.4167367-1-Vishal.Badole@amd.com Fixes: 89802ca36c96 ("lib/group_cpus: make group CPU cluster aware") Signed-off-by: Vishal Badole Cc: "Borislav Petkov (AMD)" Cc: Radu Rendec Cc: Thomas Gleixner Cc: Tim Chen Cc: Wangyang Guo Cc: Tianyou Li Cc: Dan Liang Cc: Signed-off-by: Andrew Morton --- lib/group_cpus.c | 87 ++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 85 insertions(+), 2 deletions(-) diff --git a/lib/group_cpus.c b/lib/group_cpus.c index e6e18d7a49bba6..3c2229feb9c3d2 100644 --- a/lib/group_cpus.c +++ b/lib/group_cpus.c @@ -6,6 +6,7 @@ #include #include #include +#include #include #include @@ -286,6 +287,77 @@ static void assign_cpus_to_groups(unsigned int ncpus, } } +/* + * topology_cluster_cpumask() only lists the cluster siblings that are online, + * so group_cpus_evenly() would compute a different managed-IRQ partition when + * recomputed with CPUs offline (e.g. an NVMe reset across s2idle), steering a + * queue's IRQ away from the CPU it serves. + * + * Snapshot the cluster masks once, on the first spread seen with every + * present CPU online, and reuse it so the grouping stays stable. If no + * snapshot exists (partial boot via maxcpus=/nosmp, or allocation failure) + * the cluster path is skipped and the plain present/possible spread is + * used. Only the cluster path is stabilised; grp_spread_init_one()'s + * sibling mask is unchanged. The snapshot lives for the system lifetime + * and is not refreshed for CPUs hot-added after boot. + */ +static cpumask_var_t *cluster_snapshot; +static bool cluster_snapshot_ready; +static DEFINE_MUTEX(cluster_snapshot_lock); + +static void capture_cluster_snapshot(void) +{ + cpumask_var_t *snapshot; + unsigned int cpu; + + /* Pairs with the smp_store_release() below. */ + if (smp_load_acquire(&cluster_snapshot_ready)) + return; + + /* Only capture when all present CPUs are online. */ + if (!data_race(cpumask_equal(cpu_present_mask, cpu_online_mask))) + return; + + mutex_lock(&cluster_snapshot_lock); + if (cluster_snapshot_ready) + goto out; + + snapshot = kcalloc(nr_cpu_ids, sizeof(*snapshot), GFP_KERNEL); + if (!snapshot) + goto out; + + for_each_possible_cpu(cpu) + if (!zalloc_cpumask_var(&snapshot[cpu], GFP_KERNEL)) + goto free_snapshot; + + /* Trylock: a caller may hold a lock the hotplug writer needs. */ + if (!cpus_read_trylock()) + goto free_snapshot; + + /* Recheck under the lock, which also pins the cluster masks. */ + if (!data_race(cpumask_equal(cpu_present_mask, cpu_online_mask))) { + cpus_read_unlock(); + goto free_snapshot; + } + + for_each_possible_cpu(cpu) + cpumask_copy(snapshot[cpu], topology_cluster_cpumask(cpu)); + cpus_read_unlock(); + + cluster_snapshot = snapshot; + /* Publish the filled snapshot before the ready flag. */ + smp_store_release(&cluster_snapshot_ready, true); + goto out; + +free_snapshot: + /* Unallocated entries are NULL, which free_cpumask_var() ignores. */ + for_each_possible_cpu(cpu) + free_cpumask_var(snapshot[cpu]); + kfree(snapshot); +out: + mutex_unlock(&cluster_snapshot_lock); +} + static int alloc_cluster_groups(unsigned int ncpus, unsigned int ngroups, struct cpumask *node_cpumask, @@ -299,6 +371,17 @@ static int alloc_cluster_groups(unsigned int ncpus, const struct cpumask **clusters; struct node_groups *cluster_groups; + /* + * Capture on the first spread with every present CPU online (normally + * the first device probe); later spreads reuse it. Sample the ready + * flag once so both loops below use one consistent source. + */ + capture_cluster_snapshot(); + + /* Pairs with the smp_store_release() in capture_cluster_snapshot(). */ + if (!smp_load_acquire(&cluster_snapshot_ready)) + goto no_cluster; + cpumask_copy(msk, node_cpumask); /* Probe how many clusters in this node. */ @@ -307,7 +390,7 @@ static int alloc_cluster_groups(unsigned int ncpus, if (cpu >= nr_cpu_ids) break; - cluster_mask = topology_cluster_cpumask(cpu); + cluster_mask = cluster_snapshot[cpu]; if (!cpumask_weight(cluster_mask)) goto no_cluster; /* Clean out CPUs on the same cluster. */ @@ -331,7 +414,7 @@ static int alloc_cluster_groups(unsigned int ncpus, cpumask_copy(msk, node_cpumask); for (n = 0; n < ncluster; n++) { cpu = cpumask_first(msk); - cluster_mask = topology_cluster_cpumask(cpu); + cluster_mask = cluster_snapshot[cpu]; nc = cpumask_weight_and(cluster_mask, node_cpumask); clusters[n] = cluster_mask; cluster_groups[n].id = n; From 76af7376018a2a834dbaaecb43261ba143576e0a Mon Sep 17 00:00:00 2001 From: Chris Gellermann Date: Mon, 3 Aug 2026 14:48:59 +0200 Subject: [PATCH 746/857] selftests/membarrier: introduce helper to get membarrier command registrations Patch series "selftests/membarrier: Skip an unregistered memory barrier test on Musl". The membarrier test "membarrier MEMBARRIER_CMD_PRIVATE_EXPEDITED not registered failure" fails in the multithreaded test scenario when using Musl libc as the command gets preregistered implicitly during thread creation. Skip the test if command registration is detected. This patch (of 2): Add a new membarrier_get_registrations() for reusage. Link: https://lore.kernel.org/20260803124900.3328789-1-christian.gellermann@codasip.com Link: https://lore.kernel.org/20260803124900.3328789-2-christian.gellermann@codasip.com Signed-off-by: Chris Gellermann Tested-by: Michael Jeanson Cc: Ben Segall Cc: Dietmar Eggemann Cc: Ingo Molnar Cc: Juri Lelli Cc: K Prateek Nayak Cc: Mathieu Desnoyers Cc: Mel Gorman Cc: "Paul E . McKenney" Cc: Peter Zijlstra Cc: Shuah Khan Cc: Steven Rostedt Cc: Valentin Schneider Cc: Vincent Guittot Cc: Wei Yang Signed-off-by: Andrew Morton --- tools/testing/selftests/membarrier/membarrier_test_impl.h | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/membarrier/membarrier_test_impl.h b/tools/testing/selftests/membarrier/membarrier_test_impl.h index f6d7c44b2288da..29aac3bc498871 100644 --- a/tools/testing/selftests/membarrier/membarrier_test_impl.h +++ b/tools/testing/selftests/membarrier/membarrier_test_impl.h @@ -16,6 +16,11 @@ static int sys_membarrier(int cmd, int flags) return syscall(__NR_membarrier, cmd, flags); } +static int membarrier_get_registrations(void) +{ + return sys_membarrier(MEMBARRIER_CMD_GET_REGISTRATIONS, 0); +} + static int test_membarrier_get_registrations(int cmd) { int ret, flags = 0; @@ -24,7 +29,7 @@ static int test_membarrier_get_registrations(int cmd) registrations |= cmd; - ret = sys_membarrier(MEMBARRIER_CMD_GET_REGISTRATIONS, 0); + ret = membarrier_get_registrations(); if (ret < 0) { ksft_exit_fail_msg( "%s test: flags = %d, errno = %d\n", From d8ce1b35606d9d2e66a03be85ad7e841cb61272e Mon Sep 17 00:00:00 2001 From: Chris Gellermann Date: Mon, 3 Aug 2026 14:49:00 +0200 Subject: [PATCH 747/857] selftests/membarrier: skip unpermitted membarrier command test if preregistered by libc On thread creation, Musl registers the private expedited memory barrier, see pthread_create [1]. Thus, invoking the barrier command will no longer be rejected by the kernel with EPERM. The test checking this will fail. Check if the memory barrier command has been registered and skip the test in this case. Link: https://git.musl-libc.org/cgit/musl/tree/src/thread/pthread_create.c#n260 [1] Link: https://lore.kernel.org/20260803124900.3328789-3-christian.gellermann@codasip.com Signed-off-by: Chris Gellermann Tested-by: Michael Jeanson Cc: Ben Segall Cc: Dietmar Eggemann Cc: Ingo Molnar Cc: Juri Lelli Cc: K Prateek Nayak Cc: Mathieu Desnoyers Cc: Mel Gorman Cc: "Paul E . McKenney" Cc: Peter Zijlstra Cc: Shuah Khan Cc: Steven Rostedt Cc: Valentin Schneider Cc: Vincent Guittot Cc: Wei Yang Signed-off-by: Andrew Morton --- .../selftests/membarrier/membarrier_test_impl.h | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tools/testing/selftests/membarrier/membarrier_test_impl.h b/tools/testing/selftests/membarrier/membarrier_test_impl.h index 29aac3bc498871..b4dcbb32538d47 100644 --- a/tools/testing/selftests/membarrier/membarrier_test_impl.h +++ b/tools/testing/selftests/membarrier/membarrier_test_impl.h @@ -113,6 +113,16 @@ static int test_membarrier_private_expedited_fail(void) int cmd = MEMBARRIER_CMD_PRIVATE_EXPEDITED, flags = 0; const char *test_name = "sys membarrier MEMBARRIER_CMD_PRIVATE_EXPEDITED not registered failure"; + /* + * Some C libraries, like Musl, register the private expedited barrier + * command when creating a thread. Expecting an EPERM on an unregistered + * command will therefore no longer work. Skip the test in this case. + */ + if (MEMBARRIER_CMD_REGISTER_PRIVATE_EXPEDITED & membarrier_get_registrations()) { + ksft_test_result_skip("%s test: Command already registered\n", test_name); + return 0; + } + if (sys_membarrier(cmd, flags) != -1) { ksft_exit_fail_msg( "%s test: flags = %d. Should fail, but passed\n", From 9f6a8fcfbe16fadefef32fac684349e23c78e9e1 Mon Sep 17 00:00:00 2001 From: Florian Schmaus Date: Fri, 28 Aug 2026 17:54:07 +0200 Subject: [PATCH 748/857] selftests/epoll: fix race condition in multi-waiter wakeup tests In tests with multiple concurrent waiters on edge-triggered epoll instances where an emitter writes to multiple sockets (epoll16, epoll56, epoll58): When the emitter performs its first write(), ep_poll_callback() fires and wakes up both waiters because one waiter uses epoll_wait() and the other one uses poll(). This translates to different wait queues, ep->wq for epoll and ep->poll_wait for poll/select, which are both awoken by the kernel because of that single write. Next, both waiter threads invoke epoll_wait(), but since there is only one event, only one epoll_wait() will return non-zero because of the edge-triggered mode being used (in level-triggered mode, the kernel would re-queue the event because of remaining unread data). Since the second waiter sees an empty ready list, it does not increment ctx.count and the test fails spuriously with ctx.count == 1 instead of 2. Emitter (CPU 0) Thread 0 (CPU 1) Thread 1 (CPU 2) =============== ================ ================ epoll_wait(e0, -1) poll(e0, -1) [on e0->wq] [on e0->poll_wait] write(sfd[1]) | +--(Kernel wakes BOTH e0->wq and e0->poll_wait via callback)--+ | | | wakes up wakes up | | epoll_wait() reaps e1 poll() returns 1 | | (e1 removed via ET) (wants event) | | e0->rdllist is EMPTY | | | count++ (count = 1) v | | epoll_wait(e0, 0) | | sees EMPTY list! | | returns 0! | | thread exits | v | write(sfd[3]) | (event arrives too late!) v EXPECT_EQ(count, 2) <-- SPURIOUS FAILURE! Introduce waiter_entry1ap_loop() to retry poll() if the initial epoll_wait(..., 0) yielded no events. This ensures the thread waits for the subsequent write rather than failing immediately. Apply this helper in epoll16, epoll56, and for both waiter threads in epoll58. Link: https://lore.kernel.org/20260828-selftest-epoll-fix-race-v2-1-953ab57fd60a@codasip.com Fixes: f2728fe80cef ("selftests: add epoll selftests") Signed-off-by: Florian Schmaus Cc: Heiher Cc: Roman Penyaev Cc: Shuah Khan Cc: Christian Brauner Signed-off-by: Andrew Morton --- .../filesystems/epoll/epoll_wakeup_test.c | 32 +++++++++++++------ 1 file changed, 22 insertions(+), 10 deletions(-) diff --git a/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c b/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c index 81a994943e121b..b4dcbcd79773a7 100644 --- a/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c +++ b/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c @@ -74,6 +74,24 @@ static void *waiter_entry1ap(void *data) return NULL; } +static void *waiter_entry1ap_loop(void *data) +{ + struct pollfd pfd; + struct epoll_event e; + struct epoll_mtcontext *ctx = data; + + pfd.fd = ctx->efd[0]; + pfd.events = POLLIN; + while (poll(&pfd, 1, 2000) > 0) { + if (epoll_wait(ctx->efd[0], &e, 1, 0) > 0) { + __sync_fetch_and_add(&ctx->count, 1); + break; + } + } + + return NULL; +} + static void *waiter_entry1o(void *data) { struct epoll_event e; @@ -809,7 +827,7 @@ TEST(epoll16) ASSERT_EQ(epoll_ctl(ctx.efd[0], EPOLL_CTL_ADD, ctx.sfd[2], events), 0); ctx.main = pthread_self(); - ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap, &ctx), 0); + ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap_loop, &ctx), 0); ASSERT_EQ(pthread_create(&emitter, NULL, emitter_entry2, &ctx), 0); if (epoll_wait(ctx.efd[0], events, 1, -1) > 0) @@ -2925,7 +2943,7 @@ TEST(epoll56) ASSERT_EQ(epoll_ctl(ctx.efd[0], EPOLL_CTL_ADD, ctx.efd[2], &e), 0); ctx.main = pthread_self(); - ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap, &ctx), 0); + ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap_loop, &ctx), 0); ASSERT_EQ(pthread_create(&emitter, NULL, emitter_entry2, &ctx), 0); if (epoll_wait(ctx.efd[0], &e, 1, -1) > 0) @@ -3030,7 +3048,6 @@ TEST(epoll57) TEST(epoll58) { pthread_t emitter; - struct pollfd pfd; struct epoll_event e; struct epoll_mtcontext ctx = { 0 }; @@ -3061,15 +3078,10 @@ TEST(epoll58) ASSERT_EQ(epoll_ctl(ctx.efd[0], EPOLL_CTL_ADD, ctx.efd[2], &e), 0); ctx.main = pthread_self(); - ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap, &ctx), 0); + ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap_loop, &ctx), 0); ASSERT_EQ(pthread_create(&emitter, NULL, emitter_entry2, &ctx), 0); - pfd.fd = ctx.efd[0]; - pfd.events = POLLIN; - if (poll(&pfd, 1, -1) > 0) { - if (epoll_wait(ctx.efd[0], &e, 1, 0) > 0) - __sync_fetch_and_add(&ctx.count, 1); - } + waiter_entry1ap_loop(&ctx); ASSERT_EQ(pthread_join(ctx.waiter, NULL), 0); EXPECT_EQ(ctx.count, 2); From d3c1859891b28954f0ed05a02538c3c40dd2935c Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Fri, 28 Aug 2026 19:28:23 +0800 Subject: [PATCH 749/857] ocfs2: exit recovery thread on mount error path When a mount fails after the cluster connection has been established, e.g. in ocfs2_mount_volume(), ocfs2_fill_super() unwinds via out_debugfs/out_super and frees the osb without disabling recovery. A node failure event can concurrently launch the recovery thread, which blocks in __ocfs2_wait_on_mount() waiting for the volume state to become VOLUME_MOUNTED or VOLUME_DISABLED. As the mount error path neither sets VOLUME_DISABLED nor wakes osb_mount_event, the thread can never make progress: the kthread leaks and stays blocked on the wait queue embedded in the freed osb, which may then be accessed as freed memory. Fix it by setting VOLUME_DISABLED and waking osb_mount_event on this path so the thread bails out, and replace the plain kfree(osb->recovery_map) with ocfs2_recovery_exit(), which waits for a running recovery thread to exit before the recovery map is freed. Link: https://lore.kernel.org/20260828112825.666097-1-joseph.qi@linux.alibaba.com Fixes: f1e75d128b46 ("ocfs2: rewrite error handling of ocfs2_fill_super") Signed-off-by: Joseph Qi Reviewed-by: Heming Zhao Cc: Changwei Ge Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Cc: Signed-off-by: Andrew Morton --- fs/ocfs2/super.c | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c index c62e389d4dd659..f785c39d1fb8a5 100644 --- a/fs/ocfs2/super.c +++ b/fs/ocfs2/super.c @@ -1169,8 +1169,17 @@ static int ocfs2_fill_super(struct super_block *sb, struct fs_context *fc) out_debugfs: debugfs_remove_recursive(osb->osb_debug_root); out_super: + /* + * A recovery thread launched by a node failure event may still be + * waiting for the volume to be mounted. Set VOLUME_DISABLED and + * wake it up, then wait for it to exit before osb is freed, + * otherwise the kthread would leak and stay blocked on the wait + * queue embedded in the freed osb. + */ + atomic_set(&osb->vol_state, VOLUME_DISABLED); + wake_up(&osb->osb_mount_event); ocfs2_release_system_inodes(osb); - kfree(osb->recovery_map); + ocfs2_recovery_exit(osb); ocfs2_delete_osb(osb); kfree(osb); out: From e2894afec320b3d5d6b17e432adc94b873ad2c87 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Fri, 28 Aug 2026 19:28:24 +0800 Subject: [PATCH 750/857] ocfs2: free replay slots in ocfs2_recovery_exit() Commit ce2fcf1516d6 ("ocfs2: fix memory leak in ocfs2_mount_volume()") added ocfs2_free_replay_slots() calls to the mount error paths out_dismount and out_check_volume to fix a leak of osb->replay_map. However these calls are unlocked while the bail path of the recovery thread, which is woken up by out_dismount right before the call, also frees the replay slots under osb->recovery_lock. Both sides can thus observe a non-NULL osb->replay_map and trigger a double free. Fix this by moving ocfs2_free_replay_slots() into ocfs2_recovery_exit() after ocfs2_recovery_disable(), which waits for a running recovery thread to exit under osb->recovery_lock, and drop the unlocked call sites. Both ocfs2_dismount_volume() and the out_super path of ocfs2_fill_super() call ocfs2_recovery_exit(), so the replay slots are freed on every path. Since super.c no longer references it, make ocfs2_free_replay_slots() static again. Link: https://lore.kernel.org/20260828112825.666097-2-joseph.qi@linux.alibaba.com Fixes: ce2fcf1516d6 ("ocfs2: fix memory leak in ocfs2_mount_volume()") Signed-off-by: Joseph Qi Reviewed-by: Heming Zhao Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Cc: Signed-off-by: Andrew Morton --- fs/ocfs2/journal.c | 3 ++- fs/ocfs2/journal.h | 1 - fs/ocfs2/super.c | 5 +---- 3 files changed, 3 insertions(+), 6 deletions(-) diff --git a/fs/ocfs2/journal.c b/fs/ocfs2/journal.c index d8afbc1a76bb8a..a3938a03e93bf7 100644 --- a/fs/ocfs2/journal.c +++ b/fs/ocfs2/journal.c @@ -156,7 +156,7 @@ static void ocfs2_queue_replay_slots(struct ocfs2_super *osb, replay_map->rm_state = REPLAY_DONE; } -void ocfs2_free_replay_slots(struct ocfs2_super *osb) +static void ocfs2_free_replay_slots(struct ocfs2_super *osb) { struct ocfs2_replay_map *replay_map = osb->replay_map; @@ -243,6 +243,7 @@ void ocfs2_recovery_exit(struct ocfs2_super *osb) /* XXX: Should we bug if there are dirty entries? */ kfree(rm); + ocfs2_free_replay_slots(osb); } static int __ocfs2_recovery_map_test(struct ocfs2_super *osb, diff --git a/fs/ocfs2/journal.h b/fs/ocfs2/journal.h index f8b3b2a3d6309e..19fc920d26b1cd 100644 --- a/fs/ocfs2/journal.h +++ b/fs/ocfs2/journal.h @@ -151,7 +151,6 @@ void ocfs2_recovery_exit(struct ocfs2_super *osb); void ocfs2_recovery_disable_quota(struct ocfs2_super *osb); int ocfs2_compute_replay_slots(struct ocfs2_super *osb); -void ocfs2_free_replay_slots(struct ocfs2_super *osb); /* * Journal Control: * Initialize, Load, Shutdown, Wipe a journal. diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c index f785c39d1fb8a5..6a8092b65bb558 100644 --- a/fs/ocfs2/super.c +++ b/fs/ocfs2/super.c @@ -1162,7 +1162,6 @@ static int ocfs2_fill_super(struct super_block *sb, struct fs_context *fc) out_dismount: atomic_set(&osb->vol_state, VOLUME_DISABLED); wake_up(&osb->osb_mount_event); - ocfs2_free_replay_slots(osb); ocfs2_dismount_volume(sb, 1); goto out; @@ -1776,14 +1775,12 @@ static int ocfs2_mount_volume(struct super_block *sb) status = ocfs2_truncate_log_init(osb); if (status < 0) { mlog_errno(status); - goto out_check_volume; + goto out_system_inodes; } ocfs2_super_unlock(osb, 1); return 0; -out_check_volume: - ocfs2_free_replay_slots(osb); out_system_inodes: if (osb->local_alloc_state == OCFS2_LA_ENABLED) ocfs2_shutdown_local_alloc(osb); From 107451a4d73242a4b01961632239c77acbfeea12 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Fri, 28 Aug 2026 19:28:25 +0800 Subject: [PATCH 751/857] ocfs2: defer suballocator block group reclaim to workqueue When the last bit in a suballocator block group is freed, _ocfs2_free_suballoc_bits() reclaims the group back to the global bitmap. The reclaim takes inode_lock() on the global bitmap inode while running inside the freeing transaction, adding a lock dependency of j_trans_barrier -> global bitmap inode i_rwsem This forms a circular dependency with paths such as ocfs2_shutdown_local_alloc(), which take the global bitmap inode lock before starting a transaction: Task1 (dealloc): ocfs2_run_deallocs ocfs2_free_cached_blocks ocfs2_start_trans down_read(j_trans_barrier) _ocfs2_free_suballoc_bits _ocfs2_reclaim_suballoc_to_main inode_lock(main_bm_inode) <- wait on Task2 Task2 (dismount): ocfs2_shutdown_local_alloc inode_lock(main_bm_inode) ocfs2_start_trans down_read(j_trans_barrier) <- wait on Task3 Task3 (ocfs2cmt): ocfs2_commit_cache down_write(j_trans_barrier) <- wait on Task1's handle jbd2_journal_flush Task1 waits for Task2's inode_lock(), Task2 waits for the j_trans_barrier down_write() held by ocfs2cmt, and ocfs2cmt waits for Task1's running transaction to commit - a real deadlock, observed with aio-stress direct IO writes racing dismount. Fix it by deferring the reclaim to the per-superblock ocfs2_wq workqueue, so the freeing transaction no longer takes the global bitmap inode lock. The worker re-checks under the suballocator locks that the block group is still fully freed (it may have been allocated from again in the meantime), takes the global bitmap inode locks before starting its own transaction, and performs the same suballocator cleanup and space return. The inode lock order (suballocator inode -> global bitmap inode) is consistent with the existing "inode lock before transaction" order, breaking the cycle. Reclaim work can still be queued late in dismount, e.g. when the truncate log is flushed or orphan dir recovery frees inode bits, so both ocfs2_dismount_volume() and the mount error path flush ocfs2_wq right before the system inodes are released, while the journal is still alive, to make sure no reclaim work is left running. The worker also bails out if the journal is already gone. Tested with the ocfs2 testsuite (including aio-stress direct IO) and umount/mount cycles on a CONFIG_PROVE_LOCKING kernel: the circular locking dependency is gone and freed block groups are still returned to the global bitmap. Link: https://lore.kernel.org/20260828112825.666097-3-joseph.qi@linux.alibaba.com Fixes: 4a54331616b3 ("ocfs2: give ocfs2 the ability to reclaim suballocator free bg") Signed-off-by: Joseph Qi Reviewed-by: Heming Zhao Assisted-by: Qoder:Qwen3.8-Max Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao Signed-off-by: Andrew Morton --- fs/ocfs2/ocfs2.h | 5 ++ fs/ocfs2/suballoc.c | 202 +++++++++++++++++++++++++++++++++++++------- fs/ocfs2/suballoc.h | 2 +- fs/ocfs2/super.c | 16 ++++ 4 files changed, 195 insertions(+), 30 deletions(-) diff --git a/fs/ocfs2/ocfs2.h b/fs/ocfs2/ocfs2.h index 62cad6522c7a31..b747cdec178758 100644 --- a/fs/ocfs2/ocfs2.h +++ b/fs/ocfs2/ocfs2.h @@ -502,6 +502,11 @@ struct ocfs2_super */ struct workqueue_struct *ocfs2_wq; + /* deferred reclaim of fully freed suballocator block groups */ + spinlock_t os_suballoc_reclaim_lock; + struct list_head os_suballoc_reclaim_list; + struct work_struct os_suballoc_reclaim_work; + /* sysfs directory per partition */ struct kset *osb_dev_kset; diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index 20c3aec6b9873c..453b56be9624c6 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -2687,16 +2687,24 @@ static int ocfs2_block_group_clear_bits(handle_t *handle, * cleanup rec/alloc_inode job, then switches to the main bitmap * to reclaim released space. * + * Callers must hold inode_lock() and ocfs2_inode_lock() on + * main_bm_inode, i.e. the global bitmap inode locks must be taken + * before starting the transaction. + * * handle: The transaction handle * alloc_inode: The suballoc inode * alloc_bh: The buffer_head of suballoc inode * group_bh: The group descriptor buffer_head of suballocator managed. - * Caller should release the input group_bh. + * This function takes ownership of it and will release it. + * main_bm_inode: The global bitmap inode + * main_bm_bh: The buffer_head of the global bitmap inode */ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, struct inode *alloc_inode, struct buffer_head *alloc_bh, - struct buffer_head *group_bh) + struct buffer_head *group_bh, + struct inode *main_bm_inode, + struct buffer_head *main_bm_bh) { int idx, status = 0; int i, next_free_rec, len = 0; @@ -2706,8 +2714,6 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, u64 bg_blkno, start_blk; unsigned int count; struct ocfs2_chain_rec *rec; - struct buffer_head *main_bm_bh = NULL; - struct inode *main_bm_inode = NULL; struct ocfs2_super *osb = OCFS2_SB(alloc_inode->i_sb); struct ocfs2_dinode *fe = (struct ocfs2_dinode *) alloc_bh->b_data; struct ocfs2_chain_list *cl = &fe->id2.i_chain; @@ -2794,24 +2800,12 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, ocfs2_remove_from_cache(INODE_CACHE(alloc_inode), group_bh); memset(group, 0, sizeof(struct ocfs2_group_desc)); - /* prepare job for reclaim clusters */ - main_bm_inode = ocfs2_get_system_file_inode(osb, - GLOBAL_BITMAP_SYSTEM_INODE, - OCFS2_INVALID_SLOT); - if (!main_bm_inode) - goto bail; /* ignore the error in reclaim path */ - - inode_lock(main_bm_inode); - - status = ocfs2_inode_lock(main_bm_inode, &main_bm_bh, 1); - if (status < 0) - goto free_bm_inode; /* ignore the error in reclaim path */ - ocfs2_block_to_cluster_group(main_bm_inode, start_blk, &bg_blkno, &start_bit); fe = (struct ocfs2_dinode *) main_bm_bh->b_data; cl = &fe->id2.i_chain; - /* reuse group_bh, caller will release the input group_bh */ + /* release the suballocator group descriptor before reuse */ + brelse(group_bh); group_bh = NULL; /* reclaim clusters to global_bitmap */ @@ -2819,7 +2813,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, &group_bh); if (status < 0) { mlog_errno(status); - goto free_bm_bh; + goto bail; } group = (struct ocfs2_group_desc *) group_bh->b_data; @@ -2827,7 +2821,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, ocfs2_error(alloc_inode->i_sb, "reclaim length (%d) beyands block group length (%d)", count + start_bit, le16_to_cpu(group->bg_bits)); - goto free_group_bh; + goto bail; } old_bg_contig_free_bits = group->bg_contig_free_bits; @@ -2837,7 +2831,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, _ocfs2_clear_bit); if (status < 0) { mlog_errno(status); - goto free_group_bh; + goto bail; } status = ocfs2_journal_access_di(handle, INODE_CACHE(main_bm_inode), @@ -2847,7 +2841,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, ocfs2_block_group_set_bits(handle, main_bm_inode, group, group_bh, start_bit, count, le16_to_cpu(old_bg_contig_free_bits), 1); - goto free_group_bh; + goto bail; } idx = le16_to_cpu(group->bg_chain); @@ -2858,19 +2852,168 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, fe->id1.bitmap1.i_used = cpu_to_le32(tmp_used - count); ocfs2_journal_dirty(handle, main_bm_bh); -free_group_bh: +bail: brelse(group_bh); + return status; +} + +/* + * When a suballocator block group becomes fully freed, its space is + * reclaimed back to the global bitmap. Taking the global bitmap inode + * lock inside the freeing transaction would create a lock dependency + * of "j_trans_barrier -> global bitmap inode i_rwsem", which forms a + * circular dependency with paths like ocfs2_shutdown_local_alloc() that + * take the inode lock before starting a transaction, and can lead to a + * real deadlock with the ocfs2cmt journal commit thread. So queue the + * reclaim to the workqueue and let it run outside the freeing + * transaction. + */ +struct ocfs2_suballoc_reclaim_work { + struct list_head list; + struct inode *alloc_inode; + u64 bg_blkno; +}; + +static void ocfs2_queue_suballoc_reclaim(struct ocfs2_super *osb, + struct inode *alloc_inode, + u64 bg_blkno) +{ + struct ocfs2_suballoc_reclaim_work *reclaim_work; + + reclaim_work = kmalloc_obj(*reclaim_work, GFP_NOFS); + if (!reclaim_work) { + /* + * Reclaim is only a space return optimization. If we can't + * queue it, the freed block group just stays owned by the + * suballocator. + */ + return; + } + + igrab(alloc_inode); + reclaim_work->alloc_inode = alloc_inode; + reclaim_work->bg_blkno = bg_blkno; + + spin_lock(&osb->os_suballoc_reclaim_lock); + list_add_tail(&reclaim_work->list, &osb->os_suballoc_reclaim_list); + spin_unlock(&osb->os_suballoc_reclaim_lock); -free_bm_bh: + queue_work(osb->ocfs2_wq, &osb->os_suballoc_reclaim_work); +} + +static void ocfs2_do_suballoc_reclaim(struct ocfs2_super *osb, + struct ocfs2_suballoc_reclaim_work *reclaim_work) +{ + int status, i; + handle_t *handle; + struct inode *alloc_inode = reclaim_work->alloc_inode; + struct inode *main_bm_inode; + struct buffer_head *alloc_bh = NULL, *group_bh = NULL; + struct buffer_head *main_bm_bh = NULL; + struct ocfs2_dinode *fe; + struct ocfs2_chain_list *cl; + struct ocfs2_chain_rec *rec; + + /* journal already gone, e.g. during dismount cleanup */ + if (!osb->journal) + return; + + inode_lock(alloc_inode); + status = ocfs2_inode_lock(alloc_inode, &alloc_bh, 1); + if (status < 0) + goto out_alloc; + + fe = (struct ocfs2_dinode *) alloc_bh->b_data; + cl = &fe->id2.i_chain; + + /* + * The block group may have been allocated from again since the + * reclaim work was queued, re-check that it is still fully freed. + * A stale work item can also reference a group that is no longer + * chained, whose descriptor would fail validation and trigger a + * spurious ocfs2_error(), so verify chain membership first. + */ + for (i = 0; i < le16_to_cpu(cl->cl_next_free_rec); i++) { + rec = &cl->cl_recs[i]; + if (le64_to_cpu(rec->c_blkno) == reclaim_work->bg_blkno) + break; + } + if (i == le16_to_cpu(cl->cl_next_free_rec) || + ocfs2_is_cluster_bitmap(alloc_inode) || + (le32_to_cpu(rec->c_free) != (le32_to_cpu(rec->c_total) - 1)) || + (le16_to_cpu(cl->cl_next_free_rec) == 1)) + goto out_alloc_unlock; + + status = ocfs2_read_group_descriptor(alloc_inode, fe, + reclaim_work->bg_blkno, &group_bh); + if (status < 0) + goto out_alloc_unlock; + + main_bm_inode = ocfs2_get_system_file_inode(osb, + GLOBAL_BITMAP_SYSTEM_INODE, + OCFS2_INVALID_SLOT); + if (!main_bm_inode) + goto out_group; + + inode_lock(main_bm_inode); + status = ocfs2_inode_lock(main_bm_inode, &main_bm_bh, 1); + if (status < 0) + goto out_main; + + handle = ocfs2_start_trans(osb, OCFS2_SUBALLOC_FREE); + if (IS_ERR(handle)) { + status = PTR_ERR(handle); + mlog_errno(status); + goto out_main_unlock; + } + + status = _ocfs2_reclaim_suballoc_to_main(handle, alloc_inode, + alloc_bh, group_bh, + main_bm_inode, main_bm_bh); + /* group_bh ownership passed to _ocfs2_reclaim_suballoc_to_main() */ + group_bh = NULL; + if (status < 0) + mlog_errno(status); + + ocfs2_commit_trans(osb, handle); + +out_main_unlock: ocfs2_inode_unlock(main_bm_inode, 1); brelse(main_bm_bh); - -free_bm_inode: +out_main: inode_unlock(main_bm_inode); iput(main_bm_inode); +out_group: + brelse(group_bh); +out_alloc_unlock: + ocfs2_inode_unlock(alloc_inode, 1); + brelse(alloc_bh); +out_alloc: + inode_unlock(alloc_inode); +} -bail: - return status; +void ocfs2_suballoc_reclaim_worker(struct work_struct *work) +{ + struct ocfs2_super *osb = container_of(work, struct ocfs2_super, + os_suballoc_reclaim_work); + struct ocfs2_suballoc_reclaim_work *reclaim_work; + + while (1) { + spin_lock(&osb->os_suballoc_reclaim_lock); + if (list_empty(&osb->os_suballoc_reclaim_list)) { + spin_unlock(&osb->os_suballoc_reclaim_lock); + break; + } + reclaim_work = list_first_entry(&osb->os_suballoc_reclaim_list, + struct ocfs2_suballoc_reclaim_work, + list); + list_del(&reclaim_work->list); + spin_unlock(&osb->os_suballoc_reclaim_lock); + + ocfs2_do_suballoc_reclaim(osb, reclaim_work); + iput(reclaim_work->alloc_inode); + kfree(reclaim_work); + } } /* @@ -2955,7 +3098,8 @@ static int _ocfs2_free_suballoc_bits(handle_t *handle, goto bail; } - _ocfs2_reclaim_suballoc_to_main(handle, alloc_inode, alloc_bh, group_bh); + ocfs2_queue_suballoc_reclaim(OCFS2_SB(alloc_inode->i_sb), alloc_inode, + bg_blkno); bail: brelse(group_bh); diff --git a/fs/ocfs2/suballoc.h b/fs/ocfs2/suballoc.h index bcf2ed4a86310b..6042abc032f96e 100644 --- a/fs/ocfs2/suballoc.h +++ b/fs/ocfs2/suballoc.h @@ -206,7 +206,7 @@ int ocfs2_lock_allocators(struct inode *inode, struct ocfs2_extent_tree *et, int ocfs2_test_inode_bit(struct ocfs2_super *osb, u64 blkno, int *res); - +void ocfs2_suballoc_reclaim_worker(struct work_struct *work); /* * The following two interfaces are for ocfs2_create_inode_in_orphan(). diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c index 6a8092b65bb558..1e76b1d9fe0af4 100644 --- a/fs/ocfs2/super.c +++ b/fs/ocfs2/super.c @@ -1784,6 +1784,9 @@ static int ocfs2_mount_volume(struct super_block *sb) out_system_inodes: if (osb->local_alloc_state == OCFS2_LA_ENABLED) ocfs2_shutdown_local_alloc(osb); + /* Drain pending suballoc reclaim work before the journal goes away */ + if (osb->ocfs2_wq) + flush_workqueue(osb->ocfs2_wq); ocfs2_release_system_inodes(osb); /* before journal shutdown, we should release slot_info */ ocfs2_free_slot_info(osb); @@ -1854,6 +1857,14 @@ static void ocfs2_dismount_volume(struct super_block *sb, int mnt_err) if (osb->cconn) ocfs2_super_unlock(osb, 1); + /* + * Drain pending suballoc reclaim work while the system inodes and + * the journal are still alive, since the worker needs to look up + * the global bitmap inode and start a transaction. + */ + if (osb->ocfs2_wq) + flush_workqueue(osb->ocfs2_wq); + ocfs2_release_system_inodes(osb); ocfs2_journal_shutdown(osb); @@ -2140,6 +2151,11 @@ static int ocfs2_initialize_super(struct super_block *sb, INIT_WORK(&osb->dquot_drop_work, ocfs2_drop_dquot_refs); init_llist_head(&osb->dquot_drop_list); + spin_lock_init(&osb->os_suballoc_reclaim_lock); + INIT_LIST_HEAD(&osb->os_suballoc_reclaim_list); + INIT_WORK(&osb->os_suballoc_reclaim_work, + ocfs2_suballoc_reclaim_worker); + /* get some pseudo constants for clustersize bits */ osb->s_clustersize_bits = le32_to_cpu(di->id2.i_super.s_clustersize_bits); From 545b9943b34d3168e2d7765e5ddec1be481e3fde Mon Sep 17 00:00:00 2001 From: Thorsten Blum Date: Fri, 28 Aug 2026 19:43:37 +0200 Subject: [PATCH 752/857] init: simplify early_hostname() Inline the strscpy() check and remove the redundant arglen variable. Use %zu to format the unsigned maxlen argument and add a newline after the truncation warning. Link: https://lore.kernel.org/20260828174337.609333-2-blum@kernel.org Signed-off-by: Thorsten Blum Signed-off-by: Andrew Morton --- init/version.c | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/init/version.c b/init/version.c index 0bd5c45aabc463..cbae11b6f7688b 100644 --- a/init/version.c +++ b/init/version.c @@ -21,16 +21,14 @@ static int __init early_hostname(char *arg) { size_t bufsize = sizeof(init_uts_ns.name.nodename); size_t maxlen = bufsize - 1; - ssize_t arglen; if (!arg) return -EINVAL; - arglen = strscpy(init_uts_ns.name.nodename, arg, bufsize); - if (arglen < 0) { - pr_warn("hostname parameter exceeds %zd characters and will be truncated", + if (strscpy(init_uts_ns.name.nodename, arg, bufsize) < 0) + pr_warn("hostname parameter exceeds %zu characters and will be truncated\n", maxlen); - } + return 0; } early_param("hostname", early_hostname); From 7265d18741e2d0926f4042af86b1acb9a0f0c2d1 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 22 Aug 2026 16:33:27 +0200 Subject: [PATCH 753/857] squashfs: fix fragment index table sizing overflow on 32-bit Patch series "squashfs: harden fragment index table sizing". Two integer overflows undermine fragment index table handling. One is in the original fragment sizing macros. The other is in a bounds check added by commit 1cac63cc9b2f ("Squashfs: add sanity checks to fragment reading at mount time"). Patch 1: the fragment byte count wraps on 32-bit, so the index table is allocated too small and squashfs_frag_lookup() reads out of bounds. A crafted image triggers a KASAN out-of-bounds read on a 32-bit build. With the fix the same image fails cleanly at mount. Patch 2: the check that the table fits before the next one adds two u64 values controlled by the filesystem image and can wrap. This patch (of 2): SQUASHFS_FRAGMENT_BYTES() multiplies the on-disk fragment count (an unsigned int) by sizeof(struct squashfs_fragment_entry), a size_t. On a 32-bit kernel that product is 32-bit and can wrap. squashfs_read_fragment_index_table() sizes the fragment index table from it, but squashfs_frag_lookup() bounds the fragment number against msblk->fragments, the unwrapped superblock value. The two disagree: an image declaring 0x10000001 fragments wraps the product to 16, so a single index entry is allocated, yet the lookup still accepts fragment 0x0fffffff: if (fragment >= msblk->fragments) return -EIO; block = SQUASHFS_FRAGMENT_INDEX(fragment); ... start_block = le64_to_cpu(msblk->fragment_index[block]); block is then 524287 and the read lands ~4MB past an 8-byte allocation. On a 32-bit build KASAN catches it when the crafted image is mounted and the file is stat'd. Cast to u64 in the macro so the multiplication is 64-bit on all targets. After conversion to index-table entries, SQUASHFS_FRAGMENT_INDEX_BYTES() is at most 64 MiB for any u32 count, so it fits both the unsigned int local and the int argument it feeds. 64-bit builds are unchanged. Link: https://lore.kernel.org/20260822143328.68867-1-kmehltretter@gmail.com Link: https://lore.kernel.org/20260822143328.68867-2-kmehltretter@gmail.com Fixes: ffae2cd73a9e ("Squashfs: header files") Signed-off-by: Karl Mehltretter Assisted-by: Claude:claude-opus-5 Cc: Phillip Lougher Cc: Signed-off-by: Andrew Morton --- fs/squashfs/squashfs_fs.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/squashfs/squashfs_fs.h b/fs/squashfs/squashfs_fs.h index a955d9369749f2..93436c7d80c973 100644 --- a/fs/squashfs/squashfs_fs.h +++ b/fs/squashfs/squashfs_fs.h @@ -136,7 +136,7 @@ static inline int squashfs_block_size(__le32 raw) /* fragment and fragment table defines */ #define SQUASHFS_FRAGMENT_BYTES(A) \ - ((A) * sizeof(struct squashfs_fragment_entry)) + ((u64)(A) * sizeof(struct squashfs_fragment_entry)) #define SQUASHFS_FRAGMENT_INDEX(A) (SQUASHFS_FRAGMENT_BYTES(A) / \ SQUASHFS_METADATA_SIZE) From 35f00119cd14f2d8142fe602f75b3a5d5af30e3f Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 22 Aug 2026 16:33:28 +0200 Subject: [PATCH 754/857] squashfs: make the fragment index table bounds check overflow-safe squashfs_read_fragment_index_table() checks that the table fits before the next one with: if (fragment_table_start + length > next_table) return ERR_PTR(-EINVAL); fragment_table_start comes from the superblock and is not validated before this point. A start of 2^64 - length wraps the sum to zero, so the check passes regardless of next_table and fails to reject the invalid table ordering. length then reaches kmalloc() through squashfs_read_table(). A fragment count of 0xffffffff asks for 64MB, order 14. GFP_KERNEL does not include __GFP_NOWARN, so the page allocator warns before the mount fails with -ENOMEM. With panic_on_warn, the warning panics the kernel. Compare the operands instead of adding them. id.c and export.c avoid the same wrap with an exact-size check. Keep the inequality here because a gap before the next table is still allowed. Link: https://lore.kernel.org/20260822143328.68867-3-kmehltretter@gmail.com Fixes: 1cac63cc9b2f ("Squashfs: add sanity checks to fragment reading at mount time") Signed-off-by: Karl Mehltretter Assisted-by: Claude:claude-opus-5 Cc: Phillip Lougher Cc: Signed-off-by: Andrew Morton --- fs/squashfs/fragment.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/squashfs/fragment.c b/fs/squashfs/fragment.c index 49602b9a42e19e..c46e946fa47447 100644 --- a/fs/squashfs/fragment.c +++ b/fs/squashfs/fragment.c @@ -69,9 +69,11 @@ __le64 *squashfs_read_fragment_index_table(struct super_block *sb, /* * Sanity check, length bytes should not extend into the next table - * this check also traps instances where fragment_table_start is - * incorrectly larger than the next table start + * incorrectly larger than the next table start. Both values are read + * from the filesystem image, so compare without adding them. */ - if (fragment_table_start + length > next_table) + if (fragment_table_start > next_table || + length > next_table - fragment_table_start) return ERR_PTR(-EINVAL); table = squashfs_read_table(sb, fragment_table_start, length); From 2ca39c1d0e20875f66bb2888a10fecea878dfdd4 Mon Sep 17 00:00:00 2001 From: Aaron Tomlin Date: Sat, 29 Aug 2026 10:53:20 -0400 Subject: [PATCH 755/857] hung_task: reset warning budget when problem gets resolved Patch series "hung_task: Improve warning budget handling and task reporting", v10. The hung_task watchdog detects tasks stuck in TASK_UNINTERRUPTIBLE (D) state for longer than CONFIG_DEFAULT_HUNG_TASK_TIMEOUT seconds. To prevent log spam during system spikes, sysctl_hung_task_warnings enforces a budget on the number of logged warnings. However, the current implementation has two major limitations: 1. Permanent exhaustion of warning budget sysctl_hung_task_warnings is decremented directly when printing warnings. Once this budget hits zero, no further warnings are reported until an administrator manually updates the sysctl value or reboots the system. Consequently, a single temporary hang episode permanently blinds the kernel watchdog to any subsequent hung tasks after system recovery. 2. Total log suppression when budget is exhausted Once the warning budget reaches zero, hung_task_info() completely suppresses all output, including the basic single-line alert. While suppressing verbose stack dumps and lock debugging is desirable to prevent dmesg flooding, hiding basic task alerts leaves administrators entirely unaware that tasks are hanging. This patch series resolves both limitations by decoupling the configured warning limit from the active runtime budget, automatically resetting the budget upon system recovery or sysctl updates, and emitting a single aggregate summary line when hung tasks are detected under an exhausted warning budget. Patch 1 separates the configured sysctl hung_task_warnings from the runtime budget, making khungtaskd the sole owner of runtime budget updates. The budget is reloaded directly when a scan finds zero hung tasks, or via an atomic reset request published on sysctl write. Patch 2 prevents dmesg flooding during system-wide hangs by keeping non-panic per-task stack dumps budgeted, while providing ongoing visibility by logging a single aggregate summary line at the end of each scan iteration when the warning budget is exhausted. This patch (of 2): The sysctl hung_task_warnings currently holds both the configured warning limit and the remaining budget. Each detailed report decrements the sysctl, so once it reaches zero, the configured limit is lost and cannot be restored automatically. Keep sysctl_hung_task_warnings as the configured warning limit and make khungtaskd the sole owner of the remaining budget. A check that finds no hung tasks reloads the budget directly from the configured limit. A successful sysctl write publishes an atomic reset request, which khungtaskd consumes at the start of the next check. Link: https://lore.kernel.org/20260829145321.18423-1-atomlin@atomlin.com Link: https://lore.kernel.org/20260829145321.18423-2-atomlin@atomlin.com Signed-off-by: Aaron Tomlin Suggested-by: Petr Mladek Suggested-by: Lance Yang Tested-by: Lance Yang Reviewed-by: Lance Yang Reviewed-by: Bradley Morgan Cc: David Laight Cc: "Masami Hiramatsu (Google)" Signed-off-by: Andrew Morton --- Documentation/admin-guide/sysctl/kernel.rst | 5 ++- kernel/hung_task.c | 49 +++++++++++++++++---- 2 files changed, 44 insertions(+), 10 deletions(-) diff --git a/Documentation/admin-guide/sysctl/kernel.rst b/Documentation/admin-guide/sysctl/kernel.rst index b6328cd0f43e94..31fbc8c9184cb1 100644 --- a/Documentation/admin-guide/sysctl/kernel.rst +++ b/Documentation/admin-guide/sysctl/kernel.rst @@ -459,8 +459,9 @@ hung_task_warnings ================== The maximum number of warnings to report. During a check interval -if a hung task is detected, this value is decreased by 1. -When this value reaches 0, no more warnings will be reported. +if a hung task is detected, the internal warning budget is decreased by 1. +When this budget reaches 0, no more detailed warnings will be reported. The +warning budget is reset to the configured limit when no hung task is found. This file shows up if ``CONFIG_DETECT_HUNG_TASK`` is enabled. -1: report an infinite number of warnings. diff --git a/kernel/hung_task.c b/kernel/hung_task.c index 6fcc94ce4ca9d2..a5043188456d42 100644 --- a/kernel/hung_task.c +++ b/kernel/hung_task.c @@ -57,8 +57,20 @@ unsigned long __read_mostly sysctl_hung_task_timeout_secs = CONFIG_DEFAULT_HUNG_ */ static unsigned long __read_mostly sysctl_hung_task_check_interval_secs; +/* + * Limit the number of printed hung tasks to prevent printing + * the same or similar backtraces repeatedly. + */ static int __read_mostly sysctl_hung_task_warnings = 10; +/* + * The number of hung tasks which still can be reported. + * The budget gets restored to the original limit when + * the previous stall is resolved. + */ +static int hung_task_warnings_budget = 10; +static atomic_t reset_hung_task_warnings = ATOMIC_INIT(0); + static int __read_mostly did_panic; static bool hung_task_call_panic; @@ -245,11 +257,11 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout, /* * The given task did not get scheduled for more than * CONFIG_DEFAULT_HUNG_TASK_TIMEOUT. Therefore, complain - * accordingly + * accordingly with full details if the budget is not exhausted. */ - if (sysctl_hung_task_warnings || hung_task_call_panic) { - if (sysctl_hung_task_warnings > 0) - sysctl_hung_task_warnings--; + if (hung_task_warnings_budget || hung_task_call_panic) { + if (hung_task_warnings_budget > 0) + hung_task_warnings_budget--; pr_err("INFO: task %s:%d blocked%s for more than %ld seconds.\n", t->comm, t->pid, t->in_iowait ? " in I/O wait" : "", (jiffies - t->last_switch_time) / HZ); @@ -264,7 +276,7 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout, sched_show_task(t); debug_show_blocker(t, timeout); - if (!sysctl_hung_task_warnings) + if (!hung_task_warnings_budget) pr_info("Future hung task reports are suppressed, see sysctl kernel.hung_task_warnings\n"); } @@ -304,7 +316,7 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) unsigned long last_break = jiffies; struct task_struct *g, *t; unsigned long this_round_count; - int need_warning = sysctl_hung_task_warnings; + int need_warning; unsigned long si_mask = hung_task_si_mask; /* @@ -314,6 +326,11 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) if (test_taint(TAINT_DIE) || did_panic) return; + if (atomic_xchg_acquire(&reset_hung_task_warnings, 0)) + hung_task_warnings_budget = + READ_ONCE(sysctl_hung_task_warnings); + need_warning = hung_task_warnings_budget; + this_round_count = 0; rcu_read_lock(); for_each_process_thread(g, t) { @@ -340,8 +357,11 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) unlock: rcu_read_unlock(); - if (!this_round_count) + if (!this_round_count) { + hung_task_warnings_budget = + READ_ONCE(sysctl_hung_task_warnings); return; + } if (need_warning || hung_task_call_panic) { si_mask |= SYS_INFO_LOCKS; @@ -425,6 +445,19 @@ static int proc_dohung_task_timeout_secs(const struct ctl_table *table, int writ return ret; } +static int proc_dohung_task_warnings(const struct ctl_table *table, int write, + void *buffer, + size_t *lenp, loff_t *ppos) +{ + int ret; + + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + if (!ret && write) + atomic_set_release(&reset_hung_task_warnings, 1); + + return ret; +} + /* * This is needed for proc_doulongvec_minmax of sysctl_hung_task_timeout_secs * and hung_task_check_interval_secs @@ -480,7 +513,7 @@ static const struct ctl_table hung_task_sysctls[] = { .data = &sysctl_hung_task_warnings, .maxlen = sizeof(int), .mode = 0644, - .proc_handler = proc_dointvec_minmax, + .proc_handler = proc_dohung_task_warnings, .extra1 = SYSCTL_NEG_ONE, }, { From ce713cff74a92014177aff0eb58cde28821b5c69 Mon Sep 17 00:00:00 2001 From: Aaron Tomlin Date: Sat, 29 Aug 2026 10:53:21 -0400 Subject: [PATCH 756/857] hung_task: log summary line when warning budget is exhausted Once the warning budget is exhausted, hung_task_info() normally stops printing per-task details. When panic is triggered, full details are still printed so diagnostics remain available before panic. To retain visibility without restoring per-task output after budget exhaustion, emit a single aggregate summary line at the end of each watchdog scan that detects hung tasks with an exhausted budget. This keeps non-panic per-task reports budgeted during system-wide hangs. Link: https://lore.kernel.org/20260829145321.18423-3-atomlin@atomlin.com Signed-off-by: Aaron Tomlin Suggested-by: Petr Mladek Suggested-by: Lance Yang Reviewed-by: Petr Mladek Reviewed-by: Lance Yang Reviewed-by: Bradley Morgan Cc: David Laight Cc: "Masami Hiramatsu (Google)" Signed-off-by: Andrew Morton --- kernel/hung_task.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/kernel/hung_task.c b/kernel/hung_task.c index a5043188456d42..49d47ae475ab1c 100644 --- a/kernel/hung_task.c +++ b/kernel/hung_task.c @@ -277,7 +277,7 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout, debug_show_blocker(t, timeout); if (!hung_task_warnings_budget) - pr_info("Future hung task reports are suppressed, see sysctl kernel.hung_task_warnings\n"); + pr_info("hung_task: further per-task details suppressed until warning budget is reset or panic is triggered (see sysctl kernel.hung_task_warnings)\n"); } touch_nmi_watchdog(); @@ -363,6 +363,10 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) return; } + if (!hung_task_warnings_budget && !hung_task_call_panic) + pr_info("hung_task: %lu hung tasks detected, warning budget exhausted\n", + this_round_count); + if (need_warning || hung_task_call_panic) { si_mask |= SYS_INFO_LOCKS; From 6b98e27978cc062792360dc18d8b7b6a318f2f5d Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 23:16:32 +0000 Subject: [PATCH 757/857] init, arch: make CONFIG_COMMAND_LINE_SIZE globally configurable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently, s390 has the ability to configure the maximum kernel command line size via Kconfig (CONFIG_COMMAND_LINE_SIZE). Other architectures define a hardcoded COMMAND_LINE_SIZE macro in their setup.h headers. In some use cases, such as netboot kernels, rootfs configurations, or larger initramfs setups, a larger command line size is required. While for embedded workloads, it can be reduced to save memory. Move CONFIG_COMMAND_LINE_SIZE out of arch/s390/Kconfig and into init/Kconfig under General setup, and update every architecture's setup.h header to define COMMAND_LINE_SIZE as CONFIG_COMMAND_LINE_SIZE. For user-space API (uapi) headers, wrap the definition in an `#ifdef __KERNEL__` guard and retain the historical hardcoded default in the `#else` block. When user-space headers are installed via `make headers_install`, unifdef strips out the kernel section, ensuring the same value as before for user-space applications including ``. For S390, the range is kept the same, but other architectures have varying constraints. S390 requires a minimum of 896 bytes to protect legacy bootloaders from overwriting the .text section. ARM, M68K, and NIOS2 allocate the command line directly on severely constrained decompressor stacks, so their ranges are strictly capped at 2048 bytes to prevent deterministic stack exhaustion and boot panics. PowerPC (PPC) boot wrappers silently truncate arguments past 2048 bytes, so it is also capped at 2048 to prevent silent parameter loss. The SuperH (SUPERH) boot parameter page allocates exactly PAGE_SIZE (typically 4096 bytes), and placing a 4096-byte command line starting at offset 256 would cause strscpy() to read out of bounds; it is capped at 3840 bytes. Alpha physically limits its boot parameter block to 256 bytes, so its limit is strictly locked to 256. All other architectures are capped at 4096 bytes to prevent unreasonable allocations. Link: https://lore.kernel.org/20260818231646.804507-2-wfelipe@google.com Signed-off-by: Maciej Żenczykowski Signed-off-by: Wilson Felipe Pereira Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Arnd Bergmann Cc: Christian Borntraeger Cc: Heiko Carstens Cc: Palmer Dabbelt Cc: Sven Schnelle Cc: Vasily Gorbik Signed-off-by: Andrew Morton --- arch/alpha/include/uapi/asm/setup.h | 4 ++++ arch/arc/include/asm/setup.h | 2 +- arch/arm/include/uapi/asm/setup.h | 6 +++++- arch/arm64/include/uapi/asm/setup.h | 4 ++++ arch/loongarch/include/uapi/asm/setup.h | 4 ++++ arch/m68k/include/uapi/asm/setup.h | 6 +++++- arch/microblaze/include/uapi/asm/setup.h | 4 ++++ arch/mips/include/uapi/asm/setup.h | 4 ++++ arch/parisc/include/uapi/asm/setup.h | 4 ++++ arch/powerpc/include/uapi/asm/setup.h | 4 ++++ arch/riscv/include/uapi/asm/setup.h | 4 ++++ arch/s390/Kconfig | 8 -------- arch/sparc/include/uapi/asm/setup.h | 10 +++++++--- arch/um/include/asm/setup.h | 2 +- arch/x86/include/asm/setup.h | 2 +- arch/xtensa/include/uapi/asm/setup.h | 4 ++++ include/uapi/asm-generic/setup.h | 4 ++++ init/Kconfig | 16 ++++++++++++++++ 18 files changed, 76 insertions(+), 16 deletions(-) diff --git a/arch/alpha/include/uapi/asm/setup.h b/arch/alpha/include/uapi/asm/setup.h index f881ea5947cbc1..169f743ef7658d 100644 --- a/arch/alpha/include/uapi/asm/setup.h +++ b/arch/alpha/include/uapi/asm/setup.h @@ -2,6 +2,10 @@ #ifndef _UAPI__ALPHA_SETUP_H #define _UAPI__ALPHA_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 256 +#endif #endif /* _UAPI__ALPHA_SETUP_H */ diff --git a/arch/arc/include/asm/setup.h b/arch/arc/include/asm/setup.h index 1c6db599e1fcc9..60e158d58cec77 100644 --- a/arch/arc/include/asm/setup.h +++ b/arch/arc/include/asm/setup.h @@ -9,7 +9,7 @@ #include #include -#define COMMAND_LINE_SIZE 256 +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE /* * Data structure to map a ID to string diff --git a/arch/arm/include/uapi/asm/setup.h b/arch/arm/include/uapi/asm/setup.h index 8e50e034fec73a..4aa93558af1e7f 100644 --- a/arch/arm/include/uapi/asm/setup.h +++ b/arch/arm/include/uapi/asm/setup.h @@ -17,7 +17,11 @@ #include -#define COMMAND_LINE_SIZE 1024 +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else +#define COMMAND_LINE_SIZE 1024 +#endif /* The list ends with an ATAG_NONE node. */ #define ATAG_NONE 0x00000000 diff --git a/arch/arm64/include/uapi/asm/setup.h b/arch/arm64/include/uapi/asm/setup.h index 5d703888f35110..2236890175a5ab 100644 --- a/arch/arm64/include/uapi/asm/setup.h +++ b/arch/arm64/include/uapi/asm/setup.h @@ -22,6 +22,10 @@ #include +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 2048 +#endif #endif diff --git a/arch/loongarch/include/uapi/asm/setup.h b/arch/loongarch/include/uapi/asm/setup.h index d46363ce3e024c..03c7bfa1e5d9f2 100644 --- a/arch/loongarch/include/uapi/asm/setup.h +++ b/arch/loongarch/include/uapi/asm/setup.h @@ -3,6 +3,10 @@ #ifndef _UAPI_ASM_LOONGARCH_SETUP_H #define _UAPI_ASM_LOONGARCH_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 4096 +#endif #endif /* _UAPI_ASM_LOONGARCH_SETUP_H */ diff --git a/arch/m68k/include/uapi/asm/setup.h b/arch/m68k/include/uapi/asm/setup.h index 25fe26d5597cc6..2d5b24a5345f92 100644 --- a/arch/m68k/include/uapi/asm/setup.h +++ b/arch/m68k/include/uapi/asm/setup.h @@ -12,6 +12,10 @@ #ifndef _UAPI_M68K_SETUP_H #define _UAPI_M68K_SETUP_H -#define COMMAND_LINE_SIZE 256 +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else +#define COMMAND_LINE_SIZE 256 +#endif #endif /* _UAPI_M68K_SETUP_H */ diff --git a/arch/microblaze/include/uapi/asm/setup.h b/arch/microblaze/include/uapi/asm/setup.h index 16c56807f86a2d..e4b253064c7d99 100644 --- a/arch/microblaze/include/uapi/asm/setup.h +++ b/arch/microblaze/include/uapi/asm/setup.h @@ -12,6 +12,10 @@ #ifndef _UAPI_ASM_MICROBLAZE_SETUP_H #define _UAPI_ASM_MICROBLAZE_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 256 +#endif #endif /* _UAPI_ASM_MICROBLAZE_SETUP_H */ diff --git a/arch/mips/include/uapi/asm/setup.h b/arch/mips/include/uapi/asm/setup.h index 7d48c433b0c27d..8d6c474835aece 100644 --- a/arch/mips/include/uapi/asm/setup.h +++ b/arch/mips/include/uapi/asm/setup.h @@ -2,7 +2,11 @@ #ifndef _UAPI_MIPS_SETUP_H #define _UAPI_MIPS_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 4096 +#endif #endif /* _UAPI_MIPS_SETUP_H */ diff --git a/arch/parisc/include/uapi/asm/setup.h b/arch/parisc/include/uapi/asm/setup.h index 78b2f4ec7d6522..cfa77e84205dc6 100644 --- a/arch/parisc/include/uapi/asm/setup.h +++ b/arch/parisc/include/uapi/asm/setup.h @@ -2,6 +2,10 @@ #ifndef _PARISC_SETUP_H #define _PARISC_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 1024 +#endif #endif /* _PARISC_SETUP_H */ diff --git a/arch/powerpc/include/uapi/asm/setup.h b/arch/powerpc/include/uapi/asm/setup.h index c54940b09d065c..daa15ac4e94c6e 100644 --- a/arch/powerpc/include/uapi/asm/setup.h +++ b/arch/powerpc/include/uapi/asm/setup.h @@ -2,6 +2,10 @@ #ifndef _UAPI_ASM_POWERPC_SETUP_H #define _UAPI_ASM_POWERPC_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 2048 +#endif #endif /* _UAPI_ASM_POWERPC_SETUP_H */ diff --git a/arch/riscv/include/uapi/asm/setup.h b/arch/riscv/include/uapi/asm/setup.h index eb4f0209c6960e..a6c1f4b0987e35 100644 --- a/arch/riscv/include/uapi/asm/setup.h +++ b/arch/riscv/include/uapi/asm/setup.h @@ -3,6 +3,10 @@ #ifndef _UAPI_ASM_RISCV_SETUP_H #define _UAPI_ASM_RISCV_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 2048 +#endif #endif /* _UAPI_ASM_RISCV_SETUP_H */ diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index 4b51bc6e8948d7..026ba041ca9509 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -521,14 +521,6 @@ endchoice config 64BIT def_bool y -config COMMAND_LINE_SIZE - int "Maximum size of kernel command line" - default 4096 - range 896 1048576 - help - This allows you to specify the maximum length of the kernel command - line. - config SMP def_bool y diff --git a/arch/sparc/include/uapi/asm/setup.h b/arch/sparc/include/uapi/asm/setup.h index 3c208a4dd46405..7054f5249a3a68 100644 --- a/arch/sparc/include/uapi/asm/setup.h +++ b/arch/sparc/include/uapi/asm/setup.h @@ -6,10 +6,14 @@ #ifndef _UAPI_SPARC_SETUP_H #define _UAPI_SPARC_SETUP_H -#if defined(__sparc__) && defined(__arch64__) -# define COMMAND_LINE_SIZE 2048 +#ifdef __KERNEL__ +# define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE #else -# define COMMAND_LINE_SIZE 256 +# if defined(__sparc__) && defined(__arch64__) +# define COMMAND_LINE_SIZE 2048 +# else +# define COMMAND_LINE_SIZE 256 +# endif #endif diff --git a/arch/um/include/asm/setup.h b/arch/um/include/asm/setup.h index 80ada899f25426..bc83dc4d467d30 100644 --- a/arch/um/include/asm/setup.h +++ b/arch/um/include/asm/setup.h @@ -6,6 +6,6 @@ * command line, so this choice is ok. */ -#define COMMAND_LINE_SIZE 4096 +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE #endif /* SETUP_H_INCLUDED */ diff --git a/arch/x86/include/asm/setup.h b/arch/x86/include/asm/setup.h index 895d09faaf832e..1b333bb091d7f2 100644 --- a/arch/x86/include/asm/setup.h +++ b/arch/x86/include/asm/setup.h @@ -4,7 +4,7 @@ #include -#define COMMAND_LINE_SIZE 2048 +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE #include #include diff --git a/arch/xtensa/include/uapi/asm/setup.h b/arch/xtensa/include/uapi/asm/setup.h index 5356a5fd4d1737..dcf33a403527a3 100644 --- a/arch/xtensa/include/uapi/asm/setup.h +++ b/arch/xtensa/include/uapi/asm/setup.h @@ -12,6 +12,10 @@ #ifndef _XTENSA_SETUP_H #define _XTENSA_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 256 +#endif #endif diff --git a/include/uapi/asm-generic/setup.h b/include/uapi/asm-generic/setup.h index 88ac5100df3598..b8d06e6d56bd7e 100644 --- a/include/uapi/asm-generic/setup.h +++ b/include/uapi/asm-generic/setup.h @@ -2,6 +2,10 @@ #ifndef __ASM_GENERIC_SETUP_H #define __ASM_GENERIC_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 512 +#endif #endif /* __ASM_GENERIC_SETUP_H */ diff --git a/init/Kconfig b/init/Kconfig index 8583d9f06c522e..edc79893e1ac4f 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -1614,6 +1614,22 @@ config CMDLINE_FROM_BOOTCONFIG If unsure, say N. +config COMMAND_LINE_SIZE + int "Maximum size of kernel command line" + default 4096 if S390 || LOONGARCH || MIPS || UML + default 2048 if X86 || ARM64 || PPC || SPARC64 || RISCV + default 1024 if ARM || PARISC + default 256 if ALPHA || ARC || M68K || MICROBLAZE || SPARC32 || XTENSA + default 512 + range 896 1048576 if S390 + range 256 2048 if ARM || M68K || NIOS2 || PPC + range 256 3840 if SUPERH + range 256 256 if ALPHA + range 256 4096 + help + This allows you to specify the maximum length of the kernel command + line. + config CMDLINE_LOG_WRAP_IDEAL_LEN int "Length to try to wrap the cmdline when logged at boot" default 1021 From 7ec117c6318a362783afa03e5bcc7f7ace76a78b Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 23:16:33 +0000 Subject: [PATCH 758/857] init/Kconfig: make config INIT_ENV_ARG_LIMIT user-configurable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "init, arch: make command line size and init arg limit configurable", v2. This patch series promotes COMMAND_LINE_SIZE from arch/s390/Kconfig to init/Kconfig to be generally available to other architectures. In some use cases, such as netboot kernels, rootfs configurations, larger initramfs setups, require larger sizes. While for embedded workloads, it can be reduced to save memory. Since COMMAND_LINE_SIZE can be larger, it also makes sense to allow INIT_ENV_ARG_LIMIT to be configured. This patch (of 2): INIT_ENV_ARG_LIMIT is defined without a prompt string (`int`), making it a hidden Kconfig symbol that defaults to 32 (or 128 for UML) and cannot be configured in `make menuconfig`. Now that CONFIG_COMMAND_LINE_SIZE is configurable across all architectures, users who select larger kernel command lines (e.g., 4096 bytes) may pass more than 32 command-line arguments or environment variables (`foo=bar`) to `/sbin/init`. If INIT_ENV_ARG_LIMIT remains hardcoded at 32, any argument after the 32nd sets the panic_later flag and causes a hard kernel panic on boot. Add a prompt string ("Maximum number of kernel command line arguments") and a `range 32 4096` to `config INIT_ENV_ARG_LIMIT` so that users can configure their init argument and environment variable limit when needed, while preserving the existing default of 32 for standard builds. Link: https://lore.kernel.org/20260818231646.804507-1-wfelipe@google.com Link: https://lore.kernel.org/20260818231646.804507-3-wfelipe@google.com Signed-off-by: Maciej Żenczykowski Signed-off-by: Wilson Felipe Pereira Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Arnd Bergmann Cc: Christian Borntraeger Cc: Heiko Carstens Cc: Palmer Dabbelt Cc: Sven Schnelle Cc: Vasily Gorbik Signed-off-by: Andrew Morton --- init/Kconfig | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/init/Kconfig b/init/Kconfig index edc79893e1ac4f..3b141ba427d5e0 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -231,9 +231,10 @@ config BROKEN_ON_SMP default y config INIT_ENV_ARG_LIMIT - int + int "Maximum number of kernel command line arguments" default 32 if !UML default 128 if UML + range 32 4096 help Maximum of each of the number of arguments and environment variables passed to init from the kernel command line. From 963c5e0e78102a6ae3f9be830cbcf90f90c2d3f0 Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Mon, 17 Aug 2026 20:16:13 +0800 Subject: [PATCH 759/857] minmax.h: update the stale 'x' versus 'ux' comment Commit b280bb27a9f7 ("minmax.h: reduce the #define expansion of min(), max() and clamp()") made __sign_use(), __is_nonneg() and __types_ok() take only 'ux', and commit a5743f32baec ("minmax.h: use BUILD_BUG_ON_MSG() for the lo < hi test in clamp()") did the same for the clamp() limit test. The comment describing the old split was added one patch earlier and was never updated. 'ux' now carries the value check too, since __is_nonneg() tests it rather than the original expression, and nothing here looks at the value of 'x' any more: it is expanded only to initialise 'ux' and in the error message, as the first of those changes intended. Link: https://lore.kernel.org/20260817121613.3846511-1-zhanxusheng@xiaomi.com Signed-off-by: Zhan Xusheng Cc: David Laight Cc: "H. Peter Anvin" Signed-off-by: Andrew Morton --- include/linux/minmax.h | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/include/linux/minmax.h b/include/linux/minmax.h index a0158db54a0411..5ef4d58c0c4291 100644 --- a/include/linux/minmax.h +++ b/include/linux/minmax.h @@ -38,9 +38,9 @@ * Note that 'x' is the original expression, and 'ux' is the unique variable * that contains the value. * - * We use 'ux' for pure type checking, and 'x' for when we need to look at the - * value (but without evaluating it for side effects! - * Careful to only ever evaluate it with sizeof() or __builtin_constant_p() etc). + * We use 'ux' for both the type and the value checks, so 'x' itself is only + * expanded twice: once to initialise 'ux', and once quoted in the error + * message. * * Pointers end up being checked by the normal C type rules at the actual * comparison, and these expressions only need to be careful to not cause From 332a62161ac92f0ec2207e3f498c7afcd2c20612 Mon Sep 17 00:00:00 2001 From: Julian Braha Date: Mon, 17 Aug 2026 22:03:11 +0100 Subject: [PATCH 760/857] lib: cleanup "fake" tristates in Kconfig These 7 DECOMPRESS_ options (e.g. DECOMPRESS_GZIP) currently have the tristate type, but can never be set to M. Their only valid values are Y and N, making them effectively booleans. Let's make their types more accurate by changing them to 'bool'. Note that this is only a code cleanup, there is no functional change. These bistates were found by kconfirm, a static analysis tool for Kconfig. Link: https://lore.kernel.org/20260817210311.2142999-1-julianbraha@gmail.com Signed-off-by: Julian Braha Cc: Arnd Bergmann Cc: Jani Nikula Cc: Julia Lawall Signed-off-by: Andrew Morton --- lib/Kconfig | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/lib/Kconfig b/lib/Kconfig index 4e6b34c3346d56..1e42f5167c68bc 100644 --- a/lib/Kconfig +++ b/lib/Kconfig @@ -219,29 +219,29 @@ source "lib/xz/Kconfig" # config DECOMPRESS_GZIP select ZLIB_INFLATE - tristate + bool config DECOMPRESS_BZIP2 - tristate + bool config DECOMPRESS_LZMA - tristate + bool config DECOMPRESS_XZ select XZ_DEC - tristate + bool config DECOMPRESS_LZO select LZO_DECOMPRESS - tristate + bool config DECOMPRESS_LZ4 select LZ4_DECOMPRESS - tristate + bool config DECOMPRESS_ZSTD select ZSTD_DECOMPRESS - tristate + bool # # Generic allocator support is selected if needed From 562676f80ef7dff8c1b2ef472a31e83a8a425ba9 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 04:53:46 +0000 Subject: [PATCH 761/857] init/main: fix off-by-one in argv_init cleanup Patch series "init: fix array boundary bugs in boot parameter parsing". This series fixes two distinct boundary logic edge-case bugs in `init/main.c` related to parsing boot command-line arguments and environment variables. Both bugs have been present since the early git history (Linux-2.6.12-rc2). 1. The first patch fixes an off-by-one error in `init_setup()` where the final slot of the `argv_init` array was left uncleared. This allowed a stale kernel parameter to leak into the `init` process's user-space command line if exactly `MAX_INIT_ARGS` unknown parameters were passed. 2. The second patch fixes a false-positive kernel panic in `unknown_bootoption()`. If a user filled the environment variable array up to its exact limit (32) and then attempted to overwrite the final variable, the kernel would panic before evaluating whether it was a harmless duplicate. Exact QEMU reproduction steps for both edge cases are documented inside their respective commit descriptions. This patch (of 2): When cleaning up argv_init in init_setup() and rdinit_setup(), the loop terminates one element early due to using '<' instead of '<='. Since argv_init is sized MAX_INIT_ARGS+2, index MAX_INIT_ARGS is a valid element that should be cleared to NULL. If exactly MAX_INIT_ARGS unknown arguments are passed before 'init=', the uncleared argv_init[MAX_INIT_ARGS] can act as a ghost argument to /sbin/init or cause a spurious kernel panic when later appended to. To verify the argument leak, boot a VM into a shell with 32 unknown kernel arguments, the init parameter, and 31 user arguments: STALE_ARGS=$(for i in {1..32}; do echo -n "stale$i "; done) USER_ARGS=$(for i in {1..31}; do echo -n "user$i "; done) qemu-system-x86_64 -kernel bzImage \ -append "$STALE_ARGS init=/bin/sh $USER_ARGS" Running `cat /proc/1/cmdline` inside the shell reveals that the 32nd kernel argument ('stale32') incorrectly leaked into the init process's command line. This patch zeroes the final slot, cleanly terminating the array. Link: https://lore.kernel.org/20260818045357.4123784-1-wfelipe@google.com Link: https://lore.kernel.org/20260818045357.4123784-2-wfelipe@google.com Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Fixes: ffdfc40976dd ("[PATCH] Add rdinit parameter to pick early userspace init") Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton --- init/main.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/init/main.c b/init/main.c index 2613d3f9b3ce93..87bd282437667e 100644 --- a/init/main.c +++ b/init/main.c @@ -576,7 +576,7 @@ static int __init init_setup(char *str) * the shell think it should execute a script with such name. * So we ignore all arguments entered _before_ init=... [MJ] */ - for (i = 1; i < MAX_INIT_ARGS; i++) + for (i = 1; i <= MAX_INIT_ARGS; i++) argv_init[i] = NULL; return 1; } @@ -589,7 +589,7 @@ static int __init rdinit_setup(char *str) ramdisk_execute_command = str; ramdisk_execute_command_set = true; /* See "auto" comment in init_setup */ - for (i = 1; i < MAX_INIT_ARGS; i++) + for (i = 1; i <= MAX_INIT_ARGS; i++) argv_init[i] = NULL; return 1; } From e2b7bcabf4c5e13621e55083dc0af8088a4b2858 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 04:53:47 +0000 Subject: [PATCH 762/857] init/main: fix false-positive kernel panic on environment variable overwrite In unknown_bootoption(), the limit checking for environment variables sets panic_later *before* checking if the variable already exists in envp_init. If a user passes exactly MAX_INIT_ENVS custom variables and then overwrites the final variable by matching its key, it causes a false-positive hard panic on boot despite not actually exceeding the array bounds or increasing the total variable count. Swapping the order of these checks allows the duplicate check to break out of the loop before the panic flag is erroneously latched. To verify, boot a VM with 31 custom variables (filling the array up to its limit of 32) and then overwrite the very last variable: ENV_VARS=$(for i in {1..31}; do echo -n "var$i=$i "; done) qemu-system-x86_64 -kernel bzImage -append "$ENV_VARS var31=overwrite" Without this patch, the kernel crashes instantly with: Kernel panic - not syncing: Too many boot env vars at 'var31=overwrite' With this patch, the kernel safely overwrites the variable and boots. Link: https://lore.kernel.org/20260818045357.4123784-3-wfelipe@google.com Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton --- init/main.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/init/main.c b/init/main.c index 87bd282437667e..591d97246c1a88 100644 --- a/init/main.c +++ b/init/main.c @@ -543,12 +543,12 @@ static int __init unknown_bootoption(char *param, char *val, /* Environment option */ unsigned int i; for (i = 0; envp_init[i]; i++) { + if (!strncmp(param, envp_init[i], len+1)) + break; if (i == MAX_INIT_ENVS) { panic_later = "env"; panic_param = param; } - if (!strncmp(param, envp_init[i], len+1)) - break; } envp_init[i] = param; } else { From f5a9cf938481d66115bb592944b21638f2a3b77e Mon Sep 17 00:00:00 2001 From: Andrei Vagin Date: Sun, 16 Aug 2026 16:12:15 +0000 Subject: [PATCH 763/857] proc: report SIGEV_NONE in /proc/pid/timers if target task has died When a posix timer is created targeting a specific thread (using SIGEV_SIGNAL | SIGEV_THREAD_ID), it takes a reference to the target struct pid in timer->it_pid. If the target thread subsequently terminates, its numeric tid is freed and can be recycled for an unrelated task. However, the timer holds its reference to the original struct pid. show_timer() in /proc/[pid]/timers previously called pid_nr_ns() directly on timer->it_pid without checking whether any task remained attached to that struct pid. As a result: 1. It reported the stale tid, which could mistakenly refer to a recycled pid. 2. In the kernel, expired signals for dead target threads are dropped by posixtimer_send_sigqueue() because posixtimer_get_target() returns NULL, so the timer functionally acts as SIGEV_NONE. 3. Checkpoint/restore tools (CRIU) parsing /proc/[pid]/timers would try to restore a timer with SIGEV_SIGNAL | SIGEV_THREAD_ID targeting a non-existent or unrelated thread. Check pid_has_task(timer->it_pid, timer->it_pid_type) in show_timer(). If the target task has died, override notify to SIGEV_NONE and report PID 0 (e.g., 'notify: none/pid.0'). Link: https://lore.kernel.org/20260816161216.984580-1-avagin@google.com Fixes: 57b8015e07a7 ("posix-timers: Show sigevent info in proc file") Signed-off-by: Andrei Vagin Reviewed-by: Pavel Tikhomirov Cc: Thomas Gleixner Signed-off-by: Andrew Morton --- fs/proc/base.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/fs/proc/base.c b/fs/proc/base.c index 6a39de424f62a1..1455de58e53cf0 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -2519,17 +2519,23 @@ static int show_timer(struct seq_file *m, void *v) struct k_itimer *timer = hlist_entry((struct hlist_node *)v, struct k_itimer, list); struct timers_private *tp = m->private; int notify = timer->it_sigev_notify; + pid_t nr = 0; guard(spinlock_irq)(&timer->it_lock); if (!posixtimer_valid(timer)) return 0; + if (timer->it_pid && pid_has_task(timer->it_pid, timer->it_pid_type)) + nr = pid_nr_ns(timer->it_pid, tp->ns); + else + notify = SIGEV_NONE; + seq_printf(m, "ID: %d\n", timer->it_id); seq_printf(m, "signal: %d/%px\n", timer->sigq.info.si_signo, timer->sigq.info.si_value.sival_ptr); seq_printf(m, "notify: %s/%s.%d\n", nstr[notify & ~SIGEV_THREAD_ID], (notify & SIGEV_THREAD_ID) ? "tid" : "pid", - pid_nr_ns(timer->it_pid, tp->ns)); + nr); seq_printf(m, "ClockID: %d\n", timer->it_clock); return 0; From 65e303e49954b429d173c7bd4d1a353301643efe Mon Sep 17 00:00:00 2001 From: Konstantin Khorenko Date: Fri, 14 Aug 2026 18:57:09 +0200 Subject: [PATCH 764/857] selftests/core: fix unshare_test with large fs.nr_open The test assumes fs.nr_open is close to the default 1048576, but some systems set it much higher (e.g. 1073741816). This is systemd's doing: since systemd v240 (2018), PID 1 bumps fs.nr_open and fs.file-max to their largest possible values on boot, as file descriptors are already accounted for by memcg [1]. In that case, dup2() to nr_open + 64 requires the kernel to allocate a file descriptor table with ~1 billion entries, which fails with ENOMEM. On a kernel that already carries 04a2c4b4511d1, dup2() no longer fails with ENOMEM. The allocation is now rejected up front and the caller gets EMFILE instead, without the WARNING, but the test still fails. Cap the nr_open value used for the test's own arithmetic to a known reasonable base value (1048576) and restore the true original value once the test has completed. Link: https://lore.kernel.org/20260814165709.513263-1-khorenko@virtuozzo.com Link: https://github.com/systemd/systemd/commit/a8b627aaed409a15260c25988970c795bf963812 [1] Signed-off-by: Konstantin Khorenko Signed-off-by: Eva Kurchatova Cc: Shuah Khan Cc: Wei Yang Cc: Signed-off-by: Andrew Morton --- tools/testing/selftests/core/unshare_test.c | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/core/unshare_test.c b/tools/testing/selftests/core/unshare_test.c index ffce75a6c228f1..d40e963dd5205e 100644 --- a/tools/testing/selftests/core/unshare_test.c +++ b/tools/testing/selftests/core/unshare_test.c @@ -40,6 +40,14 @@ TEST(unshare_EMFILE) ASSERT_EQ(sscanf(buf, "%d", &nr_open), 1); + /* + * Cap nr_open for the duration of the test to avoid ENOMEM from a + * huge fd table allocation; buf/n keep the real original value so + * fs.nr_open can be restored to it once the test is done. + */ + if (nr_open > 1024 * 1024) + nr_open = 1024 * 1024; + ASSERT_EQ(0, getrlimit(RLIMIT_NOFILE, &rlimit)); /* bump fs.nr_open */ @@ -73,10 +81,13 @@ TEST(unshare_EMFILE) if (pid == 0) { int err; + char buf3[32]; + ssize_t n3; - /* restore fs.nr_open */ + /* restore fs.nr_open to the (possibly capped) test baseline */ + n3 = sprintf(buf3, "%d\n", nr_open); lseek(fd, 0, SEEK_SET); - write(fd, buf, n); + write(fd, buf3, n3); /* ... and now unshare(CLONE_FILES) must fail with EMFILE */ err = unshare(CLONE_FILES); EXPECT_EQ(err, -1) @@ -89,6 +100,10 @@ TEST(unshare_EMFILE) EXPECT_EQ(waitpid(pid, &status, 0), pid); EXPECT_EQ(true, WIFEXITED(status)); EXPECT_EQ(0, WEXITSTATUS(status)); + + /* restore the real fs.nr_open value */ + lseek(fd, 0, SEEK_SET); + write(fd, buf, n); } TEST_HARNESS_MAIN From 6c2b6d8d1190816a99e4914aee7a42db45580b74 Mon Sep 17 00:00:00 2001 From: Thomas Maarseveen Date: Wed, 12 Aug 2026 20:55:33 +0200 Subject: [PATCH 765/857] lib/tests: add KUnit tests for errseq The errseq_t infrastructure (lib/errseq.c) underpins writeback error reporting but has no regression tests. Its semantics are subtle enough to have needed fixing before: commit b4678df184b3 ("errseq: Always report a writeback error once") changed how unseen errors reach new samplers. Add a KUnit suite covering the documented single-threaded semantics: - a zeroed errseq_t is the "no error yet" epoch - errors are recorded, overwrite one another, and both ends of the valid errno range round-trip exactly - an error nobody has seen samples as zero, so a check against a fresh sample still reports it - errseq_check_and_advance() reports a given error exactly once per cursor and leaves the cursor in place when nothing has changed - once an error has been seen, a fresh sample is current and a check against it reports nothing - the same error recorded again after being seen is reported again, even to a cursor that consumed the first occurrence while another cursor marked the repeat as seen - independent cursors each observe each error The lockless behaviour of errseq_t under concurrent updates and the WARN path for invalid error values are deliberately out of scope. Tested with ./tools/testing/kunit/kunit.py run, with a kunitconfig enabling CONFIG_KUNIT=y and CONFIG_ERRSEQ_KUNIT_TEST=y; all 13 tests pass under ARCH=um. Link: https://lore.kernel.org/20260812-errseq-kunit-v1-1-312be4c3aa0d@gmail.com Signed-off-by: Thomas Maarseveen Acked-by: Jeff Layton Cc: David Gow Signed-off-by: Andrew Morton --- MAINTAINERS | 1 + lib/Kconfig.debug | 15 +++ lib/tests/Makefile | 1 + lib/tests/errseq_kunit.c | 237 +++++++++++++++++++++++++++++++++++++++ 4 files changed, 254 insertions(+) create mode 100644 lib/tests/errseq_kunit.c diff --git a/MAINTAINERS b/MAINTAINERS index 2133aec4a20046..366ae4b9eb888d 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -9698,6 +9698,7 @@ M: Jeff Layton S: Maintained F: include/linux/errseq.h F: lib/errseq.c +F: lib/tests/errseq_kunit.c ESD CAN NETWORK DRIVERS M: Stefan Mätje diff --git a/lib/Kconfig.debug b/lib/Kconfig.debug index 134b15a44625e8..c3f448f3b8f13c 100644 --- a/lib/Kconfig.debug +++ b/lib/Kconfig.debug @@ -2797,6 +2797,21 @@ config SYSCTL_KUNIT_TEST If unsure, say N. +config ERRSEQ_KUNIT_TEST + tristate "KUnit test for errseq" if !KUNIT_ALL_TESTS + depends on KUNIT + default KUNIT_ALL_TESTS + help + This builds the errseq KUnit test suite. + It tests the documented semantics of the errseq_t error-tracking + infrastructure (lib/errseq.c), which underpins writeback error + reporting. + + For more information on KUnit and unit tests in general please refer + to the KUnit documentation in Documentation/dev-tools/kunit/. + + If unsure, say N. + config KFIFO_KUNIT_TEST tristate "KUnit Test for the generic kernel FIFO implementation" if !KUNIT_ALL_TESTS depends on KUNIT diff --git a/lib/tests/Makefile b/lib/tests/Makefile index 3cac3b63a7522c..8e11b125433bf1 100644 --- a/lib/tests/Makefile +++ b/lib/tests/Makefile @@ -13,6 +13,7 @@ obj-$(CONFIG_BLACKHOLE_DEV_KUNIT_TEST) += blackhole_dev_kunit.o obj-$(CONFIG_CHECKSUM_KUNIT) += checksum_kunit.o obj-$(CONFIG_CMDLINE_KUNIT_TEST) += cmdline_kunit.o obj-$(CONFIG_CPUMASK_KUNIT_TEST) += cpumask_kunit.o +obj-$(CONFIG_ERRSEQ_KUNIT_TEST) += errseq_kunit.o obj-$(CONFIG_FFS_KUNIT_TEST) += ffs_kunit.o CFLAGS_fortify_kunit.o += $(call cc-disable-warning, unsequenced) CFLAGS_fortify_kunit.o += $(call cc-disable-warning, stringop-overread) diff --git a/lib/tests/errseq_kunit.c b/lib/tests/errseq_kunit.c new file mode 100644 index 00000000000000..8f39ebc4a2488e --- /dev/null +++ b/lib/tests/errseq_kunit.c @@ -0,0 +1,237 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * KUnit tests for the errseq_t error-tracking infrastructure. + * + * These exercise the documented single-threaded semantics of the errseq + * API (see Documentation/core-api/errseq.rst and lib/errseq.c): error + * recording and overwriting, the "seen" handoff between errseq_sample() + * and errseq_check_and_advance(), and the re-reporting of an error that + * is recorded again after it has been seen. + * + * The lockless properties of errseq_t under concurrent updates are + * outside the scope of these deterministic tests, as is the WARN path + * for invalid error values. + */ +#include + +#include +#include +#include + +/* + * A zeroed errseq_t is the "no error has ever occurred" epoch: it + * samples as zero and no check against it reports anything. + */ +static void errseq_test_zero_epoch_reports_no_error(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + + KUNIT_EXPECT_EQ(test, errseq_sample(&eseq), 0); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, 0), 0); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); + KUNIT_EXPECT_EQ(test, since, 0); +} + +static void errseq_test_set_records_error(struct kunit *test) +{ + errseq_t eseq = 0; + + /* errseq_set() returns the previous value; the epoch is zero. */ + KUNIT_EXPECT_EQ(test, errseq_set(&eseq, -EIO), 0); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, 0), -EIO); +} + +/* Any error set always overwrites an existing error. */ +static void errseq_test_set_overwrites_error(struct kunit *test) +{ + errseq_t eseq = 0; + + errseq_set(&eseq, -EIO); + errseq_set(&eseq, -ENOSPC); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, 0), -ENOSPC); +} + +/* Both ends of the valid error range are recorded exactly. */ +static void errseq_test_errno_range_extremes(struct kunit *test) +{ + errseq_t lo = 0; + errseq_t hi = 0; + + errseq_set(&lo, -1); + KUNIT_EXPECT_EQ(test, errseq_check(&lo, 0), -1); + + errseq_set(&hi, -MAX_ERRNO); + KUNIT_EXPECT_EQ(test, errseq_check(&hi, 0), -MAX_ERRNO); +} + +/* + * An error nobody has seen yet samples as zero, so that a check against + * the sample still reports it (see commit b4678df184b3 ("errseq: Always + * report a writeback error once")). + */ +static void errseq_test_sample_of_unseen_error_is_zero(struct kunit *test) +{ + errseq_t eseq = 0; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_sample(&eseq), 0); +} + +static void errseq_test_new_sampler_sees_unseen_error(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since; + + errseq_set(&eseq, -EIO); + since = errseq_sample(&eseq); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, since), -EIO); +} + +/* A given error is reported exactly once per advancing cursor. */ +static void errseq_test_check_and_advance_reports_once(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = errseq_sample(&eseq); + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); +} + +/* + * Once an error has been seen, a fresh sample is non-zero and checking + * against it reports nothing: handled errors do not reach new samplers. + */ +static void errseq_test_sample_after_seen_is_current(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + errseq_t sample; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + sample = errseq_sample(&eseq); + KUNIT_EXPECT_NE(test, sample, 0); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, sample), 0); +} + +static void errseq_test_new_error_after_advance(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + errseq_set(&eseq, -ENOSPC); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -ENOSPC); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); +} + +/* + * Recording the same error again after it has been seen must bump the + * sequence, so cursors that consumed the first occurrence see the + * second one too. + */ +static void errseq_test_same_error_reported_again_after_seen(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + errseq_t seen_cursor; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + seen_cursor = since; + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, since), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + /* The repeat must advance the sequence, not just re-toggle "seen". */ + KUNIT_EXPECT_NE(test, since, seen_cursor); +} + +/* + * A cursor that consumed an error must still observe a repeat of that + * error even when another cursor has already marked the repeat seen: + * recording over a seen value must advance the sequence. + */ +static void errseq_test_repeat_error_visible_to_all_cursors(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t cursor_a = 0; + errseq_t cursor_b = 0; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), -EIO); + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_b), -EIO); + + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, cursor_a), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), 0); +} + +/* An advance with no new error reports nothing and leaves the cursor put. */ +static void errseq_test_advance_stable_when_unchanged(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + errseq_t cursor; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + cursor = since; + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); + KUNIT_EXPECT_EQ(test, since, cursor); +} + +/* + * Cursors are independent: one subscriber consuming an error does not + * consume it for another, and each subscriber sees each error once. + */ +static void errseq_test_two_subscribers_independent(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t cursor_a = errseq_sample(&eseq); + errseq_t cursor_b = errseq_sample(&eseq); + + errseq_set(&eseq, -EIO); + + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, cursor_b), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_b), -EIO); + + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), 0); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_b), 0); +} + +static struct kunit_case errseq_test_cases[] = { + KUNIT_CASE(errseq_test_zero_epoch_reports_no_error), + KUNIT_CASE(errseq_test_set_records_error), + KUNIT_CASE(errseq_test_set_overwrites_error), + KUNIT_CASE(errseq_test_errno_range_extremes), + KUNIT_CASE(errseq_test_sample_of_unseen_error_is_zero), + KUNIT_CASE(errseq_test_new_sampler_sees_unseen_error), + KUNIT_CASE(errseq_test_check_and_advance_reports_once), + KUNIT_CASE(errseq_test_sample_after_seen_is_current), + KUNIT_CASE(errseq_test_new_error_after_advance), + KUNIT_CASE(errseq_test_same_error_reported_again_after_seen), + KUNIT_CASE(errseq_test_repeat_error_visible_to_all_cursors), + KUNIT_CASE(errseq_test_advance_stable_when_unchanged), + KUNIT_CASE(errseq_test_two_subscribers_independent), + {} +}; + +static struct kunit_suite errseq_test_suite = { + .name = "errseq", + .test_cases = errseq_test_cases, +}; + +kunit_test_suite(errseq_test_suite); + +MODULE_DESCRIPTION("KUnit tests for the errseq infrastructure"); +MODULE_LICENSE("GPL"); From 0f8dedfe850e906e550bc3065d0efff1ad51c319 Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Mon, 31 Aug 2026 14:22:48 -0700 Subject: [PATCH 766/857] xor: add missing vzeroupper to AVX code Since the AVX optimized XOR code uses YMM registers, execute vzeroupper before returning from it. This is needed to avoid degrading the performance of any later SSE code that may happen to be executed. Link: https://lore.kernel.org/20260831212248.213805-1-ebiggers@kernel.org Fixes: ea4d26ae24e5 ("raid5: add AVX optimized RAID5 checksumming") Signed-off-by: Eric Biggers Cc: Christoph Hellwig Cc: Signed-off-by: Andrew Morton --- lib/raid/xor/x86/xor-avx.c | 1 + 1 file changed, 1 insertion(+) diff --git a/lib/raid/xor/x86/xor-avx.c b/lib/raid/xor/x86/xor-avx.c index f7777d7aa269bd..95b21e7225e8d7 100644 --- a/lib/raid/xor/x86/xor-avx.c +++ b/lib/raid/xor/x86/xor-avx.c @@ -147,6 +147,7 @@ static void xor_gen_avx(void *dest, void **srcs, unsigned int src_cnt, { kernel_fpu_begin(); xor_gen_avx_inner(dest, srcs, src_cnt, bytes); + asm volatile("vzeroupper"); kernel_fpu_end(); } From c4daa4162a6c03a65d0a3d6068082466725ec267 Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Mon, 31 Aug 2026 14:23:08 -0700 Subject: [PATCH 767/857] raid6: add missing vzeroupper to AVX2 code Since the AVX2 optimized RAID6 code uses YMM registers, execute vzeroupper before returning from it. This is needed to avoid degrading the performance of any later SSE code that may happen to be executed. Link: https://lore.kernel.org/20260831212308.213855-1-ebiggers@kernel.org Fixes: 2c935842bdb4 ("lib/raid6: Add AVX2 optimized gen_syndrome functions") Fixes: 7056741fd9fc ("lib/raid6: Add AVX2 optimized recovery functions") Signed-off-by: Eric Biggers Cc: Christoph Hellwig Cc: Signed-off-by: Andrew Morton --- lib/raid/raid6/x86/avx2.c | 6 ++++++ lib/raid/raid6/x86/recov_avx2.c | 2 ++ 2 files changed, 8 insertions(+) diff --git a/lib/raid/raid6/x86/avx2.c b/lib/raid/raid6/x86/avx2.c index 7d829c669ea795..3cc2fe7ac42c57 100644 --- a/lib/raid/raid6/x86/avx2.c +++ b/lib/raid/raid6/x86/avx2.c @@ -67,6 +67,7 @@ static void raid6_avx21_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -115,6 +116,7 @@ static void raid6_avx21_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -175,6 +177,7 @@ static void raid6_avx22_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -243,6 +246,7 @@ static void raid6_avx22_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -334,6 +338,7 @@ static void raid6_avx24_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -444,6 +449,7 @@ static void raid6_avx24_xor_syndrome(int disks, int start, int stop, asm volatile("vmovntdq %%ymm14,%0" : "=m" (q[d+96])); } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } diff --git a/lib/raid/raid6/x86/recov_avx2.c b/lib/raid/raid6/x86/recov_avx2.c index a714a780a2d8f6..820871046e3058 100644 --- a/lib/raid/raid6/x86/recov_avx2.c +++ b/lib/raid/raid6/x86/recov_avx2.c @@ -176,6 +176,7 @@ static void raid6_2data_recov_avx2(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -293,6 +294,7 @@ static void raid6_datap_recov_avx2(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } From a7cf810b68a060e3924986e0855b452ff8e67c62 Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Mon, 31 Aug 2026 14:23:16 -0700 Subject: [PATCH 768/857] raid6: add missing vzeroupper to AVX-512 code Since the AVX-512 optimized RAID6 code uses ZMM registers, execute vzeroupper before returning from it. This is needed to avoid degrading the performance of any later SSE code that may happen to be executed. Link: https://lore.kernel.org/20260831212316.213896-1-ebiggers@kernel.org Fixes: e0a491c12968 ("lib/raid6: Add AVX512 optimized gen_syndrome functions") Fixes: 13c520b2993c ("lib/raid6: Add AVX512 optimized recovery functions") Signed-off-by: Eric Biggers Cc: Christoph Hellwig Cc: Signed-off-by: Andrew Morton --- lib/raid/raid6/x86/avx512.c | 6 ++++++ lib/raid/raid6/x86/recov_avx512.c | 2 ++ 2 files changed, 8 insertions(+) diff --git a/lib/raid/raid6/x86/avx512.c b/lib/raid/raid6/x86/avx512.c index e671eb5bde63e4..772bfc4af6dfd7 100644 --- a/lib/raid/raid6/x86/avx512.c +++ b/lib/raid/raid6/x86/avx512.c @@ -78,6 +78,7 @@ static void raid6_avx5121_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -137,6 +138,7 @@ static void raid6_avx5121_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -208,6 +210,7 @@ static void raid6_avx5122_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -292,6 +295,7 @@ static void raid6_avx5122_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -396,6 +400,7 @@ static void raid6_avx5124_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -529,6 +534,7 @@ static void raid6_avx5124_xor_syndrome(int disks, int start, int stop, "m" (q[d+128]), "m" (q[d+192])); } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } const struct raid6_calls raid6_avx512x4 = { diff --git a/lib/raid/raid6/x86/recov_avx512.c b/lib/raid/raid6/x86/recov_avx512.c index ec72d5a30c01ef..299a3f044d6162 100644 --- a/lib/raid/raid6/x86/recov_avx512.c +++ b/lib/raid/raid6/x86/recov_avx512.c @@ -211,6 +211,7 @@ static void raid6_2data_recov_avx512(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -353,6 +354,7 @@ static void raid6_datap_recov_avx512(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } From 3f693883df2337053a3e6d47f56013d8fb0282ca Mon Sep 17 00:00:00 2001 From: Hrushiraj Gandhi Date: Mon, 31 Aug 2026 20:19:53 +0530 Subject: [PATCH 769/857] gcov: use strscpy() instead of strcpy() in init_node() node->name is a flexible array member sized to exactly strlen(name) + 1 bytes at allocation time in new_node(), so this copy can never actually overflow. Still, prefer the bounded strscpy() over strcpy() on general principle; pass the same strlen(name) + 1 bound the allocation used, since sizeof() cannot be applied to a flexible array member. No functional change. Link: https://lore.kernel.org/20260831144953.324441-1-hrushirajg23@gmail.com Signed-off-by: Hrushiraj Gandhi Reviewed-by: Bradley Morgan Cc: Peter Oberparleiter Signed-off-by: Andrew Morton --- kernel/gcov/fs.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/gcov/fs.c b/kernel/gcov/fs.c index 1d19b1be207a7d..764918570de13d 100644 --- a/kernel/gcov/fs.c +++ b/kernel/gcov/fs.c @@ -529,7 +529,7 @@ static void init_node(struct gcov_node *node, struct gcov_info *info, } node->parent = parent; if (name) - strcpy(node->name, name); + strscpy(node->name, name, strlen(name) + 1); } /* From aa4b37d1fc2134fbbb85db782893889ebe37fb06 Mon Sep 17 00:00:00 2001 From: Mete Durlu Date: Mon, 31 Aug 2026 11:57:15 +0200 Subject: [PATCH 770/857] panic: introduce arch_do_panic Patch series "Introduce arch_do_panic", v6. Replace architecture-specific ifdef sections in vpanic() with a clean arch_do_panic() hook. Currently s390 and sparc embed their panic handlers directly in vpanic() using preprocessor conditionals, making the common code path harder to maintain. Introduce arch_do_panic() as an architecture extension point called at the end of vpanic(). Architectures can use this hook to implement their specific panic handling without polluting the generic panic code. Remove s390s ifdef block in vpanic() and move the corresponding code block to s390s own arch_do_panic() implementation in architecture specific code. Move sparc panic handling from ifdef blocks to arch_do_panic(). Remove the preprocessor conditionals from vpanic() and place the Stop-A enablement code in architecture-specific files where it belongs. Stop-A enablement markers are now printed after "end Kernel panic" line. To me, there are no better alternatives other than setup.c to put sparc's arch_do_panic() implementation. The other files under arch/sparc/kernel are either divided to *_32.c and *_64.c variants, which mean code duplication, or unrelated. The cleanup reduces vpanic() complexity and establishes a pattern for other architectures needing custom panic behavior. No functional changes, only minor print order changes. This patch (of 3): Introduce a hook for architectures to put their specific panic handlers. s390 and sparc already have ifdef preprocessor checks to execute architecture specific code. Pave the way for vpanic() cleanup. Link: https://lore.kernel.org/20260831-arch_do_panic-v6-1-a1e170a9e7fd@linux.ibm.com Link: https://lore.kernel.org/all/20260730-arch_do_panic-v3-0-d5401e683cdb@linux.ibm.com/ [1] Signed-off-by: Mete Durlu Reviewed-by: Bradley Morgan Suggested-by: Sven Schnelle Reviewed-by: Andrew Morton Cc: Alexander Gordeev Cc: Andreas Larsson Cc: Christian Borntraeger Cc: David S. Miller Cc: Heiko Carstens Cc: Petr Mladek Cc: Vasily Gorbik Signed-off-by: Andrew Morton --- include/linux/panic.h | 2 ++ kernel/panic.c | 3 +++ 2 files changed, 5 insertions(+) diff --git a/include/linux/panic.h b/include/linux/panic.h index f1dd417e54b294..98dd7dfd27de7a 100644 --- a/include/linux/panic.h +++ b/include/linux/panic.h @@ -110,4 +110,6 @@ extern void add_taint(unsigned flag, enum lockdep_ok); extern int test_taint(unsigned flag); extern unsigned long get_taint(void); +void arch_do_panic(void); + #endif /* _LINUX_PANIC_H */ diff --git a/kernel/panic.c b/kernel/panic.c index 213725b612aa11..726a978422326f 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -567,6 +567,8 @@ static void panic_other_cpus_shutdown(bool crash_kexec) crash_smp_send_stop(); } +void __weak arch_do_panic(void) {} + /** * vpanic - halt the system * @fmt: The text string to print @@ -756,6 +758,7 @@ void vpanic(const char *fmt, va_list args) #endif pr_emerg("---[ end Kernel panic - not syncing: %s ]---\n", buf); + arch_do_panic(); /* Do not scroll important messages printed above */ suppress_printk = 1; From a3d815a05a39e74f515c1e5ce4f4ccc8373c6490 Mon Sep 17 00:00:00 2001 From: Mete Durlu Date: Mon, 31 Aug 2026 11:57:16 +0200 Subject: [PATCH 771/857] s390: implement arch_do_panic Implement s390 specific arch_do_panic() instead of using s390 specific ifdef sections in vpanic() code. disabled_wait() is now called after "end Kernel panic" marker. No functional changes. Link: https://lore.kernel.org/20260831-arch_do_panic-v6-2-a1e170a9e7fd@linux.ibm.com Signed-off-by: Mete Durlu Acked-by: Heiko Carstens Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Cc: Alexander Gordeev Cc: Andreas Larsson Cc: Christian Borntraeger Cc: David S. Miller Cc: Petr Mladek Cc: Sven Schnelle Cc: Vasily Gorbik Signed-off-by: Andrew Morton --- arch/s390/kernel/traps.c | 7 +++++++ kernel/panic.c | 3 --- 2 files changed, 7 insertions(+), 3 deletions(-) diff --git a/arch/s390/kernel/traps.c b/arch/s390/kernel/traps.c index b6ba4465f59dea..115cb337324769 100644 --- a/arch/s390/kernel/traps.c +++ b/arch/s390/kernel/traps.c @@ -26,6 +26,7 @@ #include #include #include +#include #include #include #include @@ -33,6 +34,7 @@ #include #include #include +#include #include "entry.h" struct pgm_stat { @@ -283,6 +285,11 @@ static void monitor_event_exception(struct pt_regs *regs) } } +void arch_do_panic(void) +{ + disabled_wait(); +} + void kernel_stack_invalid(struct pt_regs *regs) { /* diff --git a/kernel/panic.c b/kernel/panic.c index 726a978422326f..ee6e3f9e39002e 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -752,9 +752,6 @@ void vpanic(const char *fmt, va_list args) pr_emerg("Press Stop-A (L1-A) from sun keyboard or send break\n" "twice on console to return to the boot prom\n"); } -#endif -#if defined(CONFIG_S390) - disabled_wait(); #endif pr_emerg("---[ end Kernel panic - not syncing: %s ]---\n", buf); From 8d68177c3befb59ebb333d3bf9ec1d9ac1b960e1 Mon Sep 17 00:00:00 2001 From: Mete Durlu Date: Mon, 31 Aug 2026 11:57:17 +0200 Subject: [PATCH 772/857] sparc: implement arch_do_panic Implement sparc specific arch_do_panic() instead of using sparc specific ifdef sections in vpanic() code. Reorder arch specific panic handling, sparc's Stop-A messages are now printed after "end Kernel panic" marker. Link: https://lore.kernel.org/20260831-arch_do_panic-v6-3-a1e170a9e7fd@linux.ibm.com Signed-off-by: Mete Durlu Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Cc: Alexander Gordeev Cc: Andreas Larsson Cc: Christian Borntraeger Cc: David S. Miller Cc: Heiko Carstens Cc: Petr Mladek Cc: Sven Schnelle Cc: Vasily Gorbik Signed-off-by: Andrew Morton --- arch/sparc/kernel/setup.c | 9 +++++++++ kernel/panic.c | 9 --------- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/arch/sparc/kernel/setup.c b/arch/sparc/kernel/setup.c index 4975867d9001b6..5f43cef8063825 100644 --- a/arch/sparc/kernel/setup.c +++ b/arch/sparc/kernel/setup.c @@ -2,6 +2,8 @@ #include #include +#include +#include static const struct ctl_table sparc_sysctl_table[] = { { @@ -36,6 +38,13 @@ static const struct ctl_table sparc_sysctl_table[] = { #endif }; +void arch_do_panic(void) +{ + /* Make sure the user can actually press Stop-A (L1-A) */ + stop_a_enabled = 1; + pr_emerg("Press Stop-A (L1-A) from sun keyboard or send break\n" + "twice on console to return to the boot prom\n"); +} static int __init init_sparc_sysctls(void) { diff --git a/kernel/panic.c b/kernel/panic.c index ee6e3f9e39002e..7dda841c16f9cc 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -744,15 +744,6 @@ void vpanic(const char *fmt, va_list args) reboot_mode = panic_reboot_mode; emergency_restart(); } -#ifdef __sparc__ - { - extern int stop_a_enabled; - /* Make sure the user can actually press Stop-A (L1-A) */ - stop_a_enabled = 1; - pr_emerg("Press Stop-A (L1-A) from sun keyboard or send break\n" - "twice on console to return to the boot prom\n"); - } -#endif pr_emerg("---[ end Kernel panic - not syncing: %s ]---\n", buf); arch_do_panic(); From 3d282cc5037bf4cfc4659048f98d345e5b53dac7 Mon Sep 17 00:00:00 2001 From: Ivy Lopez Date: Mon, 31 Aug 2026 19:41:38 -0600 Subject: [PATCH 773/857] lib: decompress_unxz: make it obvious that there is no memory leak Calling __decompress() or unxz() with fill == NULL && flush == NULL && in == NULL is invalid, thus there were no memory leaks even though it might have looked like that. Move the conditional free() calls so that it's obvious that there are no leaks. Link: https://lore.kernel.org/20260901014138.22699-1-skunkolee@gmail.com Signed-off-by: Ivy Lopez Closes: https://bugzilla.kernel.org/show_bug.cgi?id=207113 Link: https://lore.kernel.org/lkml/20241006072542.66442-2-t.v.s10123@gmail.com/T/ Link: https://lore.kernel.org/lkml/20260825191333.34276-1-skunkolee@gmail.com/T/ Reviewed-by: Lasse Collin Signed-off-by: Andrew Morton --- lib/decompress_unxz.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/lib/decompress_unxz.c b/lib/decompress_unxz.c index 05d5cb490a44e7..9ccded9934c667 100644 --- a/lib/decompress_unxz.c +++ b/lib/decompress_unxz.c @@ -342,13 +342,13 @@ STATIC int INIT unxz(unsigned char *in, long in_size, b.out_pos = 0; } } while (ret == XZ_OK); + } - if (must_free_in) - free(in); + if (must_free_in) + free(in); - if (flush != NULL) - free(b.out); - } + if (flush != NULL) + free(b.out); if (in_used != NULL) *in_used += b.in_pos; From 73a6a5261be409209c62f76c6f6c753fc01290e4 Mon Sep 17 00:00:00 2001 From: Julian Braha Date: Mon, 31 Aug 2026 00:00:16 +0100 Subject: [PATCH 774/857] arch/Kconfig: fix dead conditions by removing dead options These two 'int' options: ARCH_MMAP_RND_BITS_DEFAULT ARCH_MMAP_RND_COMPAT_BITS_DEFAULT are used directly as conditions for defaults. 'int' options should not be used as conditions, because they will always evaluate to false. Let's remove these options, because they are not used anywhere else. This dead code was found by kconfirm, a static analysis tool for Kconfig. Link: https://lore.kernel.org/20260830230016.2730093-1-julianbraha@gmail.com Signed-off-by: Julian Braha Reviewed-by: Arnd Bergmann Reviewed-by: Jinjie Ruan Signed-off-by: Andrew Morton --- arch/Kconfig | 8 -------- 1 file changed, 8 deletions(-) diff --git a/arch/Kconfig b/arch/Kconfig index 45c65777236231..72890200d049cc 100644 --- a/arch/Kconfig +++ b/arch/Kconfig @@ -1238,13 +1238,9 @@ config ARCH_MMAP_RND_BITS_MIN config ARCH_MMAP_RND_BITS_MAX int -config ARCH_MMAP_RND_BITS_DEFAULT - int - config ARCH_MMAP_RND_BITS int "Number of bits to use for ASLR of mmap base address" if EXPERT range ARCH_MMAP_RND_BITS_MIN ARCH_MMAP_RND_BITS_MAX - default ARCH_MMAP_RND_BITS_DEFAULT if ARCH_MMAP_RND_BITS_DEFAULT default ARCH_MMAP_RND_BITS_MIN depends on HAVE_ARCH_MMAP_RND_BITS help @@ -1272,13 +1268,9 @@ config ARCH_MMAP_RND_COMPAT_BITS_MIN config ARCH_MMAP_RND_COMPAT_BITS_MAX int -config ARCH_MMAP_RND_COMPAT_BITS_DEFAULT - int - config ARCH_MMAP_RND_COMPAT_BITS int "Number of bits to use for ASLR of mmap base address for compatible applications" if EXPERT range ARCH_MMAP_RND_COMPAT_BITS_MIN ARCH_MMAP_RND_COMPAT_BITS_MAX - default ARCH_MMAP_RND_COMPAT_BITS_DEFAULT if ARCH_MMAP_RND_COMPAT_BITS_DEFAULT default ARCH_MMAP_RND_COMPAT_BITS_MIN depends on HAVE_ARCH_MMAP_RND_COMPAT_BITS help From b3625f5abefb5ec52a87fb8cce8bbba1db9cc158 Mon Sep 17 00:00:00 2001 From: Yuntao Wang Date: Tue, 11 Aug 2026 20:18:30 +0800 Subject: [PATCH 775/857] dyndbg: fix incorrect mod_ct value in dynamic_debug_init() Patch series "dyndbg: fix incorrect mod_ct value in dynamic_debug_init()". Fix and clean up dynamic_debug_init(). This patch (of 2): Suppose all `struct _ddebug` instances belong to the same module, mod_ct should be 1, but it is currently 0. mod_ct is incremented only when iter->modname changes, i.e. when the loop encounters the first _ddebug entry of a new module: if (strcmp(modname, iter->modname)) { mod_ct++; ... } If all _ddebug entries belong to the same module, strcmp() never returns nonzero, so mod_ct remains 0. However, the last (and in this case only) module is added after the loop: di.num_descs = mod_sites; di.descs = iter_mod_start; ret = ddebug_add_module(&di, modname); Thus, mod_ct should be incremented before adding this final module. The bug only affects the diagnostic message printed by vpr_info(): "%d prdebugs in %d modules, ..." It reports one fewer module than the actual number of modules. There is no userspace-visible runtime effect; the dynamic debug tables themselves are initialized correctly. Fix it. Link: https://lore.kernel.org/20260811121831.577848-1-yuntao.wang@linux.dev Link: https://lore.kernel.org/20260811121831.577848-2-yuntao.wang@linux.dev Signed-off-by: Yuntao Wang Cc: Jason Baron Cc: Jim Cromie Signed-off-by: Andrew Morton --- lib/dynamic_debug.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/dynamic_debug.c b/lib/dynamic_debug.c index 18a71a9108d3e5..16fad5454d6a50 100644 --- a/lib/dynamic_debug.c +++ b/lib/dynamic_debug.c @@ -1456,6 +1456,8 @@ static int __init dynamic_debug_init(void) iter_mod_start = iter; } } + + mod_ct++; di.num_descs = mod_sites; di.descs = iter_mod_start; ret = ddebug_add_module(&di, modname); From 037683608ea82391a45ea27b9a084310d521b03b Mon Sep 17 00:00:00 2001 From: Yuntao Wang Date: Tue, 11 Aug 2026 20:18:31 +0800 Subject: [PATCH 776/857] dyndbg: clean up dynamic_debug_init() to improve readability Keep variable assignments in the same order throughout the function to make the code easier to follow. No functional changes. Link: https://lore.kernel.org/20260811121831.577848-3-yuntao.wang@linux.dev Signed-off-by: Yuntao Wang Cc: Jason Baron Cc: Jim Cromie Signed-off-by: Andrew Morton --- lib/dynamic_debug.c | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/lib/dynamic_debug.c b/lib/dynamic_debug.c index 16fad5454d6a50..49334d1aa4b3af 100644 --- a/lib/dynamic_debug.c +++ b/lib/dynamic_debug.c @@ -1442,28 +1442,29 @@ static int __init dynamic_debug_init(void) i = mod_sites = mod_ct = 0; for (; iter < __stop___dyndbg; iter++, i++, mod_sites++) { - if (strcmp(modname, iter->modname)) { - mod_ct++; - di.num_descs = mod_sites; di.descs = iter_mod_start; + di.num_descs = mod_sites; ret = ddebug_add_module(&di, modname); if (ret) goto out_err; - mod_sites = 0; - modname = iter->modname; + mod_ct++; + iter_mod_start = iter; + modname = iter->modname; + mod_sites = 0; } } - mod_ct++; - di.num_descs = mod_sites; di.descs = iter_mod_start; + di.num_descs = mod_sites; ret = ddebug_add_module(&di, modname); if (ret) goto out_err; + mod_ct++; + ddebug_init_success = 1; vpr_info("%d prdebugs in %d modules, %d KiB in ddebug tables, %d kiB in __dyndbg section\n", i, mod_ct, (int)((mod_ct * sizeof(struct ddebug_table)) >> 10), From 9e27597f5a7f5196784bc123e3205ff7051897f4 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Fri, 10 Jul 2026 14:39:57 +0200 Subject: [PATCH 777/857] fork: honor task_struct's declared alignment Since commit cb7ca40a3882 ("x86/fpu: Make task_struct::thread constant size"), struct task_struct is declared __attribute__((aligned(64))) on all architectures. But fork_init() still sets the task_struct slab cache's alignment to align = max(L1_CACHE_BYTES, ARCH_MIN_TASKALIGN) which is smaller than 64 on architectures whose cache lines are below 64 bytes: e.g. 32 on ARMv5. In practice plain SLUB happens to hand out 64-byte-aligned objects anyway. With CONFIG_SLUB_DEBUG_ON the red-zone padding shifts objects to the requested alignment. With CONFIG_UBSAN_ALIGNMENT=y a boot on QEMU versatilepb (ARM926EJ-S, v7.2-rc2, gcc 13.3) floods the console with reports like: UBSAN: misaligned-access in include/linux/sched.h:2087:9 member access within misaligned address c295d7e0 for type 'struct task_struct' which requires 64 byte alignment CPU: 0 UID: 0 PID: 15 Comm: pr/ttyAMA-1 Not tainted 7.2.0-rc2 #1 VOLUNTARY Set the slab alignment to at least the type's declared alignment. Link: https://lore.kernel.org/20260710123957.31774-1-kmehltretter@gmail.com Fixes: cb7ca40a3882 ("x86/fpu: Make task_struct::thread constant size") Signed-off-by: Karl Mehltretter Reviewed-by: Bradley Morgan Cc: Ingo Molnar Cc: Karl Mehltretter Cc: Kees Cook Cc: Peter Zijlstra Cc: Vlastimil Babka Assisted-by: Claude:claude-fable-5 Cc: Signed-off-by: Andrew Morton --- kernel/fork.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/kernel/fork.c b/kernel/fork.c index 416758c8a3d431..fcc2f10e1faab7 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -858,7 +858,8 @@ void __init fork_init(void) #ifndef ARCH_MIN_TASKALIGN #define ARCH_MIN_TASKALIGN 0 #endif - int align = max_t(int, L1_CACHE_BYTES, ARCH_MIN_TASKALIGN); + int align = max3(L1_CACHE_BYTES, ARCH_MIN_TASKALIGN, + __alignof__(struct task_struct)); unsigned long useroffset, usersize; /* create a slab on which task_structs can be allocated */ From 41ba6fbe56cf05a4d14562acddde5745629c2bcb Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Sat, 1 Aug 2026 22:58:44 +0000 Subject: [PATCH 778/857] taskstats: fold the two pid/tgid handlers into one cmd_attr_pid() and cmd_attr_tgid() are copy paste. fold them into one handler that takes the attr type and fill function as parameters, same pattern as the cpumask fold in 59c0bc949c5e. No functional change. Link: https://lore.kernel.org/20260801225845.23855-1-include@grrlz.net Signed-off-by: Bradley Morgan Reviewed-by: Andrew Morton Cc: Balbir Singh Cc: Balbir Singh Signed-off-by: Andrew Morton --- kernel/taskstats.c | 47 ++++++++++++---------------------------------- 1 file changed, 12 insertions(+), 35 deletions(-) diff --git a/kernel/taskstats.c b/kernel/taskstats.c index 9a48827e22bce3..598d9cd8325018 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -473,7 +473,8 @@ static size_t taskstats_packet_size(void) return size; } -static int cmd_attr_pid(struct genl_info *info) +static int cmd_attr_pid_tgid(struct genl_info *info, int attr, + int (*fill)(pid_t, struct taskstats *)) { struct taskstats *stats; struct sk_buff *rep_skb; @@ -488,41 +489,15 @@ static int cmd_attr_pid(struct genl_info *info) return rc; rc = -EINVAL; - pid = nla_get_u32(info->attrs[TASKSTATS_CMD_ATTR_PID]); - stats = mk_reply(rep_skb, TASKSTATS_TYPE_PID, pid); + pid = nla_get_u32(info->attrs[attr]); + stats = mk_reply(rep_skb, + attr == TASKSTATS_CMD_ATTR_PID + ? TASKSTATS_TYPE_PID : TASKSTATS_TYPE_TGID, + pid); if (!stats) goto err; - rc = fill_stats_for_pid(pid, stats); - if (rc < 0) - goto err; - return send_reply(rep_skb, info); -err: - nlmsg_free(rep_skb); - return rc; -} - -static int cmd_attr_tgid(struct genl_info *info) -{ - struct taskstats *stats; - struct sk_buff *rep_skb; - size_t size; - u32 tgid; - int rc; - - size = taskstats_packet_size(); - - rc = prepare_reply(info, TASKSTATS_CMD_NEW, &rep_skb, size); - if (rc < 0) - return rc; - - rc = -EINVAL; - tgid = nla_get_u32(info->attrs[TASKSTATS_CMD_ATTR_TGID]); - stats = mk_reply(rep_skb, TASKSTATS_TYPE_TGID, tgid); - if (!stats) - goto err; - - rc = fill_stats_for_tgid(tgid, stats); + rc = fill(pid, stats); if (rc < 0) goto err; return send_reply(rep_skb, info); @@ -542,9 +517,11 @@ static int taskstats_user_cmd(struct sk_buff *skb, struct genl_info *info) TASKSTATS_CMD_ATTR_DEREGISTER_CPUMASK, DEREGISTER); else if (info->attrs[TASKSTATS_CMD_ATTR_PID]) - return cmd_attr_pid(info); + return cmd_attr_pid_tgid(info, TASKSTATS_CMD_ATTR_PID, + fill_stats_for_pid); else if (info->attrs[TASKSTATS_CMD_ATTR_TGID]) - return cmd_attr_tgid(info); + return cmd_attr_pid_tgid(info, TASKSTATS_CMD_ATTR_TGID, + fill_stats_for_tgid); else return -EINVAL; } From c529afc2f5b9aade2ec0bd54bc28186561e8aa66 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:18 +0800 Subject: [PATCH 779/857] ocfs2: restrict OCFS2_INVALID_SLOT suballoc slot to system inodes Patch series "ocfs2: validate suballoc slot and bit of metadata blocks", v3. The ocfs2 metadata validators trust the on-disk suballoc slot and bit without checking them against the slot range of the mounted filesystem and the capacity of the block group bitmap. A corrupted image can carry OCFS2_INVALID_SLOT or another out-of-range slot, or a suballoc bit beyond the bitmap, and once the corresponding inode, extent block, xattr block, dir index root or refcount block gets freed, the bad value goes straight into ocfs2_get_system_file_inode() or _ocfs2_free_suballoc_bits() and hits a BUG_ON() or runs off the end of local_system_inodes[]. This series rejects such values at read time, in the existing validators, so corrupted objects fail with -EROFS (and a read-only remount) instead of crashing: patch 1 restricts OCFS2_INVALID_SLOT dinodes to system inodes, completing fe7a283b3916 ("ocfs2: add suballoc slot check in ocfs2_validate_inode_block()"), and turns the "system file state is ambiguous" BUG_ON() in ocfs2_read_locked_inode() into an ocfs2_error(); patch 2 rejects oversized dinode suballoc bits; patch 3 validates the suballoc slot and bit of xattr and dir index blocks; patch 4 validates the suballoc slot and bit of extent and refcount blocks. The checks only enforce what the kernel and mkfs.ocfs2 already write: a valid slot from meta_ac->ac_alloc_slot and a bit within the block group bitmap, with system inodes carrying OCFS2_INVALID_SLOT plus OCFS2_SYSTEM_FL and extent blocks using slot 0. Nothing changes for healthy filesystems. Each new check was exercised under QEMU by corrupting the field in question with an out-of-range value; with the series applied the access fails with -EROFS and the filesystem remounts read-only instead of hitting the BUG_ON(). This patch (of 4): ocfs2_validate_inode_block() currently permits i_suballoc_slot to be OCFS2_INVALID_SLOT for any dinode. Only system inodes created by mkfs.ocfs2 are allocated from the global allocator and thus legitimately carry this value; regular inodes are always allocated from a per-slot suballocator and hence must have a valid slot. If a corrupted regular inode with OCFS2_INVALID_SLOT is accepted, ocfs2_remove_inode() will pass the slot to ocfs2_get_system_file_inode() and get_local_system_inode() will hit BUG_ON(slot == OCFS2_INVALID_SLOT) when the inode is deleted. This can be triggered by an unprivileged user unlinking such a corrupted file. Reject OCFS2_INVALID_SLOT for non-system dinodes during validation, while still accepting it for system inodes. Note that a crafted dinode carrying OCFS2_SYSTEM_FL passes the check above, yet a plain lookup of it still used to BUG() in ocfs2_read_locked_inode() ("system file state is ambiguous"). Since i_flags comes from disk, handle that mismatch with ocfs2_error() instead of BUG_ON() as well. Link: https://lore.kernel.org/20260901125221.1634686-1-joseph.qi@linux.alibaba.com Link: https://lore.kernel.org/20260901125221.1634686-2-joseph.qi@linux.alibaba.com Fixes: fe7a283b3916 ("ocfs2: add suballoc slot check in ocfs2_validate_inode_block()") Signed-off-by: Joseph Qi Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Cc: Signed-off-by: Andrew Morton --- fs/ocfs2/inode.c | 37 ++++++++++++++++++++++++++++--------- 1 file changed, 28 insertions(+), 9 deletions(-) diff --git a/fs/ocfs2/inode.c b/fs/ocfs2/inode.c index 180107a11046c5..9228d6ef23c24d 100644 --- a/fs/ocfs2/inode.c +++ b/fs/ocfs2/inode.c @@ -638,14 +638,18 @@ static int ocfs2_read_locked_inode(struct inode *inode, fe = (struct ocfs2_dinode *) bh->b_data; /* - * This is a code bug. Right now the caller needs to - * understand whether it is asking for a system file inode or - * not so the proper lock names can be built. + * The caller must know whether it is asking for a system file inode + * or not so the proper lock names can be built. Since i_flags comes + * from disk, a mismatch is filesystem corruption instead of a code + * bug, so handle it with ocfs2_error() rather than BUG_ON(). */ - mlog_bug_on_msg(!!(fe->i_flags & cpu_to_le32(OCFS2_SYSTEM_FL)) != - !!(args->fi_flags & OCFS2_FI_FLAG_SYSFILE), - "Inode %llu: system file state is ambiguous\n", - (unsigned long long)args->fi_blkno); + if (!!(fe->i_flags & cpu_to_le32(OCFS2_SYSTEM_FL)) != + !!(args->fi_flags & OCFS2_FI_FLAG_SYSFILE)) { + status = ocfs2_error(osb->sb, + "Inode %llu: system file state is ambiguous\n", + (unsigned long long)args->fi_blkno); + goto bail; + } if (S_ISCHR(le16_to_cpu(fe->i_mode)) || S_ISBLK(le16_to_cpu(fe->i_mode))) @@ -1520,8 +1524,23 @@ int ocfs2_validate_inode_block(struct super_block *sb, goto bail; } - if (le16_to_cpu(di->i_suballoc_slot) != (u16)OCFS2_INVALID_SLOT && - (u32)le16_to_cpu(di->i_suballoc_slot) > OCFS2_SB(sb)->max_slots - 1) { + /* + * Only system inodes created by mkfs.ocfs2 are allocated from the + * global allocator and thus legitimately carry OCFS2_INVALID_SLOT. + * Regular inodes are always allocated from a per-slot suballocator. + * If a regular inode with OCFS2_INVALID_SLOT was accepted here, + * deleting it would pass the slot to get_local_system_inode() via + * ocfs2_remove_inode() and trigger BUG_ON(slot == OCFS2_INVALID_SLOT). + */ + if (le16_to_cpu(di->i_suballoc_slot) == (u16)OCFS2_INVALID_SLOT) { + if (!(le32_to_cpu(di->i_flags) & OCFS2_SYSTEM_FL)) { + rc = ocfs2_error(sb, + "Invalid dinode %llu: suballoc slot %u for non-system inode\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(di->i_suballoc_slot)); + goto bail; + } + } else if ((u32)le16_to_cpu(di->i_suballoc_slot) > OCFS2_SB(sb)->max_slots - 1) { rc = ocfs2_error(sb, "Invalid dinode %llu: suballoc slot %u\n", (unsigned long long)bh->b_blocknr, le16_to_cpu(di->i_suballoc_slot)); From 7e0c47138df47af454755f95fd27aefb52c2c456 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:19 +0800 Subject: [PATCH 780/857] ocfs2: validate suballoc bit during inode read i_suballoc_bit of a dinode is currently not validated at all. A corrupted dinode can carry an abnormally large i_suballoc_bit, which bypasses ocfs2_validate_inode_block(). When the inode is deleted, ocfs2_remove_inode() calls ocfs2_free_dinode(), which passes the unvalidated bit to _ocfs2_free_suballoc_bits() and triggers BUG_ON((count + start_bit) > ocfs2_bits_per_group(cl)). A suballocator block group bitmap is contained in a single block and starts after the group descriptor header, so a valid suballoc bit must be smaller than the number of bits fitting in the remaining space. Reject oversized i_suballoc_bit values during dinode validation. The bound is derived from ocfs2_group_bitmap_size() so it is also tight when discontig_bg caps the suballocator bitmap at OCFS2_MAX_BG_BITMAP_SIZE. Note the above check alone is not sufficient since the freeing path compares the bit against ocfs2_bits_per_group(), which is derived from cl_cpg/cl_bpc of the allocator dinode that is not validated against the actual group capacity and can be artificially smaller on a corrupted image. Convert this BUG_ON in _ocfs2_free_suballoc_bits() to ocfs2_error() as well. Link: https://lore.kernel.org/20260901125221.1634686-3-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Signed-off-by: Andrew Morton --- fs/ocfs2/inode.c | 16 ++++++++++++++++ fs/ocfs2/ocfs2.h | 12 ++++++++++++ fs/ocfs2/suballoc.c | 18 +++++++++++++++--- 3 files changed, 43 insertions(+), 3 deletions(-) diff --git a/fs/ocfs2/inode.c b/fs/ocfs2/inode.c index 9228d6ef23c24d..92f3450010fbb0 100644 --- a/fs/ocfs2/inode.c +++ b/fs/ocfs2/inode.c @@ -1547,6 +1547,22 @@ int ocfs2_validate_inode_block(struct super_block *sb, goto bail; } + /* + * A suballocator block group bitmap is contained in a single block + * and starts after the group descriptor header, so a valid suballoc + * bit can never exceed ocfs2_suballoc_bits_per_block(). Otherwise + * deleting the inode will pass the oversized bit to + * _ocfs2_free_suballoc_bits() via ocfs2_free_dinode() and trigger + * BUG_ON((count + start_bit) > ocfs2_bits_per_group(cl)), since any + * group holds at most ocfs2_suballoc_bits_per_block() bits. + */ + if (le16_to_cpu(di->i_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + rc = ocfs2_error(sb, "Invalid dinode %llu: suballoc bit %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(di->i_suballoc_bit)); + goto bail; + } + if ((le32_to_cpu(di->i_flags) & OCFS2_ORPHANED_FL) && le16_to_cpu(di->i_orphaned_slot) >= OCFS2_SB(sb)->max_slots) { rc = ocfs2_error(sb, "Invalid dinode %llu: orphaned slot %u\n", diff --git a/fs/ocfs2/ocfs2.h b/fs/ocfs2/ocfs2.h index b747cdec178758..e3bb3cc0b25a29 100644 --- a/fs/ocfs2/ocfs2.h +++ b/fs/ocfs2/ocfs2.h @@ -593,6 +593,18 @@ static inline int ocfs2_supports_discontig_bg(struct ocfs2_super *osb) return 0; } +/* + * A suballocator block group bitmap starts right after the group + * descriptor header, so a suballoc bit can never exceed this number + * of bits. Derive it from ocfs2_group_bitmap_size() which also caps + * it at OCFS2_MAX_BG_BITMAP_SIZE when discontig_bg is enabled. + */ +static inline u32 ocfs2_suballoc_bits_per_block(struct super_block *sb) +{ + return ocfs2_group_bitmap_size(sb, 1, + OCFS2_SB(sb)->s_feature_incompat) * 8; +} + static inline unsigned int ocfs2_link_max(struct ocfs2_super *osb) { if (ocfs2_supports_indexed_dirs(osb)) diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index 453b56be9624c6..ce22d0c3d28748 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -3040,10 +3040,22 @@ static int _ocfs2_free_suballoc_bits(handle_t *handle, /* The alloc_bh comes from ocfs2_free_dinode() or * ocfs2_free_clusters(). The callers have all locked the * allocator and gotten alloc_bh from the lock call. This - * validates the dinode buffer. Any corruption that has happened - * is a code bug. */ + * validates the dinode buffer. */ BUG_ON(!OCFS2_IS_VALID_DINODE(fe)); - BUG_ON((count + start_bit) > ocfs2_bits_per_group(cl)); + + /* + * ocfs2_bits_per_group() is derived from cl_cpg and cl_bpc of the + * allocator dinode, which are not validated against the volume + * geometry. A corrupted image can carry a suballoc bit beyond it, + * so error out instead of crashing. + */ + if ((count + start_bit) > ocfs2_bits_per_group(cl)) { + return ocfs2_error(alloc_inode->i_sb, + "Allocator #%llu: freeing bits %u+%u exceeds bits per group %u\n", + (unsigned long long)le64_to_cpu(fe->i_blkno), + count, start_bit, + ocfs2_bits_per_group(cl)); + } trace_ocfs2_free_suballoc_bits( (unsigned long long)OCFS2_I(alloc_inode)->ip_blkno, From a08f83559463caf3bd3f16de2d5884e158b38e55 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:20 +0800 Subject: [PATCH 781/857] ocfs2: validate suballoc slot and bit of xattr and dir index blocks ocfs2_validate_xattr_block() and ocfs2_validate_dx_root() do not validate xb_suballoc_slot, xb_suballoc_bit, dr_suballoc_slot and dr_suballoc_bit at all. Since xattr blocks and dir index root blocks are allocated from a per-slot suballocator at runtime, their suballoc slots must be within range and their suballoc bits must fit in a block group bitmap. Otherwise a corrupted image can carry an out-of-range slot. When the xattr block or dir index is removed, ocfs2_xattr_block_remove() or ocfs2_dx_dir_remove_index() passes the unvalidated slot to ocfs2_get_system_file_inode() and get_local_system_inode() will either hit BUG_ON(slot == OCFS2_INVALID_SLOT) or compute an out-of-bounds index into the local_system_inodes array. Similarly an oversized suballoc bit will error out the filesystem in _ocfs2_free_suballoc_bits(). Furthermore ocfs2_validate_dx_root() does not verify dr_blkno against the physical block number like the extent and xattr block validators do, so a misplaced dir index root block can pass validation. Reject misplaced dir index root blocks, out-of-range suballoc slots and oversized suballoc bits during validation. Link: https://lore.kernel.org/20260901125221.1634686-4-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Signed-off-by: Andrew Morton --- fs/ocfs2/dir.c | 35 +++++++++++++++++++++++++++++++++++ fs/ocfs2/xattr.c | 25 +++++++++++++++++++++++++ 2 files changed, 60 insertions(+) diff --git a/fs/ocfs2/dir.c b/fs/ocfs2/dir.c index 0075e1624310e8..6bb6aa133f0150 100644 --- a/fs/ocfs2/dir.c +++ b/fs/ocfs2/dir.c @@ -605,6 +605,41 @@ static int ocfs2_validate_dx_root(struct super_block *sb, goto bail; } + if (le64_to_cpu(dx_root->dr_blkno) != bh->b_blocknr) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has an invalid dr_blkno of %llu\n", + (unsigned long long)bh->b_blocknr, + (unsigned long long)le64_to_cpu(dx_root->dr_blkno)); + goto bail; + } + + /* + * Dir index root blocks are allocated from a per-slot suballocator, + * so the slot must be in range. Otherwise removing the index passes + * it to get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(dx_root->dr_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has invalid dr_suballoc_slot %u\n", + (unsigned long long)le64_to_cpu(dx_root->dr_blkno), + le16_to_cpu(dx_root->dr_suballoc_slot)); + goto bail; + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise removing the index will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(dx_root->dr_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has invalid dr_suballoc_bit %u\n", + (unsigned long long)le64_to_cpu(dx_root->dr_blkno), + le16_to_cpu(dx_root->dr_suballoc_bit)); + goto bail; + } + if (!(dx_root->dr_flags & OCFS2_DX_FLAG_INLINE)) { struct ocfs2_extent_list *el = &dx_root->dr_list; diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 143d6f75f9c923..34f102db2a0e02 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -532,6 +532,31 @@ static int ocfs2_validate_xattr_block(struct super_block *sb, le32_to_cpu(xb->xb_fs_generation)); } + /* + * Xattr blocks are allocated from a per-slot suballocator, so the + * slot must be in range. Otherwise freeing the block passes it to + * get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(xb->xb_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + return ocfs2_error(sb, + "Extended attribute block #%llu has an invalid xb_suballoc_slot of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(xb->xb_suballoc_slot)); + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise freeing the block will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(xb->xb_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + return ocfs2_error(sb, + "Extended attribute block #%llu has an invalid xb_suballoc_bit of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(xb->xb_suballoc_bit)); + } + if (!(le16_to_cpu(xb->xb_flags) & OCFS2_XATTR_INDEXED)) { size_t region_offset = offsetof(struct ocfs2_xattr_block, xb_attrs.xb_header); From 8d8e46989f5584f63be42966cd5a682279d8502b Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:21 +0800 Subject: [PATCH 782/857] ocfs2: validate suballoc slot and bit of extent and refcount blocks ocfs2_validate_extent_block() and ocfs2_validate_refcount_block() do not validate h_suballoc_slot, h_suballoc_bit, rf_suballoc_slot and rf_suballoc_bit at all. Since extent blocks and refcount blocks are allocated from a per-slot suballocator at runtime, their suballoc slots must be within range and their suballoc bits must fit in a block group bitmap. Otherwise a corrupted image can carry an out-of-range slot. When the extent block is freed, ocfs2_cache_extent_block_free() caches it and ocfs2_free_cached_blocks() later passes the unvalidated slot to ocfs2_get_system_file_inode(); when the refcount block is freed, ocfs2_remove_refcount_extent() passes it via ocfs2_cache_block_dealloc(). get_local_system_inode() will then either hit BUG_ON(slot == OCFS2_INVALID_SLOT) or compute an out-of-bounds index into the local_system_inodes array. Similarly an oversized suballoc bit will error out the filesystem in _ocfs2_free_suballoc_bits(). Furthermore group descriptor validation only guarantees bg_bits within the physical bitmap size, so a corrupted image can still carry a suballoc bit beyond bg_bits, which would let ocfs2_block_group_clear_bits() clear bits beyond bg_bitmap. Convert the remaining BUG_ON against group->bg_bits in _ocfs2_free_suballoc_bits() to ocfs2_error() as well. Reject out-of-range suballoc slots and oversized suballoc bits during validation. Link: https://lore.kernel.org/20260901125221.1634686-5-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao Signed-off-by: Andrew Morton --- fs/ocfs2/alloc.c | 27 +++++++++++++++++++++++++++ fs/ocfs2/refcounttree.c | 27 +++++++++++++++++++++++++++ fs/ocfs2/suballoc.c | 15 ++++++++++++++- 3 files changed, 68 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/alloc.c b/fs/ocfs2/alloc.c index be09e766ac1fc9..2fdc5403b10b04 100644 --- a/fs/ocfs2/alloc.c +++ b/fs/ocfs2/alloc.c @@ -925,6 +925,33 @@ static int ocfs2_validate_extent_block(struct super_block *sb, goto bail; } + /* + * Extent blocks are allocated from a per-slot suballocator, so the + * slot must be in range. Otherwise freeing the block passes it to + * get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(eb->h_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + rc = ocfs2_error(sb, + "Extent block #%llu has an invalid h_suballoc_slot of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(eb->h_suballoc_slot)); + goto bail; + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise freeing the block will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(eb->h_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + rc = ocfs2_error(sb, + "Extent block #%llu has an invalid h_suballoc_bit of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(eb->h_suballoc_bit)); + goto bail; + } + if (le16_to_cpu(eb->h_list.l_count) != ocfs2_extent_recs_per_eb(sb)) { rc = ocfs2_error(sb, "Extent block #%llu has invalid l_count %u (expected %u)\n", diff --git a/fs/ocfs2/refcounttree.c b/fs/ocfs2/refcounttree.c index d9f22b4a265461..3e9cccf06e48cb 100644 --- a/fs/ocfs2/refcounttree.c +++ b/fs/ocfs2/refcounttree.c @@ -117,6 +117,33 @@ static int ocfs2_validate_refcount_block(struct super_block *sb, goto out; } + /* + * Refcount blocks are allocated from a per-slot suballocator, so the + * slot must be in range. Otherwise freeing the block passes it to + * get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(rb->rf_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + rc = ocfs2_error(sb, + "Refcount block #%llu has an invalid rf_suballoc_slot of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(rb->rf_suballoc_slot)); + goto out; + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise freeing the block will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(rb->rf_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + rc = ocfs2_error(sb, + "Refcount block #%llu has an invalid rf_suballoc_bit of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(rb->rf_suballoc_bit)); + goto out; + } + /* * rf_records (rl_count/rl_used/rl_recs[]) is only meaningful when * this block is not an interior tree block (OCFS2_REFCOUNT_TREE_FL); diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index ce22d0c3d28748..624152e4f7fedf 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -3070,7 +3070,20 @@ static int _ocfs2_free_suballoc_bits(handle_t *handle, } group = (struct ocfs2_group_desc *) group_bh->b_data; - BUG_ON((count + start_bit) > le16_to_cpu(group->bg_bits)); + /* + * Group descriptor validation only guarantees bg_bits within the + * physical bitmap size, so double check the freeing range here. + * Otherwise ocfs2_block_group_clear_bits() would clear bits beyond + * bg_bitmap. + */ + if ((count + start_bit) > le16_to_cpu(group->bg_bits)) { + status = ocfs2_error(alloc_inode->i_sb, + "Group descriptor #%llu has %u bits, cannot free bits %u+%u\n", + (unsigned long long)le64_to_cpu(group->bg_blkno), + le16_to_cpu(group->bg_bits), + count, start_bit); + goto bail; + } if (ocfs2_is_cluster_bitmap(alloc_inode)) old_bg_contig_free_bits = group->bg_contig_free_bits; From b906f9af7657fcaaad15bf34a19a668b1ad73cf3 Mon Sep 17 00:00:00 2001 From: YANXIN LI Date: Sat, 18 Jul 2026 19:03:07 +0000 Subject: [PATCH 783/857] ufs: use u64 for directory size in ufs_last_byte ufs1_read_inode() and ufs2_read_inode() copy the untrusted 64-bit on-disk size into inode->i_size. ufs_last_byte() then truncates that value to an unsigned int before calculating the valid extent of a directory folio. A directory size of exactly 4 GiB is truncated to zero. In ufs_find_entry(), subtracting the requested record length from that zero forms an endpoint almost 4 GiB beyond the mapped folio. A crafted UFS2 image produces: BUG: KASAN: use-after-free in ufs_find_entry+0x583/0x6e0 [ufs] Read of size 1 ... ufs_find_entry ufs_inode_by_name ufs_lookup __lookup_slow path_lookupat The KASAN classification reflects that the adjacent physical page was free; the source-level operation is an out-of-folio read. A controlled UFS1 test with CONFIG_UFS_FS_WRITE enabled also made ufs_find_entry() return a directory entry from an adjacent anonymous page. unlink() then passed that pointer to ufs_delete_entry(), which cleared the adjacent page's 32-bit d_ino. This write result reproduced three out of three times. UFS normally requires a privileged mount path. A relevant boundary is a privileged automounter or image-processing service handling an attacker-supplied filesystem. The write primitive additionally requires UFS1 to be mounted read-write with CONFIG_UFS_FS_WRITE enabled; the read is reachable with read-only UFS2. Keep the intermediate size arithmetic at 64 bits so non-final pages are bounded at PAGE_SIZE. With this change, the same controlled tests reach the real on-disk directory entry and produce no adjacent-page write or KASAN report. A reproducer and full validation logs are available on request. The vulnerable helper is present in v7.2-rc3, v7.1, v6.18.38, v6.12.95, v6.6.111, and upstream master at 1229e2e57a5c. Runtime reproduction and fix validation were performed in an x86-64 QEMU guest running v7.2-rc3 with generic KASAN. Signed-off-by: YANXIN LI Fixes: b71034e5e67d ("[PATCH] ufs: directory and page cache: from blocks to pages") Cc: Al Viro Cc: Christian Brauner Cc: Jan Kara Cc: Signed-off-by: Andrew Morton --- fs/ufs/dir.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/ufs/dir.c b/fs/ufs/dir.c index e62fe566767107..69e6d553510ee3 100644 --- a/fs/ufs/dir.c +++ b/fs/ufs/dir.c @@ -213,7 +213,7 @@ static void *ufs_get_folio(struct inode *dir, unsigned long n, static unsigned ufs_last_byte(struct inode *inode, unsigned long page_nr) { - unsigned last_byte = inode->i_size; + u64 last_byte = inode->i_size; last_byte -= page_nr << PAGE_SHIFT; if (last_byte > PAGE_SIZE) From aa30aa315f5f84e03669afabb65e3e5804cf979b Mon Sep 17 00:00:00 2001 From: Feng Tang Date: Wed, 2 Sep 2026 19:48:51 +0800 Subject: [PATCH 784/857] panic: remove the unneeded panic_print_get() panic_print_get() was introduced in commit 2683df6539cb ("panic: add note that 'panic_print' parameter is deprecated") to print out warning message of the deprecation of 'panic_print' on read access. Since commit 90f3c123247e ("panic: only warn about deprecated panic_print on write access"), panic_print_get() wrapper is not needed anymore for read access, so remove it and use param_get_ulong() instead. Link: https://lore.kernel.org/20260902114851.77062-1-feng.tang@linux.alibaba.com Signed-off-by: Feng Tang Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Cc: Petr Mladek Signed-off-by: Andrew Morton --- kernel/panic.c | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/kernel/panic.c b/kernel/panic.c index 7dda841c16f9cc..50715f14cf04ef 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -1216,14 +1216,9 @@ static int panic_print_set(const char *val, const struct kernel_param *kp) return param_set_ulong(val, kp); } -static int panic_print_get(char *val, const struct kernel_param *kp) -{ - return param_get_ulong(val, kp); -} - static const struct kernel_param_ops panic_print_ops = { .set = panic_print_set, - .get = panic_print_get, + .get = param_get_ulong, }; __core_param_cb(panic_print, &panic_print_ops, &panic_print, 0644); From ff936b726c981e362f02755b6417654dc4ef297a Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Thu, 18 Jun 2026 16:29:43 +0200 Subject: [PATCH 785/857] KMSAN: fix memset() when using fortify-source, again Both kmsan and fortify-source replace the memset function. When both are enabled at the same time, the kmsan version gets used, which triggers a warning about fortify-source being nonfunctional: warning: unsafe memset() usage lacked '__write_overflow' symbol in /home/arnd/arm-soc/lib/test_fortify/write_overflow-memset.c warning: unsafe memset() usage lacked '__write_overflow_field' symbol in /home/arnd/arm-soc/lib/test_fortify/write_overflow_field-memset.c Commit 78a498c3a227 already tried to address this, but this seems to only have worked for memcpy() and memmove() but not memset(), which is still lacking the macro definition when KMSAN is enabled. Remove the incorrect #ifndef check around the memset() macro. Fixes: ff901d80fff6 ("x86: kmsan: use __msan_ string functions where possible.") Fixes: 78a498c3a227 ("x86: fortify: kmsan: fix KMSAN fortify builds") Signed-off-by: Arnd Bergmann Link: https://patch.msgid.link/20260618142951.1739694-1-arnd@kernel.org Signed-off-by: Kees Cook --- include/linux/fortify-string.h | 2 -- 1 file changed, 2 deletions(-) diff --git a/include/linux/fortify-string.h b/include/linux/fortify-string.h index cf841dc71feffd..7e7c369e0a6cff 100644 --- a/include/linux/fortify-string.h +++ b/include/linux/fortify-string.h @@ -458,10 +458,8 @@ __FORTIFY_INLINE bool fortify_memset_chk(__kernel_size_t size, * __struct_size() vs __member_size() must be captured here to avoid * evaluating argument side-effects further into the macro layers. */ -#ifndef CONFIG_KMSAN #define memset(p, c, s) __fortify_memset_chk(p, c, s, \ __struct_size(p), __member_size(p)) -#endif /* * To make sure the compiler can enforce protection against buffer overflows, From 37eadcb919b02a742582de781097b455f7b16d76 Mon Sep 17 00:00:00 2001 From: Oleg Nesterov Date: Mon, 6 Apr 2026 15:37:32 +0200 Subject: [PATCH 786/857] signalfd: don't dequeue the forced fatal signals These signals should act like SIGKILL, in that userspace must never dequeue them. But as Kusaram explains, io_uring-driven signalfd_read_iter() called from get_signal() -> task_work_run() paths can do this before get_signal() has a chance to dequeue such a signal and notice SA_IMMUTABLE. Change signalfd_poll() and signalfd_dequeue() to add pending SA_IMMUTABLE signals to ctx->sigmask. TODO: we should probably change force_sig_info_to_task(HANDLER_EXIT) to make fatal_signal_pending() true, or add a fatal_or_forced_signal_pending() helper. Then signalfd_dequeue() could just return -EINTR in this case. This also makes sense for get_signal(), which could prioritize a fatal signal sent by (say) force_sig_seccomp(force_coredump => true), just like it already prioritizes SIGKILL. Cc: stable@kernel.org Reported-by: syzbot+0a4c46806941297fecb9@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=0a4c46806941297fecb9 Tested-by: syzbot+0a4c46806941297fecb9@syzkaller.appspotmail.com Link: https://lore.kernel.org/all/69d122fd.050a0220.2dbe29.001c.GAE@google.com/ Suggested-by: Kusaram Devineni Signed-off-by: Oleg Nesterov Reviewed-by: Kees Cook Link: https://patch.msgid.link/adO3HG8bvwRPcmte@redhat.com Signed-off-by: Kees Cook --- fs/signalfd.c | 28 ++++++++++++++++++++++------ 1 file changed, 22 insertions(+), 6 deletions(-) diff --git a/fs/signalfd.c b/fs/signalfd.c index dff53745e35241..22bc0870a824d8 100644 --- a/fs/signalfd.c +++ b/fs/signalfd.c @@ -48,17 +48,30 @@ static int signalfd_release(struct inode *inode, struct file *file) return 0; } +static void refine_sigmask(struct signalfd_ctx *ctx, sigset_t *sigmask) +{ + struct k_sigaction *k = current->sighand->action; + int n; + + *sigmask = ctx->sigmask; + for (n = 1; n <= _NSIG; ++n, ++k) { + if (k->sa.sa_flags & SA_IMMUTABLE) + sigaddset(sigmask, n); + } +} + static __poll_t signalfd_poll(struct file *file, poll_table *wait) { struct signalfd_ctx *ctx = file->private_data; __poll_t events = 0; + sigset_t sigmask; poll_wait(file, ¤t->sighand->signalfd_wqh, wait); spin_lock_irq(¤t->sighand->siglock); - if (next_signal(¤t->pending, &ctx->sigmask) || - next_signal(¤t->signal->shared_pending, - &ctx->sigmask)) + refine_sigmask(ctx, &sigmask); + if (next_signal(¤t->pending, &sigmask) || + next_signal(¤t->signal->shared_pending, &sigmask)) events |= EPOLLIN; spin_unlock_irq(¤t->sighand->siglock); @@ -155,11 +168,13 @@ static ssize_t signalfd_dequeue(struct signalfd_ctx *ctx, kernel_siginfo_t *info int nonblock) { enum pid_type type; - ssize_t ret; DECLARE_WAITQUEUE(wait, current); + sigset_t sigmask; + ssize_t ret; spin_lock_irq(¤t->sighand->siglock); - ret = dequeue_signal(&ctx->sigmask, info, &type); + refine_sigmask(ctx, &sigmask); + ret = dequeue_signal(&sigmask, info, &type); switch (ret) { case 0: if (!nonblock) @@ -174,7 +189,7 @@ static ssize_t signalfd_dequeue(struct signalfd_ctx *ctx, kernel_siginfo_t *info add_wait_queue(¤t->sighand->signalfd_wqh, &wait); for (;;) { set_current_state(TASK_INTERRUPTIBLE); - ret = dequeue_signal(&ctx->sigmask, info, &type); + ret = dequeue_signal(&sigmask, info, &type); if (ret != 0) break; if (signal_pending(current)) { @@ -184,6 +199,7 @@ static ssize_t signalfd_dequeue(struct signalfd_ctx *ctx, kernel_siginfo_t *info spin_unlock_irq(¤t->sighand->siglock); schedule(); spin_lock_irq(¤t->sighand->siglock); + refine_sigmask(ctx, &sigmask); } spin_unlock_irq(¤t->sighand->siglock); From 153c75d93cd59268f893f3ed66b62cc81428eef4 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Thu, 20 Aug 2026 20:58:06 -0700 Subject: [PATCH 787/857] xfs: fix media verification ioctl for internal rt volumes A media scan of a filesystem containing an internal rt volume produced an error in xfs_scrub phase 6 complaining about a truncated realtime device. The rt device wasn't truncated, but the media scan code thought we were trying to start a scan past the end of m_rtdev_targp. That in turn is an alias for m_ddev_targp, but in xfs_configure_buftarg we set nr_sectors to the size of the data section. We don't account for an internal realtime section, so the kernel doesn't scan any part of it. Oops. Reproducer: # mkfs.xfs -f /dev/sda -r zoned=1 -d rtinherit=1 # mount /dev/sda /mnt # dd if=/dev/zero of=/mnt/a bs=1024k count=100 # sync # xfs_info /mnt meta-data=/dev/sda isize=512 agcount=4, agsize=32768 blks = sectsz=512 attr=2, projid32bit=1 = crc=1 finobt=1, sparse=1, rmapbt=1 = reflink=0 bigtime=1 inobtcount=1 nrext64=1 = exchange=1 metadir=1 data = bsize=4096 blocks=131072, imaxpct=25 = sunit=0 swidth=0 blks naming =version 2 bsize=4096 ascii-ci=0, ftype=1, parent=1 log =internal log bsize=4096 blocks=16384, version=2 = sectsz=512 sunit=0 blks, lazy-count=1 realtime =internal extsz=4096 blocks=1114112, rtextents=1114112 = rgcount=17 rgsize=65536 extents = zoned=1 start=131072 reserved=53248 IOWS: 512M data volume, 3.1G internal rt section. Now let's try some media verification: # xfs_io -c 'verifymedia -d' -c 'verifymedia -r' /mnt verified 536870912/536870912 bytes at offset 0 512 MiB, 1 ops; 0.0496 sec (10.067 GiB/sec and 20.1345 ops/sec) verified 536870912/536870912 bytes at offset 0 512 MiB, 1 ops; 0.0409 sec (12.222 GiB/sec and 24.4439 ops/sec) Notice how xfs_io says we only verified 512M of the rt volume? If you run btrace in the background you'll see that we read the first 512M of the volume (aka the data section) twice and never read anything from the rt section. An earlier fix tried messing with the buftarg geometry, but I've decided on a more targetted fix for the media verification code. All we have to do is calculate the starting and ending daddr for the device that we're verifying, and clamp the user's input values to that range. This leads to some bogosity in the output reporting: # xfs_io -c 'verifymedia -d' -c 'verifymedia -r' /mnt/t verified 536870912/536870912 bytes at offset 0 512 MiB, 1 ops; 0.0606 sec (8.248 GiB/sec and 16.4968 ops/sec) verified 5100273664/5100273664 bytes at offset 0 4.750 GiB, 1 ops; 0.3329 sec (14.267 GiB/sec and 3.0035 ops/sec) Because we don't have a way to report that we didn't really do anything at all for that first 512M of address space of the rt "device". But at least we're no longer ignoring real media. (Note that the fsmap/bmap/fiemap calls all report physical addresses for the internal rt volume as offsets from the start of the data device, and the media verifier call consumes the same. We baked that into the user-visible behavior in 6.15, so we're stuck with that sparse hole at the beginning.) Cc: stable@vger.kernel.org # v6.15 Fixes: bdc03eb5f98f6f ("xfs: allow internal RT devices for zoned mode") Signed-off-by: Darrick J. Wong Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_verify_media.c | 24 +++++++++++++++++------- 1 file changed, 17 insertions(+), 7 deletions(-) diff --git a/fs/xfs/xfs_verify_media.c b/fs/xfs/xfs_verify_media.c index 5ead3976d51151..b75c81f8fcc037 100644 --- a/fs/xfs/xfs_verify_media.c +++ b/fs/xfs/xfs_verify_media.c @@ -268,6 +268,8 @@ xfs_verify_media( struct xfs_buftarg *btp = NULL; struct bio *bio; struct folio *folio; + xfs_daddr_t dev_start = 0; + xfs_daddr_t dev_end = 0; xfs_daddr_t daddr; uint64_t bbcount; int error = 0; @@ -277,24 +279,33 @@ xfs_verify_media( switch (me->me_dev) { case XFS_DEV_DATA: btp = mp->m_ddev_targp; + dev_end = XFS_FSB_TO_BB(mp, mp->m_sb.sb_dblocks); break; case XFS_DEV_LOG: - if (mp->m_logdev_targp != mp->m_ddev_targp) + if (mp->m_logdev_targp != mp->m_ddev_targp) { btp = mp->m_logdev_targp; + dev_end = XFS_FSB_TO_BB(mp, mp->m_sb.sb_logblocks); + } break; case XFS_DEV_RT: btp = mp->m_rtdev_targp; + dev_start = XFS_FSB_TO_BB(mp, mp->m_sb.sb_rtstart); + dev_end = XFS_FSB_TO_BB(mp, mp->m_sb.sb_rtstart + + mp->m_sb.sb_rblocks); break; } if (!btp) return -ENODEV; /* - * If the caller told us to verify beyond the end of the disk, tell the - * user exactly where that was. + * If the caller told us to verify before the start or beyond the end + * of the disk volume, tell the user exactly where the volume starts + * and ends. */ - if (me->me_end_daddr > btp->bt_nr_sectors) - me->me_end_daddr = btp->bt_nr_sectors; + if (me->me_end_daddr > dev_end) + me->me_end_daddr = dev_end; + if (me->me_start_daddr < dev_start) + me->me_start_daddr = dev_start; /* start and end have to be aligned to the lba size */ if (!IS_ALIGNED(BBTOB(me->me_start_daddr | me->me_end_daddr), @@ -323,8 +334,7 @@ xfs_verify_media( * verifying. */ daddr = me->me_start_daddr; - bbcount = min_t(sector_t, me->me_end_daddr, btp->bt_nr_sectors) - - me->me_start_daddr; + bbcount = me->me_end_daddr - me->me_start_daddr; folio = xfs_verify_alloc_folio(xfs_verify_iosize(me, btp, bbcount)); if (!folio) From d5ae1c0959420e536c9ac3a1a32f04c73f52bd09 Mon Sep 17 00:00:00 2001 From: Hans Holmberg Date: Wed, 26 Aug 2026 14:32:19 +0200 Subject: [PATCH 788/857] xfs: prevent race in zoned space reservations xfs_zoned_add_available() checks whether the reservation list is empty before adding blocks to the available-space counter. This check is not serialized against a task adding itself to the reservation list however. This allows the space provider to observe an empty list, after which a reserver can enqueue itself and retry the counter before the new space is added. The provider then adds the space and returns without waking the now-eligible reserver, leaving it asleep until GC or another event provides a wakeup, potentially adding seconds to max write latency. Take the reservation lock before updating the counter and checking the list. Use list_empty() because the list is now inspected under its lock. Taking a per-mount lock when handing back space is far from ideal, but benchmarking with null_blk showed no measurable performance regression. Fixes: 0bb2193056b5 ("xfs: add support for zoned space reservations") Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260609075655.1698743-1-hch@lst.de?part=2 Signed-off-by: Hans Holmberg Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_zone_space_resv.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/xfs/xfs_zone_space_resv.c b/fs/xfs/xfs_zone_space_resv.c index 5c6e6ef627e471..7aa3c74fb2e01a 100644 --- a/fs/xfs/xfs_zone_space_resv.c +++ b/fs/xfs/xfs_zone_space_resv.c @@ -85,13 +85,13 @@ xfs_zoned_add_available( struct xfs_zone_info *zi = mp->m_zone_info; struct xfs_zone_reservation *reservation; - if (list_empty_careful(&zi->zi_reclaim_reservations)) { - xfs_add_freecounter(mp, XC_FREE_RTAVAILABLE, count_fsb); + spin_lock(&zi->zi_reservation_lock); + xfs_add_freecounter(mp, XC_FREE_RTAVAILABLE, count_fsb); + if (list_empty(&zi->zi_reclaim_reservations)) { + spin_unlock(&zi->zi_reservation_lock); return; } - spin_lock(&zi->zi_reservation_lock); - xfs_add_freecounter(mp, XC_FREE_RTAVAILABLE, count_fsb); count_fsb = xfs_sum_freecounter(mp, XC_FREE_RTAVAILABLE); list_for_each_entry(reservation, &zi->zi_reclaim_reservations, entry) { if (reservation->count_fsb > count_fsb) From f3772e189ee8e37860c2bb365f6ac3517db937e9 Mon Sep 17 00:00:00 2001 From: Lin Jiapeng Date: Tue, 28 Jul 2026 15:19:10 +0800 Subject: [PATCH 789/857] xfs: fix exchange-range reflink flag clearing issue with INO1_WRITTEN When exchanging two full-file ranges, xmi_can_exchange_reflink_flags() can move the reflink inode flag from the file that currently has it to the other file, as long as exactly one side is marked. This assumes that the file contents, and therefore all shared extents, are exchanged. That assumption is not true when XFS_EXCHMAPS_INO1_WRITTEN is set. xfs_exchmaps_can_skip_mapping() can skip hole and unwritten mappings from file1, so an exchange can complete without moving every mapping that the earlier flag-swap decision accounted for. In that case the post-operation cleanup can clear the reflink flag from an inode that still owns shared written extents. Later writes then take the non-reflink write path and may update blocks that should still have been protected by CoW, which shows up as data corruption between reflink-related files. Fix this by disabling the reflink flag exchange whenever XFS_EXCHMAPS_INO1_WRITTEN is requested. The contents exchange can still proceed; the conservative outcome is that both inodes keep the reflink flag. The regular reflink flag cleanup path can drop the extra flag later once the inode no longer has shared extents. Reported-by: Lin Jiapeng (TencentOS Red Team) Fixes: 966ceafc7a43 ("xfs: create deferred log items for file mapping exchanges") Cc: stable@vger.kernel.org # v6.10 Reviewed-by: Darrick J. Wong Reviewed-by: Christoph Hellwig Signed-off-by: Lin Jiapeng Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_exchmaps.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/fs/xfs/libxfs/xfs_exchmaps.c b/fs/xfs/libxfs/xfs_exchmaps.c index 3efed37cb98a8f..49eda8d0994dee 100644 --- a/fs/xfs/libxfs/xfs_exchmaps.c +++ b/fs/xfs/libxfs/xfs_exchmaps.c @@ -959,6 +959,16 @@ xmi_can_exchange_reflink_flags( { struct xfs_mount *mp = req->ip1->i_mount; + /* + * The INO1_WRITTEN optimization can skip exchanging hole and + * unwritten mappings, which means we cannot guarantee that all + * shared extents actually moved to the other file. Clearing the + * reflink flag of an inode that still holds shared extents breaks + * the CoW write path, so refuse to exchange the flags in that case. + */ + if (req->flags & XFS_EXCHMAPS_INO1_WRITTEN) + return false; + /* * The INO1_WRITTEN optimization can skip exchanging hole and * unwritten mappings, which means we cannot guarantee that all From 144d8bc332fa12ce725a0a858f6148cf863150f5 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Mon, 20 Jul 2026 11:45:38 +0200 Subject: [PATCH 790/857] xfs: fix the lock annotation on xfs_iget_cache_hit The newer clang context analysis requires __releases_shared for the RCU pseudo-lock. Signed-off-by: Christoph Hellwig Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_icache.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/xfs/xfs_icache.c b/fs/xfs/xfs_icache.c index a857b8aa255cf7..82dac88e3c4c71 100644 --- a/fs/xfs/xfs_icache.c +++ b/fs/xfs/xfs_icache.c @@ -497,7 +497,8 @@ xfs_iget_cache_hit( struct xfs_inode *ip, xfs_ino_t ino, int flags, - int lock_flags) __releases(RCU) + int lock_flags) + __releases_shared(RCU) { struct inode *inode = VFS_I(ip); struct xfs_mount *mp = ip->i_mount; From 928a0ff252b19aaed4c36466fed72cd1dff3e025 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Mon, 20 Jul 2026 11:45:39 +0200 Subject: [PATCH 791/857] xfs: fix the lock annotation in xfs_extent_busy_update_extent Name the correct lock. Signed-off-by: Christoph Hellwig Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_extent_busy.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/xfs/xfs_extent_busy.c b/fs/xfs/xfs_extent_busy.c index 41cf0605ec22e5..6da8c1f938aa33 100644 --- a/fs/xfs/xfs_extent_busy.c +++ b/fs/xfs/xfs_extent_busy.c @@ -161,8 +161,8 @@ xfs_extent_busy_update_extent( xfs_agblock_t fbno, xfs_extlen_t flen, bool userdata) - __releases(&eb->eb_lock) - __acquires(&eb->eb_lock) + __releases(&xg->xg_busy_extents->eb_lock) + __acquires(&xg->xg_busy_extents->eb_lock) { struct xfs_extent_busy_tree *eb = xg->xg_busy_extents; xfs_agblock_t fend = fbno + flen; From e3bae00ffa91377bb358f81cdb12818b4ecc0e5a Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Mon, 20 Jul 2026 11:45:40 +0200 Subject: [PATCH 792/857] xfs: fix the lock annotation in xfs_mru_cache_lookup Name the actual lock. Unlike sparse, clang wants the annotation to be correct. Signed-off-by: Christoph Hellwig Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_mru_cache.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/xfs/xfs_mru_cache.c b/fs/xfs/xfs_mru_cache.c index d61ec8cb126d67..3f3af2e2e31c15 100644 --- a/fs/xfs/xfs_mru_cache.c +++ b/fs/xfs/xfs_mru_cache.c @@ -520,7 +520,7 @@ xfs_mru_cache_lookup( if (elem) { list_del(&elem->list_node); _xfs_mru_cache_list_insert(mru, elem); - __release(mru_lock); /* help sparse not be stupid */ + __release(&mru->lock); } else spin_unlock(&mru->lock); From 167da957c7e39fe92a880caeed600f1538ef912d Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Mon, 20 Jul 2026 11:45:41 +0200 Subject: [PATCH 793/857] xfs: improve lock annotations in the log code Improve the __acquires and __releases annotations so that the new clang code that is a bit more picky than sparse is happy. This involves passing an explicit struct xlog argument in a few places because alias analysis can't figure out it is the same lock when dereferencing changing iclogs. Signed-off-by: Christoph Hellwig Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_log.c | 31 +++++++++++++++++++------------ fs/xfs/xfs_log_cil.c | 2 +- fs/xfs/xfs_log_priv.h | 4 ++-- 3 files changed, 22 insertions(+), 15 deletions(-) diff --git a/fs/xfs/xfs_log.c b/fs/xfs/xfs_log.c index f807f8f4f70584..0294ac277f357c 100644 --- a/fs/xfs/xfs_log.c +++ b/fs/xfs/xfs_log.c @@ -470,6 +470,8 @@ xlog_state_release_iclog( struct xlog *log, struct xlog_in_core *iclog, struct xlog_ticket *ticket) + __releases(&log->l_icloglock) + __acquires(&log->l_icloglock) { bool last_ref; @@ -744,13 +746,16 @@ xfs_log_mount_cancel( */ static inline int xlog_force_iclog( + struct xlog *log, struct xlog_in_core *iclog) + __releases(&log->l_icloglock) + __acquires(&log->l_icloglock) { atomic_inc(&iclog->ic_refcnt); iclog->ic_flags |= XLOG_ICL_NEED_FLUSH | XLOG_ICL_NEED_FUA; if (iclog->ic_state == XLOG_STATE_ACTIVE) - xlog_state_switch_iclogs(iclog->ic_log, iclog, 0); - return xlog_state_release_iclog(iclog->ic_log, iclog, NULL); + xlog_state_switch_iclogs(log, iclog, 0); + return xlog_state_release_iclog(log, iclog, NULL); } /* @@ -778,11 +783,10 @@ xlog_wait_iclog_completion(struct xlog *log) */ int xlog_wait_on_iclog( + struct xlog *log, struct xlog_in_core *iclog) - __releases(iclog->ic_log->l_icloglock) + __releases(log->l_icloglock) { - struct xlog *log = iclog->ic_log; - trace_xlog_iclog_wait_on(iclog, _RET_IP_); if (!xlog_is_shutdown(log) && iclog->ic_state != XLOG_STATE_ACTIVE && @@ -879,8 +883,8 @@ xlog_unmount_write( spin_lock(&log->l_icloglock); iclog = log->l_iclog; - error = xlog_force_iclog(iclog); - xlog_wait_on_iclog(iclog); + error = xlog_force_iclog(log, iclog); + xlog_wait_on_iclog(log, iclog); if (tic) { trace_xfs_log_umount_write(log, tic); @@ -2741,14 +2745,17 @@ xlog_state_switch_iclogs( */ static int xlog_force_and_check_iclog( + struct xlog *log, struct xlog_in_core *iclog, bool *completed) + __releases(&log->l_icloglock) + __acquires(&log->l_icloglock) { xfs_lsn_t lsn = be64_to_cpu(iclog->ic_header->h_lsn); int error; *completed = false; - error = xlog_force_iclog(iclog); + error = xlog_force_iclog(log, iclog); if (error) return error; @@ -2825,7 +2832,7 @@ xfs_log_force( /* We have exclusive access to this iclog. */ bool completed; - if (xlog_force_and_check_iclog(iclog, &completed)) + if (xlog_force_and_check_iclog(log, iclog, &completed)) goto out_error; if (completed) @@ -2850,7 +2857,7 @@ xfs_log_force( iclog->ic_flags |= XLOG_ICL_NEED_FLUSH | XLOG_ICL_NEED_FUA; if (flags & XFS_LOG_SYNC) - return xlog_wait_on_iclog(iclog); + return xlog_wait_on_iclog(log, iclog); out_unlock: spin_unlock(&log->l_icloglock); return 0; @@ -2920,7 +2927,7 @@ xlog_force_lsn( &log->l_icloglock); return -EAGAIN; } - if (xlog_force_and_check_iclog(iclog, &completed)) + if (xlog_force_and_check_iclog(log, iclog, &completed)) goto out_error; if (log_flushed) *log_flushed = 1; @@ -2948,7 +2955,7 @@ xlog_force_lsn( } if (flags & XFS_LOG_SYNC) - return xlog_wait_on_iclog(iclog); + return xlog_wait_on_iclog(log, iclog); out_unlock: spin_unlock(&log->l_icloglock); return 0; diff --git a/fs/xfs/xfs_log_cil.c b/fs/xfs/xfs_log_cil.c index 639f875a8fb25a..ae1ed16aeb2fc7 100644 --- a/fs/xfs/xfs_log_cil.c +++ b/fs/xfs/xfs_log_cil.c @@ -1556,7 +1556,7 @@ xlog_cil_push_work( * iclogs older than ic_prev. Hence we only need to wait * on the most recent older iclog here. */ - xlog_wait_on_iclog(ctx->commit_iclog->ic_prev); + xlog_wait_on_iclog(log, ctx->commit_iclog->ic_prev); spin_lock(&log->l_icloglock); } diff --git a/fs/xfs/xfs_log_priv.h b/fs/xfs/xfs_log_priv.h index cf1e4ce61a8c2b..6d9673c41cdf31 100644 --- a/fs/xfs/xfs_log_priv.h +++ b/fs/xfs/xfs_log_priv.h @@ -605,8 +605,8 @@ xlog_wait( remove_wait_queue(wq, &wait); } -int xlog_wait_on_iclog(struct xlog_in_core *iclog) - __releases(iclog->ic_log->l_icloglock); +int xlog_wait_on_iclog(struct xlog *log, struct xlog_in_core *iclog) + __releases(log->l_icloglock); /* Calculate the distance between two LSNs in bytes */ static inline uint64_t From f3cc0483ffe30c6a82a34a5575aceb9d40651b33 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Mon, 20 Jul 2026 11:45:42 +0200 Subject: [PATCH 794/857] xfs: add lock annotations to xfs_try_open_zone Improve the __acquires and __releases annotations so that the new clang code that is a bit more picky than sparse is happy. Signed-off-by: Christoph Hellwig Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_zone_alloc.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/xfs/xfs_zone_alloc.c b/fs/xfs/xfs_zone_alloc.c index bdbb60cc5d5b26..28c1e48909fa8d 100644 --- a/fs/xfs/xfs_zone_alloc.c +++ b/fs/xfs/xfs_zone_alloc.c @@ -475,6 +475,8 @@ static struct xfs_open_zone * xfs_try_open_zone( struct xfs_mount *mp, enum rw_hint write_hint) + __releases(&mp->m_zone_info->zi_open_zones_lock) + __acquires(&mp->m_zone_info->zi_open_zones_lock) { struct xfs_zone_info *zi = mp->m_zone_info; struct xfs_open_zone *oz; From 4bb0fb3366fa1638387c0ffbb85b1d2d077bc4bd Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Mon, 20 Jul 2026 11:45:43 +0200 Subject: [PATCH 795/857] xfs: add lock annotations to xlog_state_shutdown_callbacks Sparse used to get away without these despite dropping and reacquiring l_icloglock Signed-off-by: Christoph Hellwig Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_log.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/xfs/xfs_log.c b/fs/xfs/xfs_log.c index 0294ac277f357c..2a34611d81f6d0 100644 --- a/fs/xfs/xfs_log.c +++ b/fs/xfs/xfs_log.c @@ -422,6 +422,8 @@ xfs_log_reserve( static void xlog_state_shutdown_callbacks( struct xlog *log) + __releases(&log->l_icloglock) + __acquires(&log->l_icloglock) { struct xlog_in_core *iclog; LIST_HEAD(cb_list); From 0a9c62fb4ef554e7e6ece56ffbec51d7c2424202 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Mon, 20 Jul 2026 11:45:44 +0200 Subject: [PATCH 796/857] xfs: add a lock annotation to xlog_cil_push_background This is required to make the clang context analysis happy, which is more strict than the old sparse lock context tracking. Signed-off-by: Christoph Hellwig Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_log_cil.c | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/xfs/xfs_log_cil.c b/fs/xfs/xfs_log_cil.c index ae1ed16aeb2fc7..166531018ce444 100644 --- a/fs/xfs/xfs_log_cil.c +++ b/fs/xfs/xfs_log_cil.c @@ -1627,6 +1627,7 @@ xlog_cil_push_work( static void xlog_cil_push_background( struct xlog *log) + __releases_shared(&log->l_cilp->xc_ctx_lock) { struct xfs_cil *cil = log->l_cilp; int space_used = atomic_read(&cil->xc_ctx->space_used); From ca753f8184dbe8ff87eb6fe168544015fb6c1cda Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Mon, 20 Jul 2026 11:45:45 +0200 Subject: [PATCH 797/857] xfs: add lock annotations to xfs_ail_delete* Pass up the __must_hold as clang requires it, and also fix the formatting of the __must_hold on xfs_ail_check to match how we do it elsewhere. Signed-off-by: Christoph Hellwig Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_trans_ail.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/fs/xfs/xfs_trans_ail.c b/fs/xfs/xfs_trans_ail.c index 99a9bf3762b7e1..f955479a08fd54 100644 --- a/fs/xfs/xfs_trans_ail.c +++ b/fs/xfs/xfs_trans_ail.c @@ -33,7 +33,7 @@ STATIC void xfs_ail_check( struct xfs_ail *ailp, struct xfs_log_item *lip) - __must_hold(&ailp->ail_lock) + __must_hold(&ailp->ail_lock) { struct xfs_log_item *prev_lip; struct xfs_log_item *next_lip; @@ -321,6 +321,7 @@ static void xfs_ail_delete( struct xfs_ail *ailp, struct xfs_log_item *lip) + __must_hold(&ailp->ail_lock) { xfs_ail_check(ailp, lip); list_del(&lip->li_ail); @@ -899,6 +900,7 @@ xfs_lsn_t xfs_ail_delete_one( struct xfs_ail *ailp, struct xfs_log_item *lip) + __must_hold(&ailp->ail_lock) { struct xfs_log_item *mlip = xfs_ail_min(ailp); xfs_lsn_t lsn = lip->li_lsn; From 4f434e6c6b3c4999f58e56ac611e1d4d9022a185 Mon Sep 17 00:00:00 2001 From: Javier Tia Date: Mon, 10 Aug 2026 17:06:13 -0600 Subject: [PATCH 798/857] xfs: initialise error in xfs_defer_finish_one() xfs_defer_finish_one() declares error without an initialiser and only assigns it inside the loop over dfp->dfp_work. When that list is empty the loop body never runs, control falls through to the "Done with the dfp, free it" path, and the function returns an indeterminate value. An item-less pending item reaches this through xfs_defer_add_barrier(), which xfs_reap_ag_blocks() uses on any CONFIG_XFS_ONLINE_REPAIR kernel. xfs_defer_finish_noroll() treats any non-EAGAIN return as fatal, so a non-zero stack value turns a successful barrier into a SHUTDOWN_CORRUPT_INCORE in the middle of a repair. Zero is the correct result: reaching the free path means the item loop drained without a non-zero error. Fixes: 3f3cec031099 ("xfs: force small EFIs for reaping btree extents") Cc: stable@vger.kernel.org Signed-off-by: Javier Tia Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_defer.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/xfs/libxfs/xfs_defer.c b/fs/xfs/libxfs/xfs_defer.c index 89501e8bd2f845..843c3330444150 100644 --- a/fs/xfs/libxfs/xfs_defer.c +++ b/fs/xfs/libxfs/xfs_defer.c @@ -583,7 +583,7 @@ xfs_defer_finish_one( const struct xfs_defer_op_type *ops = dfp->dfp_ops; struct xfs_btree_cur *state = NULL; struct list_head *li, *n; - int error; + int error = 0; trace_xfs_defer_pending_finish(tp->t_mountp, dfp); From 69d300eba1ef042d947a15906f577df1a2b8ef33 Mon Sep 17 00:00:00 2001 From: Javier Tia Date: Mon, 10 Aug 2026 17:06:14 -0600 Subject: [PATCH 799/857] xfs: give the deferred barrier op type a name xfs_barrier_defer_type is the only xfs_defer_op_type with no .name. Every other one carries a short string used for tracing and reporting: attr, bmap, extent_free, agfl_free, rtextent_free, refcount, rtrefcount, rmap, rtrmap and exchmaps. That has been harmless because nothing dereferences the field, but it leaves a NULL in a table where every other entry is populated, so the first caller to print it gets "(null)" in the kernel and undefined behaviour in the userspace libxfs build of this file, where xfs_alert lands in fprintf. xfs_defer_add() already treats a missing member of this table as worth shutting the filesystem down for, so an unpopulated one is out of step with how the file handles its own ops tables. Signed-off-by: Javier Tia Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_defer.c | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/xfs/libxfs/xfs_defer.c b/fs/xfs/libxfs/xfs_defer.c index 843c3330444150..75f0d37914d504 100644 --- a/fs/xfs/libxfs/xfs_defer.c +++ b/fs/xfs/libxfs/xfs_defer.c @@ -229,6 +229,7 @@ xfs_defer_barrier_cancel_item( } static const struct xfs_defer_op_type xfs_barrier_defer_type = { + .name = "barrier", .max_items = 1, .create_intent = xfs_defer_barrier_create_intent, .abort_intent = xfs_defer_barrier_abort_intent, From 5d46cb54112a18b777671d4487344d235bb1357e Mon Sep 17 00:00:00 2001 From: Javier Tia Date: Mon, 10 Aug 2026 17:06:15 -0600 Subject: [PATCH 800/857] xfs: report the error that made deferred work shut down the fs When a deferred operation fails and shuts the filesystem down, xfs_defer_finish_noroll() reports neither the errno nor which operation originated it, so the log cannot tell a transient -ENOSPC from real corruption. Report the operation type, errno and remaining reservation. trace_xfs_defer_finish_error() runs after xfs_force_shutdown(), which BUGs under fs.xfs.panic_mask and so never fires for the first failure; move it ahead of the shutdown and mirror it to xfs_alert() for systems without tracing armed. Capture the op name while the item is live (dfp is freed once its work list drains) and suppress the alert once the fs is already down. Signed-off-by: Javier Tia Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_defer.c | 15 ++++++++++++++- 1 file changed, 14 insertions(+), 1 deletion(-) diff --git a/fs/xfs/libxfs/xfs_defer.c b/fs/xfs/libxfs/xfs_defer.c index 75f0d37914d504..3152acdc335d2f 100644 --- a/fs/xfs/libxfs/xfs_defer.c +++ b/fs/xfs/libxfs/xfs_defer.c @@ -656,6 +656,7 @@ xfs_defer_finish_noroll( struct xfs_trans **tp) { struct xfs_defer_pending *dfp = NULL; + const char *what = "chain"; int error = 0; LIST_HEAD(dop_pending); LIST_HEAD(dop_paused); @@ -705,9 +706,17 @@ xfs_defer_finish_noroll( struct xfs_defer_pending, dfp_list); if (!dfp) break; + what = dfp->dfp_ops->name; error = xfs_defer_finish_one(*tp, dfp); if (error && error != -EAGAIN) goto out_shutdown; + /* + * A finished item is no longer a candidate for a later + * failure. An -EAGAIN one is not finished, so it keeps the + * attribution across the roll that completes it. + */ + if (!error) + what = "chain"; } /* Requeue the paused items in the outgoing transaction. */ @@ -719,8 +728,12 @@ xfs_defer_finish_noroll( out_shutdown: list_splice_tail_init(&dop_paused, &dop_pending); xfs_defer_trans_abort(*tp, &dop_pending); - xfs_force_shutdown((*tp)->t_mountp, SHUTDOWN_CORRUPT_INCORE); trace_xfs_defer_finish_error(*tp, error); + if (!xfs_is_shutdown((*tp)->t_mountp)) + xfs_alert((*tp)->t_mountp, + "deferred %s work failed, error %d, %u blocks reserved", + what, error, (*tp)->t_blk_res); + xfs_force_shutdown((*tp)->t_mountp, SHUTDOWN_CORRUPT_INCORE); xfs_defer_cancel_list((*tp)->t_mountp, &dop_pending); xfs_defer_cancel(*tp); return error; From ae66b1508239af55ea5adbf40e1c68216a824b08 Mon Sep 17 00:00:00 2001 From: Javier Tia Date: Mon, 10 Aug 2026 17:06:16 -0600 Subject: [PATCH 801/857] xfs: correct the parent pointer space reservation comment The comment on xfs_parent_calc_space_res() claims parent pointers are "always the first attr in an attr tree". They are not: a parent pointer is recorded per dirent, so by the Nth hardlink the attr fork is already in leaf or node format. The reservation is still correct, because XFS_DAENTER_SPACE_RES() covers a split at every level of a maximum-depth attr dabtree whatever format the fork is in, but anyone auditing a shortfall here is led by the comment to look for a bug that is not there. Rewrite the comment to state what actually bounds the result, and record why the double split allowance and the extent-add term differ from xfs_attr_calc_size(). Signed-off-by: Javier Tia Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_trans_space.c | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/fs/xfs/libxfs/xfs_trans_space.c b/fs/xfs/libxfs/xfs_trans_space.c index 9b8f495c9049cc..c4cd547033e584 100644 --- a/fs/xfs/libxfs/xfs_trans_space.c +++ b/fs/xfs/libxfs/xfs_trans_space.c @@ -22,8 +22,23 @@ xfs_parent_calc_space_res( unsigned int namelen) { /* - * Parent pointers are always the first attr in an attr tree, and never - * larger than a block + * A parent pointer is recorded per dirent, so an inode with N links + * carries N of them and the attr fork can already be in leaf or node + * format when one is added. That does not affect the reservation: + * XFS_DAENTER_SPACE_RES covers a split at every level of a + * maximum-depth attr dabtree, whatever format the fork is in now. + * + * The name is a dirent name and the value is a struct xfs_parent_rec, + * so the leaf entry is always local and never exceeds 272 bytes. + * Parent pointers require V5, hence a 1k minimum block size, so the + * entry always stays under half a block and this needs none of the + * double split allowance that xfs_attr_calc_size() makes. + * + * The second term hands a byte count to a macro whose parameter counts + * mappings, so it asks for more extent-add allowance than the single + * mapping a parent pointer adds - how much more depends on the block + * size. It over-reserves either way, which is why it is left alone: + * correcting the unit would shrink a reservation that is only generous. */ return XFS_DAENTER_SPACE_RES(mp, XFS_ATTR_FORK) + XFS_NEXTENTADD_SPACE_RES(mp, namelen, XFS_ATTR_FORK); From 05276cd720f903f4e31d770b01249cab1087da65 Mon Sep 17 00:00:00 2001 From: Javier Tia Date: Mon, 10 Aug 2026 17:06:17 -0600 Subject: [PATCH 802/857] xfs: initialise args->total for parent pointer updates xfs_parent_da_args_init() builds an xfs_da_args from a zeroed xfs_parent_args (kmem_cache_zalloc), leaving args->total == 0. xfs_da_grow_inode_int() treats that field as a running block reservation and subtracts from it; because it is an xfs_extlen_t (uint32_t), the first attr-fork growth wraps it to ~0U. That defeats the free-space check in xfs_alloc_space_available(), and when it coincides with an AG that has exactly zero available blocks the allocation is clamped to maxlen 0 and returns -ENOSPC, which xfs_defer_finish_noroll() escalates to a filesystem shutdown. Set args->total the way the log recovery path does (xfs_attri_recover_work(), xfs_attr_item.c:706), in the add and replace paths that can grow the fork. Removals and lookups never grow it, so they leave the field alone, matching that switch. Fixes: b7c62d90c12c ("xfs: parent pointer attribute creation") Cc: stable@vger.kernel.org # v6.10 Signed-off-by: Javier Tia Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_parent.c | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/fs/xfs/libxfs/xfs_parent.c b/fs/xfs/libxfs/xfs_parent.c index 8d111c9b6527ee..a2f2f5fa640e41 100644 --- a/fs/xfs/libxfs/xfs_parent.c +++ b/fs/xfs/libxfs/xfs_parent.c @@ -193,7 +193,7 @@ xfs_parent_addname( const struct xfs_name *parent_name, struct xfs_inode *child) { - int error; + int error, local; error = xfs_parent_iread_extents(tp, child); if (error) @@ -203,6 +203,10 @@ xfs_parent_addname( xfs_parent_da_args_init(&ppargs->args, tp, &ppargs->rec, child, I_INO(child), parent_name); + /* Growing the attr fork needs a real reservation in args->total. */ + ppargs->args.total = xfs_attr_calc_size(&ppargs->args, &local); + ASSERT(local); + return xfs_attr_setname(&ppargs->args, 0); } @@ -239,7 +243,7 @@ xfs_parent_replacename( const struct xfs_name *new_name, struct xfs_inode *child) { - int error; + int error, local; error = xfs_parent_iread_extents(tp, child); if (error) @@ -249,6 +253,10 @@ xfs_parent_replacename( xfs_parent_da_args_init(&ppargs->args, tp, &ppargs->rec, child, I_INO(child), old_name); + /* Growing the attr fork needs a real reservation in args->total. */ + ppargs->args.total = xfs_attr_calc_size(&ppargs->args, &local); + ASSERT(local); + xfs_inode_to_parent_rec(&ppargs->new_rec, new_dp); ppargs->args.new_name = new_name->name; From 4476c24bd7796a9e373b0f1906df9c651df053b5 Mon Sep 17 00:00:00 2001 From: Javier Tia Date: Mon, 10 Aug 2026 17:06:18 -0600 Subject: [PATCH 803/857] xfs: assert the reservation covers each da fork growth xfs_da_grow_inode_int() subtracts the blocks it just allocated from args->total, the caller's remaining block reservation. The subtraction is unsigned, so a caller that reaches it with too small a total wraps the field instead of failing, and every allocation afterwards runs with a bogus reservation. Assert the remaining reservation still covers the step, so an under-reserved or uninitialised total trips in debug builds instead of silently wrapping. Suggested-by: Darrick J. Wong Signed-off-by: Javier Tia Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_da_btree.c | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/xfs/libxfs/xfs_da_btree.c b/fs/xfs/libxfs/xfs_da_btree.c index f190c088591bce..a10d20eb9b1bcc 100644 --- a/fs/xfs/libxfs/xfs_da_btree.c +++ b/fs/xfs/libxfs/xfs_da_btree.c @@ -2384,6 +2384,7 @@ xfs_da_grow_inode_int( } /* account for newly allocated blocks in reserved blocks total */ + ASSERT(args->total >= dp->i_nblocks - nblks); args->total -= dp->i_nblocks - nblks; out_free_map: From 556928b5018720a719e71ec6e595e1d07d996910 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Wed, 26 Aug 2026 22:31:10 -0700 Subject: [PATCH 804/857] xfs: don't spin forever on zero-length dirents when salvaging them LOLLM noticed that xrep_dir_recover_data can spin forever if it encounters an unused dirent that claims to have length zero. Fix that, and prevent the same thing from happening with a zero-length entry. Cc: stable@vger.kernel.org # v6.10 Fixes: b1991ee3e7cf85 ("xfs: online repair of directories") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/dir_repair.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/fs/xfs/scrub/dir_repair.c b/fs/xfs/scrub/dir_repair.c index 1c088cfba10ea9..0c1224d05d579a 100644 --- a/fs/xfs/scrub/dir_repair.c +++ b/fs/xfs/scrub/dir_repair.c @@ -484,18 +484,24 @@ xrep_dir_recover_data( while (offset < end) { struct xfs_dir2_data_unused *dup = bp->b_addr + offset; struct xfs_dir2_data_entry *dep = bp->b_addr + offset; + unsigned int advance; if (xchk_should_terminate(rd->sc, &error)) return error; /* Skip unused entries. */ if (be16_to_cpu(dup->freetag) == XFS_DIR2_DATA_FREE_TAG) { + if (!dup->length) + break; offset += be16_to_cpu(dup->length); continue; } /* Don't walk off the end of the block. */ - offset += xfs_dir2_data_entsize(rd->sc->mp, dep->namelen); + advance = xfs_dir2_data_entsize(rd->sc->mp, dep->namelen); + if (!advance) + break; + offset += advance; if (offset > end) break; From 3db0206c6423ed09cdcb94101888ca3cb351655c Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Wed, 26 Aug 2026 22:31:25 -0700 Subject: [PATCH 805/857] xfs: don't stash removename operations with unknown ftype LOLLM notices that the behavior of xrep_dir_replay_update changes based on the ftype recorded in the stashed removename information. It also notices that the unlink iops sometimes set that ftype to FT_UNKNOWN because the regular directory tree update code paths don't need to know the ftype of the child. Unfortunately, this results in incorrect link counts, which eventually trips link count errors in later phases of xfs_scrub, or in xfs_repair. Fix this by creating a second xfs_name with the type set correctly. Cc: stable@vger.kernel.org # v6.10 Fixes: 8559b21a64d983 ("xfs: implement live updates for directory repairs") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/dir_repair.c | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/fs/xfs/scrub/dir_repair.c b/fs/xfs/scrub/dir_repair.c index 0c1224d05d579a..31a23c5f386ae6 100644 --- a/fs/xfs/scrub/dir_repair.c +++ b/fs/xfs/scrub/dir_repair.c @@ -1381,9 +1381,24 @@ xrep_dir_live_update( if (p->delta > 0) error = xrep_dir_stash_createname(rd, p->name, I_INO(p->ip)); - else - error = xrep_dir_stash_removename(rd, p->name, + else { + /* + * xfs_dentry_to_name in unlink or rename-exchange can + * pass us names with ftype FT_UNKNOWN, but we really + * must know the ftype of the child that is being + * removed so that we can do nlink updates correctly + * without holding inode references. + */ + struct xfs_name name = { + .name = p->name->name, + .len = p->name->len, + .type = xfs_mode_to_ftype( + VFS_IC(p->ip)->i_mode), + }; + + error = xrep_dir_stash_removename(rd, &name, I_INO(p->ip)); + } mutex_unlock(&rd->pscan.lock); if (error) goto out_abort; From 6bd830b68329c6af1419f6dd9f8ad0c40c5351d7 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Wed, 26 Aug 2026 22:31:41 -0700 Subject: [PATCH 806/857] xfs: log the tempip after we convert it to extents format LOLLM points out that xrep_symlink_swap_prep converts sc->tempip to an extents format file prior to the atomic swap, but incorrectly logs sc->ip immediately afterwards. Fix that, and the other problem that we're supposed to tell xfs_trans_log_inode what to log and don't. Cc: stable@vger.kernel.org # v6.10 Fixes: 2651923d8d8db0 ("xfs: online repair of symbolic links") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/symlink_repair.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/xfs/scrub/symlink_repair.c b/fs/xfs/scrub/symlink_repair.c index 91c86ea0e0f155..18196136423336 100644 --- a/fs/xfs/scrub/symlink_repair.c +++ b/fs/xfs/scrub/symlink_repair.c @@ -291,7 +291,7 @@ xrep_symlink_swap_prep( if (error) return error; - xfs_trans_log_inode(sc->tp, sc->ip, 0); + xfs_trans_log_inode(sc->tp, sc->tempip, logflags); error = xfs_defer_finish(&sc->tp); if (error) From 5b5ea66f124c5e57a5d7236628cb7e34a6d7aa5c Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Wed, 26 Aug 2026 22:31:56 -0700 Subject: [PATCH 807/857] xfs: fix parent rec lookup initialization in xrep_metapath_unlink LOLLM notices that xrep_metapath_unlink looks for a parent pointer in the child metafile that it's removing, but initializes the parent handle using the child. This is obviously incorrect, so fix that. Cc: stable@vger.kernel.org # v6.13 Fixes: 0d2c636e489c11 ("xfs: repair metadata directory file path connectivity") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/metapath.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/xfs/scrub/metapath.c b/fs/xfs/scrub/metapath.c index ff1ff762b3003c..9f44c82910eaef 100644 --- a/fs/xfs/scrub/metapath.c +++ b/fs/xfs/scrub/metapath.c @@ -397,7 +397,7 @@ xrep_metapath_unlink( /* Figure out if we're removing a parent pointer too. */ if (xfs_has_parent(mp)) { - xfs_inode_to_parent_rec(&rec, ip); + xfs_inode_to_parent_rec(&rec, mpath->dp); error = xfs_parent_lookup(sc->tp, ip, &mpath->xname, &rec, &mpath->pptr_args); switch (error) { From cd2d27b0c32a0203dc0c9f89f4415949e30703cd Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Wed, 26 Aug 2026 22:32:12 -0700 Subject: [PATCH 808/857] xfs: handle reconnecting metadir subdirectories A longstanding weakness of the metapath repair code is that it can only reattach non-directories to the metadata directory tree. Let's fix that by allowing reconnection of subdirectories. Note that with the initial users of metadir (rtgroups and quota), there's no way to mount a filesystem with broken /rtgroups or /quota subdirectories, so this code won't be all that useful until something adds deeper directory trees. But we shouldn't leave a logic bomb for those futures users wherein we get the link count wrong for a subdir. Signed-off-by: Darrick J. Wong Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/metapath.c | 83 ++++++++++++++++++++++++++++++++++++++++- 1 file changed, 82 insertions(+), 1 deletion(-) diff --git a/fs/xfs/scrub/metapath.c b/fs/xfs/scrub/metapath.c index 9f44c82910eaef..a760523f40e6ff 100644 --- a/fs/xfs/scrub/metapath.c +++ b/fs/xfs/scrub/metapath.c @@ -23,6 +23,7 @@ #include "xfs_rtgroup.h" #include "xfs_rtrmap_btree.h" #include "xfs_rtrefcount_btree.h" +#include "xfs_ag.h" #include "scrub/scrub.h" #include "scrub/common.h" #include "scrub/trace.h" @@ -348,12 +349,78 @@ xchk_metapath( } #ifdef CONFIG_XFS_ONLINE_REPAIR +/* + * Given a directory @dp, an existing inode @ip, and a @name, link @ip into @dp + * under the given @name. + */ +static int +xrep_metadir_add_child( + struct xchk_metapath *mpath, + xfs_ino_t old_dotdot) +{ + struct xfs_trans *tp = mpath->sc->tp; + struct xfs_dir_update *du = &mpath->du; + struct xfs_inode *dp = du->dp; + const struct xfs_name *name = du->name; + struct xfs_inode *ip = du->ip; + struct xfs_mount *mp = tp->t_mountp; + const unsigned int resblks = mpath->link_resblks; + int error; + + /* + * The metadata file shouldn't be on the unlinked list, but we'll fix + * it if that is the case. + */ + if (VFS_I(ip)->i_nlink == 0) { + struct xfs_perag *pag; + + pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, I_INO(ip))); + error = xfs_iunlink_remove(tp, pag, ip); + xfs_perag_put(pag); + if (error) + return error; + } + + error = xfs_dir_createname(tp, dp, name, I_INO(ip), resblks); + if (error) + return error; + + xfs_trans_log_inode(tp, dp, XFS_ILOG_CORE); + + xfs_bumplink(tp, ip); + + /* update dotdot entry in child */ + if (S_ISDIR(VFS_I(ip)->i_mode)) { + xfs_bumplink(tp, dp); + + /* Replace the dotdot entry in the child */ + if (old_dotdot != I_INO(dp)) { + error = xfs_dir_replace(tp, ip, &xfs_name_dotdot, + I_INO(dp), resblks); + if (error) + return error; + } + } + + /* Update the child's parent pointer */ + if (du->ppargs) { + error = xfs_parent_addname(tp, du->ppargs, dp, name, ip); + if (error) + return error; + } + + xfs_dir_update_hook(dp, ip, 1, name); + return 0; +} + /* Create the dirent represented by the final component of the path. */ STATIC int xrep_metapath_link( struct xchk_metapath *mpath) { struct xfs_scrub *sc = mpath->sc; + xfs_ino_t old_dotdot = NULLFSINO; + int error; mpath->du.dp = mpath->dp; mpath->du.name = &mpath->xname; @@ -366,7 +433,21 @@ xrep_metapath_link( trace_xrep_metapath_link(sc, mpath->path, mpath->dp, I_INO(sc->ip)); - return xfs_dir_add_child(sc->tp, mpath->link_resblks, &mpath->du); + if (S_ISDIR(VFS_I(sc->ip)->i_mode)) { + error = xchk_dir_lookup(sc, sc->ip, &xfs_name_dotdot, + &old_dotdot); + if (error && error != -ENOENT) + return error; + + /* + * subdir didn't give us a dotdot entry, so we just give up + * and let the repair get marked as failed. + */ + if (old_dotdot == NULLFSINO) + return 0; + } + + return xrep_metadir_add_child(mpath, old_dotdot); } /* Remove the dirent at the final component of the path. */ From fbb8c1b636be481b6fe49bd678afc7e866d33e35 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Wed, 26 Aug 2026 22:32:28 -0700 Subject: [PATCH 809/857] xfs: lock the healthmon when inserting unmount event LOLLM complains that xfs_healthmon_unmount does an unlocked insert of the unmount event into the health monitor's event list. Fix that. Cc: stable@vger.kernel.org # v7.0 Fixes: 25ca57fa3624ca ("xfs: convey filesystem unmount events to the health monitor") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_healthmon.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/fs/xfs/xfs_healthmon.c b/fs/xfs/xfs_healthmon.c index 4521ffdab9f1ae..3ae5f4496ad1aa 100644 --- a/fs/xfs/xfs_healthmon.c +++ b/fs/xfs/xfs_healthmon.c @@ -272,6 +272,8 @@ __xfs_healthmon_insert( { struct timespec64 now; + lockdep_assert_held(&hm->lock); + ktime_get_coarse_real_ts64(&now); event->time_ns = (now.tv_sec * NSEC_PER_SEC) + now.tv_nsec; @@ -294,6 +296,8 @@ __xfs_healthmon_push( { struct timespec64 now; + lockdep_assert_held(&hm->lock); + ktime_get_coarse_real_ts64(&now); event->time_ns = (now.tv_sec * NSEC_PER_SEC) + now.tv_nsec; @@ -415,8 +419,10 @@ xfs_healthmon_unmount( * There's nothing actionable for userspace after an unmount. Once * we've inserted the unmount event, hm no longer owns that event. */ + mutex_lock(&hm->lock); __xfs_healthmon_insert(hm, hm->unmount_event); hm->unmount_event = NULL; + mutex_unlock(&hm->lock); xfs_healthmon_detach(hm); xfs_healthmon_put(hm); From 7f49bf8378c7275bb82e807fabe5f4bc7ccdd063 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Fri, 21 Aug 2026 17:03:37 -0500 Subject: [PATCH 810/857] xfs: fix reclaimed page accounting in xfs_buf_free To obtain nr. of pages in "size" bytes, we need howmany(size, PAGE_SIZE) not howmany(size, PAGE_SHIFT). This over-reports reclaim by orders of magnitude, up to 4096x on a 64k page system. Fixes: e2874632a621 ("xfs: use vmalloc instead of vm_map_area for buffer backing memory") Cc: stable@vger.kernel.org # v6.15+ Signed-off-by: Eric Sandeen Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_buf.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/xfs/xfs_buf.c b/fs/xfs/xfs_buf.c index ee7c2e9c0340c9..836515e8feeef8 100644 --- a/fs/xfs/xfs_buf.c +++ b/fs/xfs/xfs_buf.c @@ -139,7 +139,7 @@ xfs_buf_free( ASSERT(list_empty(&bp->b_lru)); if (!xfs_buftarg_is_mem(bp->b_target) && size >= PAGE_SIZE) - mm_account_reclaimed_pages(howmany(size, PAGE_SHIFT)); + mm_account_reclaimed_pages(howmany(size, PAGE_SIZE)); if (is_vmalloc_addr(bp->b_addr)) vfree(bp->b_addr); From 97a956fb8cc21e0befcfb76746975feb20c2079e Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Fri, 21 Aug 2026 17:26:39 -0500 Subject: [PATCH 811/857] xfs: mark slab-allocated xfs_buf backing memory as __GFP_RECLAIMABLE xfs_bufs have a shrinker and are therefore reclaimable, as is the memory backing them. Mark slab-allocated backing memory as __GFP_RECLAIMABLE in the kmalloc path so that it is accounted properly. Signed-off-by: Eric Sandeen Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_buf.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/xfs/xfs_buf.c b/fs/xfs/xfs_buf.c index 836515e8feeef8..8256c1d13ce20f 100644 --- a/fs/xfs/xfs_buf.c +++ b/fs/xfs/xfs_buf.c @@ -176,7 +176,7 @@ xfs_buf_alloc_kmem( ASSERT(is_power_of_2(size)); ASSERT(size < PAGE_SIZE); - bp->b_addr = kmalloc(size, gfp_mask); + bp->b_addr = kmalloc(size, gfp_mask | __GFP_RECLAIMABLE); if (!bp->b_addr) return -ENOMEM; From 724d0bae9604dc23d0c51da021c2d59f0b124646 Mon Sep 17 00:00:00 2001 From: Anuj Gupta Date: Tue, 1 Sep 2026 11:13:48 +0530 Subject: [PATCH 812/857] xfs: release alleged child inode on metapath unlink error If xchk_metapath_ilock_parent_and_child() fails after xchk_iget() succeeds, release the inode reference before returning. Fixes: 0d2c636e489c ("xfs: repair metadata directory file path connectivity") Cc: stable@vger.kernel.org # v6.13 Signed-off-by: Anuj Gupta Reviewed-by: Darrick J. Wong Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/metapath.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/xfs/scrub/metapath.c b/fs/xfs/scrub/metapath.c index a760523f40e6ff..e0ee7d9b903ff0 100644 --- a/fs/xfs/scrub/metapath.c +++ b/fs/xfs/scrub/metapath.c @@ -637,6 +637,8 @@ xrep_metapath_try_unlink( error = xchk_metapath_ilock_parent_and_child(mpath, ip); if (error) { xchk_trans_cancel(sc); + if (ip) + xchk_irele(sc, ip); return error; } xfs_trans_ijoin(sc->tp, mpath->dp, 0); From cc38e26e3f94c46591560c39857d3072544e9635 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:00 +0200 Subject: [PATCH 813/857] xfs: remove an outdated comment above xfs_file_ioctl The return code sign flipping at the method boundary is long gone in XFS, so remove this comment documenting an exception from it. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 6 ------ 1 file changed, 6 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 96ca3e480cb9fe..4a8921f38a082f 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1209,12 +1209,6 @@ xfs_ioctl_fs_counts( #define XFS_IOC_ALLOCSP64 _IOW ('X', 36, struct xfs_flock64) #define XFS_IOC_FREESP64 _IOW ('X', 37, struct xfs_flock64) -/* - * Note: some of the ioctl's return positive numbers as a - * byte count indicating success, such as readlink_by_handle. - * So we don't "sign flip" like most other routines. This means - * true errors need to be returned as a negative value. - */ long xfs_file_ioctl( struct file *filp, From 4e58106b3cf82b1b700899c5f8983efffd6050fd Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:01 +0200 Subject: [PATCH 814/857] xfs: rename xfs_ioc_swapext to xfs_swapext The usual convention is that the ioc_ prefix is used for direct ioctl handlers that take a user pointer. The current xfs_ioc_swapext does not fit that pattern, so rename it. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 8 ++++---- fs/xfs/xfs_ioctl.h | 4 +--- fs/xfs/xfs_ioctl32.c | 2 +- 3 files changed, 6 insertions(+), 8 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 4a8921f38a082f..314eeec0234171 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -962,10 +962,10 @@ xfs_ioc_getbmap( } int -xfs_ioc_swapext( - xfs_swapext_t *sxp) +xfs_swapext( + struct xfs_swapext *sxp) { - xfs_inode_t *ip, *tip; + struct xfs_inode *ip, *tip; /* Pull information for the target fd */ CLASS(fd, f)((int)sxp->sx_fdtarget); @@ -1341,7 +1341,7 @@ xfs_file_ioctl( error = mnt_want_write_file(filp); if (error) return error; - error = xfs_ioc_swapext(&sxp); + error = xfs_swapext(&sxp); mnt_drop_write_file(filp); return error; } diff --git a/fs/xfs/xfs_ioctl.h b/fs/xfs/xfs_ioctl.h index f5ed5cf9d3df65..e57d8f5148bf7f 100644 --- a/fs/xfs/xfs_ioctl.h +++ b/fs/xfs/xfs_ioctl.h @@ -10,9 +10,7 @@ struct xfs_bstat; struct xfs_ibulk; struct xfs_inogrp; -int -xfs_ioc_swapext( - xfs_swapext_t *sxp); +int xfs_swapext(struct xfs_swapext *sxp); extern int xfs_fileattr_get( diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index c66e192448a89d..250df2bbf21417 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -475,7 +475,7 @@ xfs_file_compat_ioctl( error = mnt_want_write_file(filp); if (error) return error; - error = xfs_ioc_swapext(&sxp); + error = xfs_swapext(&sxp); mnt_drop_write_file(filp); return error; } From d551ed0e7ed31ad4ad18a970c0dd7e714b2c56e2 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:02 +0200 Subject: [PATCH 815/857] xfs: rename xfs_ioctl_fs_counts to xfs_ioc_fs_counts Match the naming scheme of most other ioctl handlers. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 314eeec0234171..e0da8dfc9ae67d 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1183,7 +1183,7 @@ xfs_ioctl_getset_resblocks( } static int -xfs_ioctl_fs_counts( +xfs_ioc_fs_counts( struct xfs_mount *mp, struct xfs_fsop_counts __user *uarg) { @@ -1347,7 +1347,7 @@ xfs_file_ioctl( } case XFS_IOC_FSCOUNTS: - return xfs_ioctl_fs_counts(mp, arg); + return xfs_ioc_fs_counts(mp, arg); case XFS_IOC_SET_RESBLKS: case XFS_IOC_GET_RESBLKS: From ddb7719b20b71aa587da198f6fc84291542a13aa Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:03 +0200 Subject: [PATCH 816/857] xfs: rename xfs_ioctl_getset_resblocks to xfs_ioc_getset_resblocks Match the naming scheme of most other ioctl handlers. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index e0da8dfc9ae67d..479947456b1f02 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1144,7 +1144,7 @@ xfs_fs_eofblocks_from_user( } static int -xfs_ioctl_getset_resblocks( +xfs_ioc_getset_resblocks( struct file *filp, unsigned int cmd, void __user *arg) @@ -1351,7 +1351,7 @@ xfs_file_ioctl( case XFS_IOC_SET_RESBLKS: case XFS_IOC_GET_RESBLKS: - return xfs_ioctl_getset_resblocks(filp, cmd, arg); + return xfs_ioc_getset_resblocks(filp, cmd, arg); case XFS_IOC_FSGROWFSDATA: { struct xfs_growfs_data in; From 1e6f3dd66fc19affea43d7760d4ec2e75dafe2e7 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:04 +0200 Subject: [PATCH 817/857] xfs: split out the handler for XFS_IOC_DIOINFO Split out a helper for XFS_IOC_DIOINFO to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 47 +++++++++++++++++++++++++++------------------- 1 file changed, 28 insertions(+), 19 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 479947456b1f02..48cee2bf6691ff 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1200,6 +1200,32 @@ xfs_ioc_fs_counts( return 0; } +static int +xfs_ioc_dioinfo( + struct file *file, + void __user *arg) +{ + struct kstat st; + struct dioattr da; + int error; + + error = vfs_getattr(&file->f_path, &st, STATX_DIOALIGN, 0); + if (error) + return error; + + /* + * Some userspace directly feeds the return value to posix_memalign, + * which fails for values that are smaller than the pointer size. + * Round up the value to not break userspace. + */ + da.d_mem = roundup(st.dio_mem_align, sizeof(void *)); + da.d_miniosz = st.dio_offset_align; + da.d_maxiosz = INT_MAX & ~(da.d_miniosz - 1); + if (copy_to_user(arg, &da, sizeof(da))) + return -EFAULT; + return 0; +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1238,26 +1264,9 @@ xfs_file_ioctl( "%s should use fallocate; XFS_IOC_{ALLOC,FREE}SP ioctl unsupported", current->comm); return -ENOTTY; - case XFS_IOC_DIOINFO: { - struct kstat st; - struct dioattr da; - - error = vfs_getattr(&filp->f_path, &st, STATX_DIOALIGN, 0); - if (error) - return error; - /* - * Some userspace directly feeds the return value to - * posix_memalign, which fails for values that are smaller than - * the pointer size. Round up the value to not break userspace. - */ - da.d_mem = roundup(st.dio_mem_align, sizeof(void *)); - da.d_miniosz = st.dio_offset_align; - da.d_maxiosz = INT_MAX & ~(da.d_miniosz - 1); - if (copy_to_user(arg, &da, sizeof(da))) - return -EFAULT; - return 0; - } + case XFS_IOC_DIOINFO: + return xfs_ioc_dioinfo(filp, arg); case XFS_IOC_FSBULKSTAT_SINGLE: case XFS_IOC_FSBULKSTAT: From b6128807c04dab41194195ed27633e6c2c8a5019 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:05 +0200 Subject: [PATCH 818/857] xfs: split out the handlers for XFS_IOC_.*HANDLE Split out helpers for XFS_IOC_.*HANDLE to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 65 ++++++++++++++++++++++++++++++---------------- 1 file changed, 42 insertions(+), 23 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 48cee2bf6691ff..a038df0196f70d 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1226,6 +1226,42 @@ xfs_ioc_dioinfo( return 0; } +static int +xfs_ioc_find_handle( + unsigned int cmd, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (copy_from_user(&hreq, arg, sizeof(hreq))) + return -EFAULT; + return xfs_find_handle(cmd, &hreq); +} + +static int +xfs_ioc_open_by_handle( + struct file *file, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (copy_from_user(&hreq, arg, sizeof(hreq))) + return -EFAULT; + return xfs_open_by_handle(file, &hreq); +} + +static int +xfs_ioc_readlink_by_handle( + struct file *file, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (copy_from_user(&hreq, arg, sizeof(hreq))) + return -EFAULT; + return xfs_readlink_by_handle(file, &hreq); +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1314,31 +1350,14 @@ xfs_file_ioctl( case XFS_IOC_FD_TO_HANDLE: case XFS_IOC_PATH_TO_HANDLE: - case XFS_IOC_PATH_TO_FSHANDLE: { - xfs_fsop_handlereq_t hreq; - - if (copy_from_user(&hreq, arg, sizeof(hreq))) - return -EFAULT; - return xfs_find_handle(cmd, &hreq); - } - case XFS_IOC_OPEN_BY_HANDLE: { - xfs_fsop_handlereq_t hreq; - - if (copy_from_user(&hreq, arg, sizeof(xfs_fsop_handlereq_t))) - return -EFAULT; - return xfs_open_by_handle(filp, &hreq); - } - - case XFS_IOC_READLINK_BY_HANDLE: { - xfs_fsop_handlereq_t hreq; - - if (copy_from_user(&hreq, arg, sizeof(xfs_fsop_handlereq_t))) - return -EFAULT; - return xfs_readlink_by_handle(filp, &hreq); - } + case XFS_IOC_PATH_TO_FSHANDLE: + return xfs_ioc_find_handle(cmd, arg); + case XFS_IOC_OPEN_BY_HANDLE: + return xfs_ioc_open_by_handle(filp, arg); + case XFS_IOC_READLINK_BY_HANDLE: + return xfs_ioc_readlink_by_handle(filp, arg); case XFS_IOC_ATTRLIST_BY_HANDLE: return xfs_attrlist_by_handle(filp, arg); - case XFS_IOC_ATTRMULTI_BY_HANDLE: return xfs_attrmulti_by_handle(filp, arg); From 4f80ca267bd25af2eaa31e749cf17ac145f8b1e8 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:06 +0200 Subject: [PATCH 819/857] xfs: split out the handler for XFS_IOC_SWAPEXT Split out a helper for XFS_IOC_SWAPEXT to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 33 +++++++++++++++++++++------------ 1 file changed, 21 insertions(+), 12 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index a038df0196f70d..ecee39e2ac1943 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1262,6 +1262,25 @@ xfs_ioc_readlink_by_handle( return xfs_readlink_by_handle(file, &hreq); } +static int +xfs_ioc_swapext( + struct file *file, + void __user *arg) +{ + struct xfs_swapext sxp; + int error; + + if (copy_from_user(&sxp, arg, sizeof(sxp))) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_swapext(&sxp); + mnt_drop_write_file(file); + return error; +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1361,18 +1380,8 @@ xfs_file_ioctl( case XFS_IOC_ATTRMULTI_BY_HANDLE: return xfs_attrmulti_by_handle(filp, arg); - case XFS_IOC_SWAPEXT: { - struct xfs_swapext sxp; - - if (copy_from_user(&sxp, arg, sizeof(xfs_swapext_t))) - return -EFAULT; - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_swapext(&sxp); - mnt_drop_write_file(filp); - return error; - } + case XFS_IOC_SWAPEXT: + return xfs_ioc_swapext(filp, arg); case XFS_IOC_FSCOUNTS: return xfs_ioc_fs_counts(mp, arg); From 92b64934058722ea435c3062a2c209b1c03823c5 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:07 +0200 Subject: [PATCH 820/857] xfs: split out the handlers for XFS_IOC_FSGROWFS* Split out helpers for XFS_IOC_FSGROWFS* to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 107 ++++++++++++++++++++++++++++----------------- 1 file changed, 66 insertions(+), 41 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index ecee39e2ac1943..92bdf0e37f3fb8 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1281,6 +1281,66 @@ xfs_ioc_swapext( return error; } +static int +xfs_ioc_growfs_data( + struct file *file, + struct xfs_mount *mp, + void __user *arg) +{ + struct xfs_growfs_data in; + int error; + + if (copy_from_user(&in, arg, sizeof(in))) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_data(mp, &in); + mnt_drop_write_file(file); + return error; +} + +static int +xfs_ioc_growfs_log( + struct file *file, + struct xfs_mount *mp, + void __user *arg) +{ + struct xfs_growfs_log in; + int error; + + if (copy_from_user(&in, arg, sizeof(in))) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_log(mp, &in); + mnt_drop_write_file(file); + return error; +} + +static int +xfs_ioc_growfs_rt( + struct file *file, + struct xfs_mount *mp, + void __user *arg) +{ + struct xfs_growfs_rt in; + int error; + + if (copy_from_user(&in, arg, sizeof(in))) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_rt(mp, &in); + mnt_drop_write_file(file); + return error; +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1390,47 +1450,12 @@ xfs_file_ioctl( case XFS_IOC_GET_RESBLKS: return xfs_ioc_getset_resblocks(filp, cmd, arg); - case XFS_IOC_FSGROWFSDATA: { - struct xfs_growfs_data in; - - if (copy_from_user(&in, arg, sizeof(in))) - return -EFAULT; - - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_data(mp, &in); - mnt_drop_write_file(filp); - return error; - } - - case XFS_IOC_FSGROWFSLOG: { - struct xfs_growfs_log in; - - if (copy_from_user(&in, arg, sizeof(in))) - return -EFAULT; - - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_log(mp, &in); - mnt_drop_write_file(filp); - return error; - } - - case XFS_IOC_FSGROWFSRT: { - xfs_growfs_rt_t in; - - if (copy_from_user(&in, arg, sizeof(in))) - return -EFAULT; - - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_rt(mp, &in); - mnt_drop_write_file(filp); - return error; - } + case XFS_IOC_FSGROWFSDATA: + return xfs_ioc_growfs_data(filp, mp, arg); + case XFS_IOC_FSGROWFSLOG: + return xfs_ioc_growfs_log(filp, mp, arg); + case XFS_IOC_FSGROWFSRT: + return xfs_ioc_growfs_rt(filp, mp, arg); case XFS_IOC_GOINGDOWN: { uint32_t in; From 3534db92bfd72f4145339dc77564094a8e00ea61 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:08 +0200 Subject: [PATCH 821/857] xfs: split out the handler for XFS_IOC_GOINGDOWN Split out a helper for XFS_IOC_GOINGDOWN to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 28 ++++++++++++++++------------ 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 92bdf0e37f3fb8..1580f8d1299ca5 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1341,6 +1341,20 @@ xfs_ioc_growfs_rt( return error; } +static int +xfs_ioc_goingdown( + struct xfs_mount *mp, + uint32_t __user *arg) +{ + uint32_t in; + + if (!capable(CAP_SYS_ADMIN)) + return -EPERM; + if (get_user(in, arg)) + return -EFAULT; + return xfs_fs_goingdown(mp, in); +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1457,18 +1471,8 @@ xfs_file_ioctl( case XFS_IOC_FSGROWFSRT: return xfs_ioc_growfs_rt(filp, mp, arg); - case XFS_IOC_GOINGDOWN: { - uint32_t in; - - if (!capable(CAP_SYS_ADMIN)) - return -EPERM; - - if (get_user(in, (uint32_t __user *)arg)) - return -EFAULT; - - return xfs_fs_goingdown(mp, in); - } - + case XFS_IOC_GOINGDOWN: + return xfs_ioc_goingdown(mp, arg); case XFS_IOC_ERROR_INJECTION: { xfs_error_injection_t in; From a3083bc5a61748dc95140d668406217f73e04e0c Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:09 +0200 Subject: [PATCH 822/857] xfs: split out the handler for XFS_IOC_ERROR_INJECTION Split out a helper for XFS_IOC_ERROR_INJECTION to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 29 ++++++++++++++++------------- 1 file changed, 16 insertions(+), 13 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 1580f8d1299ca5..a226b33a74b517 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1355,6 +1355,20 @@ xfs_ioc_goingdown( return xfs_fs_goingdown(mp, in); } +static int +xfs_ioc_error_injection( + struct xfs_mount *mp, + struct xfs_error_injection __user *arg) +{ + struct xfs_error_injection in; + + if (!capable(CAP_SYS_ADMIN)) + return -EPERM; + if (copy_from_user(&in, arg, sizeof(in))) + return -EFAULT; + return xfs_errortag_add(mp, in.errtag); +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1473,22 +1487,11 @@ xfs_file_ioctl( case XFS_IOC_GOINGDOWN: return xfs_ioc_goingdown(mp, arg); - case XFS_IOC_ERROR_INJECTION: { - xfs_error_injection_t in; - - if (!capable(CAP_SYS_ADMIN)) - return -EPERM; - - if (copy_from_user(&in, arg, sizeof(in))) - return -EFAULT; - - return xfs_errortag_add(mp, in.errtag); - } - + case XFS_IOC_ERROR_INJECTION: + return xfs_ioc_error_injection(mp, arg); case XFS_IOC_ERROR_CLEARALL: if (!capable(CAP_SYS_ADMIN)) return -EPERM; - return xfs_errortag_clearall(mp); case XFS_IOC_FREE_EOFBLOCKS: { From 126092c37ae86cf74f1a945b7ece05fa5c6444d5 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:10 +0200 Subject: [PATCH 823/857] xfs: split out the handler for XFS_IOC_FREE_EOFBLOCKS Split out a helper for XFS_IOC_FREE_EOFBLOCKS to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 52 ++++++++++++++++++++++++++-------------------- 1 file changed, 29 insertions(+), 23 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index a226b33a74b517..10673cf7f2d6fd 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1369,6 +1369,33 @@ xfs_ioc_error_injection( return xfs_errortag_add(mp, in.errtag); } +static int +xfs_ioc_free_eofblocks( + struct xfs_mount *mp, + struct xfs_fs_eofblocks __user *arg) +{ + struct xfs_fs_eofblocks eofb; + struct xfs_icwalk icw; + int error; + + if (!capable(CAP_SYS_ADMIN)) + return -EPERM; + if (xfs_is_readonly(mp)) + return -EROFS; + + if (copy_from_user(&eofb, arg, sizeof(eofb))) + return -EFAULT; + + error = xfs_fs_eofblocks_from_user(&eofb, &icw); + if (error) + return error; + + trace_xfs_ioc_free_eofblocks(mp, &icw, _RET_IP_); + + guard(super_write)(mp->m_super); + return xfs_blockgc_free_space(mp, &icw); +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1388,7 +1415,6 @@ xfs_file_ioctl( struct xfs_inode *ip = XFS_I(inode); struct xfs_mount *mp = ip->i_mount; void __user *arg = (void __user *)p; - int error; trace_xfs_file_ioctl(ip); @@ -1494,28 +1520,8 @@ xfs_file_ioctl( return -EPERM; return xfs_errortag_clearall(mp); - case XFS_IOC_FREE_EOFBLOCKS: { - struct xfs_fs_eofblocks eofb; - struct xfs_icwalk icw; - - if (!capable(CAP_SYS_ADMIN)) - return -EPERM; - - if (xfs_is_readonly(mp)) - return -EROFS; - - if (copy_from_user(&eofb, arg, sizeof(eofb))) - return -EFAULT; - - error = xfs_fs_eofblocks_from_user(&eofb, &icw); - if (error) - return error; - - trace_xfs_ioc_free_eofblocks(mp, &icw, _RET_IP_); - - guard(super_write)(mp->m_super); - return xfs_blockgc_free_space(mp, &icw); - } + case XFS_IOC_FREE_EOFBLOCKS: + return xfs_ioc_free_eofblocks(mp, arg); case XFS_IOC_EXCHANGE_RANGE: return xfs_ioc_exchange_range(filp, arg); From 1e4c3089ab63d3afee702243630d9c3c919a046b Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:11 +0200 Subject: [PATCH 824/857] xfs: split out the handlers for XFS_IOC_FSGROWFS_*_32 Split out helpers for XFS_IOC_FSGROWFS_*_32 to keep the stack variables out of xfs_file_compat_ioctl and to clean up the main compat ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 76 ++++++++++++++++++++++---------------------- 1 file changed, 38 insertions(+), 38 deletions(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index 250df2bbf21417..2162eda017430f 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -44,26 +44,46 @@ xfs_compat_ioc_fsgeometry_v1( return 0; } -STATIC int -xfs_compat_growfs_data_copyin( - struct xfs_growfs_data *in, - compat_xfs_growfs_data_t __user *arg32) +static int +xfs_compat_ioc_growfs_data( + struct file *file, + struct xfs_mount *mp, + struct compat_xfs_growfs_data __user *arg32) { - if (get_user(in->newblocks, &arg32->newblocks) || - get_user(in->imaxpct, &arg32->imaxpct)) + struct xfs_growfs_data in = { }; + int error; + + if (get_user(in.newblocks, &arg32->newblocks) || + get_user(in.imaxpct, &arg32->imaxpct)) return -EFAULT; - return 0; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_data(mp, &in); + mnt_drop_write_file(file); + return error; } -STATIC int -xfs_compat_growfs_rt_copyin( - struct xfs_growfs_rt *in, - compat_xfs_growfs_rt_t __user *arg32) +static int +xfs_compat_ioc_growfs_rt( + struct file *file, + struct xfs_mount *mp, + struct compat_xfs_growfs_rt __user *arg32) { - if (get_user(in->newblocks, &arg32->newblocks) || - get_user(in->extsize, &arg32->extsize)) + struct xfs_growfs_rt in = {}; + int error; + + if (get_user(in.newblocks, &arg32->newblocks) || + get_user(in.extsize, &arg32->extsize)) return -EFAULT; - return 0; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_rt(mp, &in); + mnt_drop_write_file(file); + return error; } STATIC int @@ -434,30 +454,10 @@ xfs_file_compat_ioctl( #if defined(BROKEN_X86_ALIGNMENT) case XFS_IOC_FSGEOMETRY_V1_32: return xfs_compat_ioc_fsgeometry_v1(ip->i_mount, arg); - case XFS_IOC_FSGROWFSDATA_32: { - struct xfs_growfs_data in; - - if (xfs_compat_growfs_data_copyin(&in, arg)) - return -EFAULT; - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_data(ip->i_mount, &in); - mnt_drop_write_file(filp); - return error; - } - case XFS_IOC_FSGROWFSRT_32: { - struct xfs_growfs_rt in; - - if (xfs_compat_growfs_rt_copyin(&in, arg)) - return -EFAULT; - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_rt(ip->i_mount, &in); - mnt_drop_write_file(filp); - return error; - } + case XFS_IOC_FSGROWFSDATA_32: + return xfs_compat_ioc_growfs_data(filp, ip->i_mount, arg); + case XFS_IOC_FSGROWFSRT_32: + return xfs_compat_ioc_growfs_rt(filp, ip->i_mount, arg); #endif /* long changes size, but xfs only copiese out 32 bits */ case XFS_IOC_GETVERSION_32: From a0f3142afdcb309731e79d71b9c5eebacd7cb366 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:12 +0200 Subject: [PATCH 825/857] xfs: cleanup XFS_IOC_GETVERSION_32 handling Move the cmd assignment into the function call, fix a spelling error in the comment and move the comment next to the call. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index 2162eda017430f..ede890d4cbc38c 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -459,10 +459,9 @@ xfs_file_compat_ioctl( case XFS_IOC_FSGROWFSRT_32: return xfs_compat_ioc_growfs_rt(filp, ip->i_mount, arg); #endif - /* long changes size, but xfs only copiese out 32 bits */ case XFS_IOC_GETVERSION_32: - cmd = _NATIVE_IOC(cmd, long); - return xfs_file_ioctl(filp, cmd, p); + /* long changes size, but xfs only copies out 32 bits */ + return xfs_file_ioctl(filp, _NATIVE_IOC(cmd, long), p); case XFS_IOC_SWAPEXT_32: { struct xfs_swapext sxp; struct compat_xfs_swapext __user *sxu = arg; From c1f331382b613142c2ebe66d1c506623f9235183 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:13 +0200 Subject: [PATCH 826/857] xfs: split out the handlers for XFS_IOC_SWAPEXT_32 Split out a helper for XFS_IOC_SWAPEXT_32 to keep the stack variables out of xfs_file_compat_ioctl and to clean up the main compat ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 40 +++++++++++++++++++++++----------------- 1 file changed, 23 insertions(+), 17 deletions(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index ede890d4cbc38c..a6e3b35db6e20b 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -158,6 +158,27 @@ xfs_ioctl32_bstat_copyin( return 0; } +static int +xfs_compat_ioc_swapext( + struct file *file, + struct compat_xfs_swapext __user *sxu) +{ + struct xfs_swapext sxp; + int error; + + /* Bulk copy in up to the sx_stat field, then copy bstat */ + if (copy_from_user(&sxp, sxu, offsetof(struct xfs_swapext, sx_stat)) || + xfs_ioctl32_bstat_copyin(&sxp.sx_stat, &sxu->sx_stat)) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_swapext(&sxp); + mnt_drop_write_file(file); + return error; +} + /* XFS_IOC_FSBULKSTAT and friends */ STATIC int @@ -446,7 +467,6 @@ xfs_file_compat_ioctl( struct inode *inode = file_inode(filp); struct xfs_inode *ip = XFS_I(inode); void __user *arg = compat_ptr(p); - int error; trace_xfs_file_compat_ioctl(ip); @@ -462,22 +482,8 @@ xfs_file_compat_ioctl( case XFS_IOC_GETVERSION_32: /* long changes size, but xfs only copies out 32 bits */ return xfs_file_ioctl(filp, _NATIVE_IOC(cmd, long), p); - case XFS_IOC_SWAPEXT_32: { - struct xfs_swapext sxp; - struct compat_xfs_swapext __user *sxu = arg; - - /* Bulk copy in up to the sx_stat field, then copy bstat */ - if (copy_from_user(&sxp, sxu, - offsetof(struct xfs_swapext, sx_stat)) || - xfs_ioctl32_bstat_copyin(&sxp.sx_stat, &sxu->sx_stat)) - return -EFAULT; - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_swapext(&sxp); - mnt_drop_write_file(filp); - return error; - } + case XFS_IOC_SWAPEXT_32: + return xfs_compat_ioc_swapext(filp, arg); case XFS_IOC_FSBULKSTAT_32: case XFS_IOC_FSBULKSTAT_SINGLE_32: case XFS_IOC_FSINUMBERS_32: From 0a0a7894f3db576575ab57d0279b176271900df6 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:14 +0200 Subject: [PATCH 827/857] xfs: split out the handlers for XFS_IOC_*_BY_HANDLE_32 Split out helpers for XFS_IOC_*_BY_HANDLE_32 to keep the stack variables out of xfs_file_compat_ioctl and to clean up the main compat ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 65 +++++++++++++++++++++++++++++--------------- 1 file changed, 43 insertions(+), 22 deletions(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index a6e3b35db6e20b..688eb3300495d6 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -379,6 +379,43 @@ xfs_compat_handlereq_to_dentry( compat_ptr(hreq->ihandle), hreq->ihandlen); } +static int +xfs_compat_ioc_find_handle( + unsigned int cmd, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (xfs_compat_handlereq_copyin(&hreq, arg)) + return -EFAULT; + return xfs_find_handle(_NATIVE_IOC(cmd, struct xfs_fsop_handlereq), + &hreq); +} + +static int +xfs_compat_ioc_open_by_handle( + struct file *file, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (xfs_compat_handlereq_copyin(&hreq, arg)) + return -EFAULT; + return xfs_open_by_handle(file, &hreq); +} + +static int +xfs_compat_ioc_readlink_by_handle( + struct file *file, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (xfs_compat_handlereq_copyin(&hreq, arg)) + return -EFAULT; + return xfs_readlink_by_handle(file, &hreq); +} + STATIC int xfs_compat_attrlist_by_handle( struct file *parfilp, @@ -490,28 +527,12 @@ xfs_file_compat_ioctl( return xfs_compat_ioc_fsbulkstat(filp, cmd, arg); case XFS_IOC_FD_TO_HANDLE_32: case XFS_IOC_PATH_TO_HANDLE_32: - case XFS_IOC_PATH_TO_FSHANDLE_32: { - struct xfs_fsop_handlereq hreq; - - if (xfs_compat_handlereq_copyin(&hreq, arg)) - return -EFAULT; - cmd = _NATIVE_IOC(cmd, struct xfs_fsop_handlereq); - return xfs_find_handle(cmd, &hreq); - } - case XFS_IOC_OPEN_BY_HANDLE_32: { - struct xfs_fsop_handlereq hreq; - - if (xfs_compat_handlereq_copyin(&hreq, arg)) - return -EFAULT; - return xfs_open_by_handle(filp, &hreq); - } - case XFS_IOC_READLINK_BY_HANDLE_32: { - struct xfs_fsop_handlereq hreq; - - if (xfs_compat_handlereq_copyin(&hreq, arg)) - return -EFAULT; - return xfs_readlink_by_handle(filp, &hreq); - } + case XFS_IOC_PATH_TO_FSHANDLE_32: + return xfs_compat_ioc_find_handle(cmd, arg); + case XFS_IOC_OPEN_BY_HANDLE_32: + return xfs_compat_ioc_open_by_handle(filp, arg); + case XFS_IOC_READLINK_BY_HANDLE_32: + return xfs_compat_ioc_readlink_by_handle(filp, arg); case XFS_IOC_ATTRLIST_BY_HANDLE_32: return xfs_compat_attrlist_by_handle(filp, arg); case XFS_IOC_ATTRMULTI_BY_HANDLE_32: From c44f3db4f4c00c5250ddb290bb7cbd2a5ed40155 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:15 +0200 Subject: [PATCH 828/857] xfs: remove an extra cast in xfs_file_compat_ioctl The ioctl argument is already available as an unsigned long in the p variable, so use that directly instead of casting back from p, which has the same value but was casted to a void pointer before. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index 688eb3300495d6..f6875a2cc7065d 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -539,6 +539,6 @@ xfs_file_compat_ioctl( return xfs_compat_attrmulti_by_handle(filp, arg); default: /* try the native version */ - return xfs_file_ioctl(filp, cmd, (unsigned long)arg); + return xfs_file_ioctl(filp, cmd, p); } } From 566cb59f63623626823d832c17a773a5a692feab Mon Sep 17 00:00:00 2001 From: Chengchang Tang Date: Mon, 31 Aug 2026 10:03:23 +0800 Subject: [PATCH 829/857] RDMA/hns: Fix missing CQE when UD QP use different SL Due to the hardware contraint, CQEs may be dropped when a UD QP sends with multiple SLs. Pin the SL to the value from the first post_send on the QP to avoid this issue. Fixes: 66d86e529dd5 ("RDMA/hns: Add UD support for HIP09") Signed-off-by: Chengchang Tang Signed-off-by: Junxian Huang Link: https://patch.msgid.link/20260831020324.3540901-2-huangjunxian6@hisilicon.com Signed-off-by: Leon Romanovsky --- drivers/infiniband/hw/hns/hns_roce_device.h | 1 + drivers/infiniband/hw/hns/hns_roce_hw_v2.c | 15 ++++++++++----- 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/drivers/infiniband/hw/hns/hns_roce_device.h b/drivers/infiniband/hw/hns/hns_roce_device.h index eb7b1865e4c798..95529251e2e217 100644 --- a/drivers/infiniband/hw/hns/hns_roce_device.h +++ b/drivers/infiniband/hw/hns/hns_roce_device.h @@ -654,6 +654,7 @@ struct hns_roce_qp { u8 priority; spinlock_t flush_lock; struct hns_roce_dip *dip; + bool ud_sl_set; }; struct hns_roce_ib_iboe { diff --git a/drivers/infiniband/hw/hns/hns_roce_hw_v2.c b/drivers/infiniband/hw/hns/hns_roce_hw_v2.c index 27cc7df55ee70d..fd198faf4ee47c 100644 --- a/drivers/infiniband/hw/hns/hns_roce_hw_v2.c +++ b/drivers/infiniband/hw/hns/hns_roce_hw_v2.c @@ -431,7 +431,8 @@ static int set_ud_opcode(struct hns_roce_v2_ud_send_wqe *ud_sq_wqe, return 0; } -static int fill_ud_av(struct hns_roce_v2_ud_send_wqe *ud_sq_wqe, +static int fill_ud_av(struct hns_roce_qp *qp, + struct hns_roce_v2_ud_send_wqe *ud_sq_wqe, struct hns_roce_ah *ah) { struct ib_device *ib_dev = ah->ibah.device; @@ -441,7 +442,12 @@ static int fill_ud_av(struct hns_roce_v2_ud_send_wqe *ud_sq_wqe, hr_reg_write(ud_sq_wqe, UD_SEND_WQE_HOPLIMIT, ah->av.hop_limit); hr_reg_write(ud_sq_wqe, UD_SEND_WQE_TCLASS, ah->av.tclass); hr_reg_write(ud_sq_wqe, UD_SEND_WQE_FLOW_LABEL, ah->av.flowlabel); - hr_reg_write(ud_sq_wqe, UD_SEND_WQE_SL, ah->av.sl); + if (!qp->ud_sl_set) { + qp->sl = ah->av.sl; + qp->ud_sl_set = true; + } + + hr_reg_write(ud_sq_wqe, UD_SEND_WQE_SL, qp->sl); ud_sq_wqe->sgid_index = ah->av.gid_index; @@ -491,12 +497,10 @@ static inline int set_ud_wqe(struct hns_roce_qp *qp, qp->qkey : ud_wr(wr)->remote_qkey); hr_reg_write(ud_sq_wqe, UD_SEND_WQE_DQPN, ud_wr(wr)->remote_qpn); - ret = fill_ud_av(ud_sq_wqe, ah); + ret = fill_ud_av(qp, ud_sq_wqe, ah); if (ret) return ret; - qp->sl = to_hr_ah(ud_wr(wr)->ah)->av.sl; - set_extend_sge(qp, wr->sg_list, &curr_idx, valid_num_sge); /* @@ -5603,6 +5607,7 @@ static void v2_set_flushed_fields(struct ib_qp *ibqp, hr_reg_write(context, QPC_SQ_PRODUCER_IDX, hr_qp->sq.head); hr_reg_clear(qpc_mask, QPC_SQ_PRODUCER_IDX); hr_qp->state = IB_QPS_ERR; + hr_qp->ud_sl_set = false; spin_unlock_irqrestore(&hr_qp->sq.lock, sq_flag); if (ibqp->srq || ibqp->qp_type == IB_QPT_XRC_INI) /* no RQ */ From 6ea2157ce8c7497a2862bd3579a987afe6b129ed Mon Sep 17 00:00:00 2001 From: Chengchang Tang Date: Mon, 31 Aug 2026 10:03:24 +0800 Subject: [PATCH 830/857] RDMA/hns: Support setting GSI QP SL via debugfs Due to the hardware contraint, CQEs may be dropped when a UD QP sends with multiple SLs. For GSI QP, provide debugfs to allow users to set a fixed SL. This is only allowed when the device is link down. Example: # cat /sys/kernel/debug/hns_roce//gsi_sl 0 # echo 3 > /sys/kernel/debug/hns_roce//gsi_sl # cat /sys/kernel/debug/hns_roce//gsi_sl 3 Signed-off-by: Chengchang Tang Signed-off-by: Junxian Huang Link: https://patch.msgid.link/20260831020324.3540901-3-huangjunxian6@hisilicon.com Signed-off-by: Leon Romanovsky --- drivers/infiniband/hw/hns/hns_roce_debugfs.c | 55 ++++++++++++++++++++ drivers/infiniband/hw/hns/hns_roce_debugfs.h | 1 + drivers/infiniband/hw/hns/hns_roce_device.h | 1 + drivers/infiniband/hw/hns/hns_roce_hw_v2.c | 5 +- 4 files changed, 60 insertions(+), 2 deletions(-) diff --git a/drivers/infiniband/hw/hns/hns_roce_debugfs.c b/drivers/infiniband/hw/hns/hns_roce_debugfs.c index 05630f7c915510..2487a80b689a6b 100644 --- a/drivers/infiniband/hw/hns/hns_roce_debugfs.c +++ b/drivers/infiniband/hw/hns/hns_roce_debugfs.c @@ -339,6 +339,60 @@ static void create_cc_param_debugfs(struct hns_roce_dev *hr_dev, } } +static int gsi_sl_debugfs_show(struct seq_file *file, void *offset) +{ + struct hns_roce_dev *hr_dev = file->private; + + seq_printf(file, "%u\n", hr_dev->gsi_sl); + + return 0; +} + +static ssize_t gsi_sl_debugfs_store(char *buf, size_t count, void *data) +{ + struct hns_roce_dev *hr_dev = data; + struct net_device *netdev; + int ret; + u8 val; + + netdev = ib_device_get_netdev(&hr_dev->ib_dev, 1); + if (!netdev) + return -ENODEV; + + if (ib_get_curr_port_state(netdev) == IB_PORT_ACTIVE) { + ret = -EOPNOTSUPP; + goto out; + } + + ret = kstrtou8(buf, 0, &val); + if (ret) + goto out; + + if (!check_sl_valid(hr_dev, val)) { + ret = -EINVAL; + goto out; + } + + hr_dev->gsi_sl = val; + +out: + dev_put(netdev); + return ret ? : count; +} + +static void create_gsi_sl_debugfs(struct hns_roce_dev *hr_dev, + struct dentry *parent) +{ + struct hns_debugfs_seqfile *seqfile = &hr_dev->dbgfs.gsi_sl; + + seqfile->read = gsi_sl_debugfs_show; + seqfile->write = gsi_sl_debugfs_store; + seqfile->data = hr_dev; + + debugfs_create_file("gsi_sl", 0600, parent, seqfile, + &hns_debugfs_seqfile_fops); +} + /* debugfs for device */ void hns_roce_register_debugfs(struct hns_roce_dev *hr_dev) { @@ -349,6 +403,7 @@ void hns_roce_register_debugfs(struct hns_roce_dev *hr_dev) create_sw_stat_debugfs(hr_dev, dbgfs->root); create_cc_param_debugfs(hr_dev, dbgfs->root); + create_gsi_sl_debugfs(hr_dev, dbgfs->root); } void hns_roce_unregister_debugfs(struct hns_roce_dev *hr_dev) diff --git a/drivers/infiniband/hw/hns/hns_roce_debugfs.h b/drivers/infiniband/hw/hns/hns_roce_debugfs.h index 116b1e8b66772a..286704711785c2 100644 --- a/drivers/infiniband/hw/hns/hns_roce_debugfs.h +++ b/drivers/infiniband/hw/hns/hns_roce_debugfs.h @@ -47,6 +47,7 @@ struct hns_roce_dev_debugfs { struct dentry *root; struct hns_sw_stat_debugfs sw_stat_root; struct hns_cc_param_debugfs cc_param_root[CONG_TYPE_MAX_NUM]; + struct hns_debugfs_seqfile gsi_sl; }; struct hns_roce_dev; diff --git a/drivers/infiniband/hw/hns/hns_roce_device.h b/drivers/infiniband/hw/hns/hns_roce_device.h index 95529251e2e217..f4f899e87ea63f 100644 --- a/drivers/infiniband/hw/hns/hns_roce_device.h +++ b/drivers/infiniband/hw/hns/hns_roce_device.h @@ -1048,6 +1048,7 @@ struct hns_roce_dev { struct hns_roce_dev_debugfs dbgfs; atomic64_t *dfx_cnt; struct hns_roce_scc_param *scc_param; + u8 gsi_sl; }; enum hns_roce_trace_type { diff --git a/drivers/infiniband/hw/hns/hns_roce_hw_v2.c b/drivers/infiniband/hw/hns/hns_roce_hw_v2.c index fd198faf4ee47c..368e1d74c2833e 100644 --- a/drivers/infiniband/hw/hns/hns_roce_hw_v2.c +++ b/drivers/infiniband/hw/hns/hns_roce_hw_v2.c @@ -442,8 +442,9 @@ static int fill_ud_av(struct hns_roce_qp *qp, hr_reg_write(ud_sq_wqe, UD_SEND_WQE_HOPLIMIT, ah->av.hop_limit); hr_reg_write(ud_sq_wqe, UD_SEND_WQE_TCLASS, ah->av.tclass); hr_reg_write(ud_sq_wqe, UD_SEND_WQE_FLOW_LABEL, ah->av.flowlabel); - if (!qp->ud_sl_set) { - qp->sl = ah->av.sl; + if (!qp->ud_sl_set || qp->ibqp.qp_type == IB_QPT_GSI) { + qp->sl = qp->ibqp.qp_type == IB_QPT_GSI ? + hr_dev->gsi_sl : ah->av.sl; qp->ud_sl_set = true; } From 9d80aa4617b32f5054c5aa471d06b66704854935 Mon Sep 17 00:00:00 2001 From: Mark Brown Date: Thu, 3 Sep 2026 17:43:15 +0100 Subject: [PATCH 831/857] Add linux-next specific files for 20260903 Signed-off-by: Mark Brown --- Next/SHA1s | 428 ++++ Next/Trees | 428 ++++ Next/merge.log | 6264 +++++++++++++++++++++++++++++++++++++++++++++ localversion-next | 1 + 4 files changed, 7121 insertions(+) create mode 100644 Next/SHA1s create mode 100644 Next/Trees create mode 100644 Next/merge.log create mode 100644 localversion-next diff --git a/Next/SHA1s b/Next/SHA1s new file mode 100644 index 00000000000000..b9b516fd1e5dd4 --- /dev/null +++ b/Next/SHA1s @@ -0,0 +1,428 @@ +Name SHA1 +---- ---- +origin 940de590b839f71d6dc846160534bf202401b8b7 +ext4-fixes 981fcc5674e67158d24d23e841523eccba19d0e7 +vfs-brauner-fixes e14d4302cbd0de773960bec33c2281508c8d8855 +fscrypt-current cee9395acd8043be0644b25c34bfa86623f2b935 +fsverity-current cee9395acd8043be0644b25c34bfa86623f2b935 +btrfs-fixes 4d1d66ba287be442749482f6d340b852ddf087fe +vfs-fixes 49c5d168a3a8f4eb27d44a2a22b7e8a856ca601f +erofs-fixes 617d0d8d199ba1790c94310fd75a22d01c97a8d6 +nfsd-fixes 46db3c8a1be96a354758b44b1d4fbb4b70d09a20 +v9fs-fixes 028ef9c96e96197026887c0f092424679298aae8 +overlayfs-fixes 4549871118cf616eecdd2d939f78e3b9e1dddc48 +fscrypt cee9395acd8043be0644b25c34bfa86623f2b935 +btrfs 966bb86e7420c64f20edaac5ea096da9b16dc458 +ceph dc173b37415e8f738fc4de477490056b479ddc9f +cifs 4ee5025d18677575c7b303fdd1bc5f3b8e7ab29c +configfs 2251d0ed97c24a9b96fd47c2efb4b51ef1c39aee +ecryptfs f81cb44f9a4b88d73ee5dec4a1ccdb0232fd2e3f +dlm ed9b6a1296f10e4881d93dfe6d76013fbbaeee87 +erofs a7d28aa0e9b2c983b915d091f9596f0e3b253355 +exfat f096faa6bb169293d04194a0a5d84bc782c683b5 +ext3 a3cf61f60051ca8874f012fa8569fa971fbfbee0 +ext4 9091c97be34083587a75db174aab51551d8e8543 +f2fs cee9395acd8043be0644b25c34bfa86623f2b935 +fsverity cee9395acd8043be0644b25c34bfa86623f2b935 +fuse 10bd6b32965265688fa277d166ab224de8e86eb2 +gfs2 ccaa1524a18aaad2d3646f614f2cbd5941dca50a +jfs dad98c5b2a05ef744af4c884c97066a3c8cdad61 +ksmbd da6066cf54a9c3f54b8f1d75a9352a6588326a17 +nfs cee9395acd8043be0644b25c34bfa86623f2b935 +nfs-anna e053b624f5d36669756990743346157be9f68c34 +nfsd f5dc2038906bb0c9627c99bea06dd7786b5ac2d1 +ntfs 0fecc393f2060e6bc25138df32cb923ec7071c6b +ntfs3 cee9395acd8043be0644b25c34bfa86623f2b935 +orangefs d410cd5303ec59c7cf23dd61423752ce8e9ecb59 +overlayfs 1f6ee9be92f8df85a8c9a5a78c20fd39c0c21a95 +ubifs a5e0055eac837a1168c781653943d4a0d9920af3 +v9fs aa88278693cbfaf7a2acf961379973fbb63b165c +v9fs-ericvh 028ef9c96e96197026887c0f092424679298aae8 +xfs c44f3db4f4c00c5250ddb290bb7cbd2a5ed40155 +zonefs 3a8389d42bdf4213730f4067f8bfa78bae6564ef +vfs-brauner 21860bb220225edeb87e9c831e6bf25c620e2126 +vfs 4dda01b67c8662c5d0c53034974cb8280c575a5d +mm-fixes 60492c2aabf53276fe75bf8b4822d8f72f9d249b +fs-current 9ac4915b36faaec54332daf97960f35e22c65fba +kbuild-current cee9395acd8043be0644b25c34bfa86623f2b935 +clang-fixes 175db11786bde9061db526bf1ac5107d915f5163 +arc-current 1590cf0329716306e948a8fc29f1d3ee87d3989f +arm-current 1039bffd6ae9c75b42b7d148d6c1106134107b66 +arm64-fixes f73a8edc2ccc6ec72c37d5c578e7592d2e1f9922 +arm-soc-fixes e36c0670d5ebebd7dba493f14966051f354fbb34 +davinci-current cee9395acd8043be0644b25c34bfa86623f2b935 +realtek-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +drivers-memory-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +sophgo-fixes 19272b37aa4f83ca52bdf9c16d5d81bdd1354494 +sophgo-soc-fixes 0af2f6be1b4281385b618cb86ad946eded089ac8 +m68k-current 2f8e3cad53b5c36ab0ed5d3195bfc55c59ea61a5 +powerpc-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +s390-fixes 98d23edcd41432286cf03672252507a841323c8c +net 66817a9794263cd2a5dc4e99bf8e5fcc5ff7181e +bpf ac0aaef0aa997fcdcb2458bd584539ba8608d33e +ipsec 96f01b53c2d05e003b040892256de54a586e8529 +netfilter 1b78070aaef63512688aebfbc82365ef9d6660f1 +ipvs 1b78070aaef63512688aebfbc82365ef9d6660f1 +wireless 1b78070aaef63512688aebfbc82365ef9d6660f1 +ath 1b78070aaef63512688aebfbc82365ef9d6660f1 +iwlwifi d13d5d299c11b7bd3362d5692c56225d9e176664 +wpan 2f37fba846c9fdff5fc15b6d93656057ccd13031 +rdma-fixes 3fb905f07ea45b31c8f67ba6e4668de46f527e65 +sound-current 8ba27b90095a4c7fcc878dd013969cd6b100cea7 +sound-asoc-fixes edc5d19f9ffea5ddcb73b25a2de671544dd0e1d2 +regmap-fixes 2e42cade8ff1ff579e77976d9869db4df1feaf74 +regulator-fixes ca12149896ed040dafef92eacd2af3f903afb177 +spi-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pci-current cee9395acd8043be0644b25c34bfa86623f2b935 +driver-core.current f6d752278c13839888425294c110174fb6c87e3d +tty.current cee9395acd8043be0644b25c34bfa86623f2b935 +usb.current c9273c83885835dbd1e8835d5665dfb8503d65e0 +usb-serial-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +phy cee9395acd8043be0644b25c34bfa86623f2b935 +staging.current cc7cd2a9228175c975f62ad56ed7c767701cb4fa +iio-fixes da7c937e0abc86710e237e88bbaaff4a298e48d1 +watchdog-fixes d83b7502bb087fa54daf0fdd419d2910c34bc97d +counter-current f1a3a9946aab611dd2200c01ff122f64b033dad2 +char-misc.current cee9395acd8043be0644b25c34bfa86623f2b935 +soundwire-fixes 6d49beec658f61801e79019fd73691b17252dee3 +thunderbolt-fixes 4310c6b8e75d6a47f7548e5948fbe6318aa4440a +input-current 85f080fb87ed5cd3e46121be677f52c82f26a0ab +crypto-current ee440d4fc0d2f15894ab1f64c474a3adbc858880 +libcrypto-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +vfio-fixes e242e974e812e7a47e3088860c80d9492fac314f +kselftest-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +dmaengine-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +backlight-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +mtd-fixes 2b533e775aec580cf60074417f4ca00ac9cf3580 +mfd-fixes d5d2d7a8d8be18681a0864f58e3875f1c639e11c +v4l-dvb-fixes e04ffff543db06308a8100f6aa3a65aebcd9834c +reset-fixes 71827776667f4e4677a4fa806bcfb24d4b8dd9d7 +mips-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +at91-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +omap-fixes 2fabd2f406d0cf787be47b3bbeea99b30004033d +tegra-fixes 0bc9d07096f17c22350fb63921656136c975b8d0 +kvm-fixes 8d3ae59288f1e7d58d76558a6ee96d533bc5019f +kvms390-fixes f47190b08b71e8482072978373ee88cb2dfbdaf4 +kvm-arm-fixes 679d7201c1f09e37fa1c12ce28d84079c17fc87f +hwmon-fixes a15f90964998e1d7b5f3aa34d29f3acac5971038 +nvdimm-fixes a8aec14230322ed8f1e8042b6d656c1631d41163 +cxl-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +dma-mapping-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +drivers-x86-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +samsung-krzk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-samsung-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-qcom-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +devicetree-fixes 5bb01c657ff9fc807c2c592ca18af34c4fc3bc6f +dt-krzk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +scsi-fixes af8c27375733fb6a6df9fa484cda77cc3dd0cb80 +drm-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +drm-intel-fixes 3785d40831ba5601296283e0197e10e089392757 +mmc-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +rtc-fixes 254f49634ee16a731174d2ae34bc50bd5f45e731 +gnss-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +hyperv-fixes 0fd49f7bfb8f1952ae157344ea9712da74734200 +risc-v-fixes b94cec5761d22624d109d859467d7d4ce0a1b88b +riscv-dt-fixes 42c57c049054dfaa0be83f6721f7c4ce4a880e56 +riscv-soc-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +fpga-fixes 19272b37aa4f83ca52bdf9c16d5d81bdd1354494 +spdx cee9395acd8043be0644b25c34bfa86623f2b935 +gpio-brgl-fixes 1f1d0812f6a8ab8e6f709c599f137c99646512cc +gpio-intel-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-intel-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +auxdisplay-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +kunit-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +renesas-fixes 2ac7bad110be6ebe478d6bd57821f7f5259a1f54 +perf-current aadea57f532882d8bab444646863c7ef8a778ff1 +efi-fixes d8809f6931065cbbf3554647a50a65a471ab5983 +battery-fixes 160a783aa65b74782bc17cb874af1a6d3f5fba3c +iommufd-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +rust-fixes e510334fbaeaa016ac76d80b4c5f47611c5f7860 +w1-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pmdomain-fixes 2b0ac85512b7f67479127b2713254490662eb13d +i2c-andi-fixes b15b548d52b43ba8ac4652bc2c7244a8dd1e9622 +i2c-rust-fixes 4eb422482ca5d924d7212ad2ca1cb7ea6f5b524d +sparc-fixes 254f49634ee16a731174d2ae34bc50bd5f45e731 +clk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +thead-clk-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +tenstorrent-clk-fixes 6de23f81a5e08be8fbf5e8d7e9febc72a5b5f27f +fustini-config-fixes 254f49634ee16a731174d2ae34bc50bd5f45e731 +pwrseq-fixes 3b54dbd119805361695cb50ca6a875f4c7518b74 +thead-dt-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +ftrace-fixes 1650a1b6cb1ae6cb99bb4fce21b30ebdf9fc238e +ring-buffer-fixes 057caace5214da3b457bbd295e1a2ad34d3685ea +trace-fixes 5eab74874d11160725c42ab676ba97a797a362eb +tracefs-fixes 07004a8c4b572171934390148ee48c4175c77eed +spacemit-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +tip-fixes 9cbd68d2653732ea4fb4f1831d541436cc6b864b +kexec-fixes a901b0778ae82d46084b55237e65d3a7bb5f0c86 +liveupdate-fixes 3a0b8fa2eb36afc88b62a95f33f0c77c71fa5ded +drm-msm-fixes e2332abed2a4d3caa59052095dc16e4ce44791ea +uml-fixes af421e9aed3920c7ac88c24daa48606c7112feca +fwctl-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +devsec-tsm-fixes c3fd16c3b98ed726294feab2f94f876290bf7b61 +drm-rust-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +tenstorrent-dt-fixes 6de23f81a5e08be8fbf5e8d7e9febc72a5b5f27f +nfc-fixes 50a0fd24027765eada2532ca27d2056cfe304711 +drm-misc-fixes d3609b540838945ab2ca5b65f32a2eb67bb284c8 +rust cee9395acd8043be0644b25c34bfa86623f2b935 +rust-interop 05f7e89ab9731565d8a62e3b5d1ec206485eeb0b +rust-alloc 82134823a07b4f45096a287f790c169511859617 +rust-io 86731a2a651e58953fc949573895f2fa6d456841 +rust-pin-init 1e26aea0355ad2afa1ccbc62885c01f5bcfc58ca +rust-timekeeping ddb1444d3335129ae87d9796ab1debf41c0ee51b +rust-xarray c455f19bbe6104debd980bb15515faf716bd81b8 +rust-analyzer 5f45afb8ab04d934fc0601a202f95267ebc20059 +mm 19e5fec518a91fc9dfec3f43029f1e0285ef8ecd +mm-nonmm-stable 786262be6048deab760f68c8acc2c85607165894 +mm-nonmm-unstable aa30aa315f5f84e03669afabb65e3e5804cf979b +kbuild 51794b107d54b3c9a8ebbe12a3b81d8495bd482c +clang-format 8f0b4cce4481fb22653697cced8d0d04027cb1e8 +perf 92d50319b4f0c0bbee8a236a09063272cd22faab +compiler-attributes 8f0b4cce4481fb22653697cced8d0d04027cb1e8 +dma-mapping cee9395acd8043be0644b25c34bfa86623f2b935 +asm-generic adbbd9714f8058730f93c8df5c5bf1679456424b +alpha d58041d2c63e09a1c9083e0e9f4151e487c4e16a +arm 1a89abc009cb5035d0cb4e5ba48d32b94f30e8e4 +arm64 2bd533739234d79b74afabfece1abfe9c6d52c83 +arm-perf 5936245125f78d896fdb1bbc2ae79213e28a6579 +arm-soc ae77d827fb0e6fb2faf8186f075c6c337f2dcf08 +amlogic 87e20720aaadf9540c1596b05b93073a99d02258 +asahi-soc 3c709d42fc7a42e6e608497301aa5f4a556e67b9 +at91 d6e7bce8d43920ea01a217b5bccf4a30bf4165b5 +bmc cd7d1ef7d74ed5f5a1790a40389bc38382924556 +broadcom fca5276165993e9ca18e746f9b41af7627cbe349 +cix a0cffbd8878c55b12dc4555f883adf3372cecd91 +davinci cee9395acd8043be0644b25c34bfa86623f2b935 +drivers-memory c8d0e87bc2413c6a3f92c7130a4cbad51f8703da +fsl e14b8d39388095fb806fe9f98a95b70619e4cabd +imx-mxs ff1bb9ccce43cd06c766c3ddef796a7e40e623fe +mediatek f5be25e697e0362103625b1b197af126ae4ba5f7 +mvebu 1dd243147c2fa918eb19cc4f5d67907ff99e7e06 +omap e1743a4a9b0b1bb6249454263f65889936fe9d6f +qcom e61cfc881090cf9de9dbd3b6b7452661dfb0261f +realtek 3c778f0c9fa36b6b9a26bd2146df24ed4e830ba4 +renesas 288e04e37c645c76cfb70a15bf71f6ec65a1f366 +reset d373605cd514837d8a6de3d00c786d4bae6dbaf8 +rockchip 32e0f64640d558a0f5410ac5cdd8ebf371c5e6a6 +samsung-krzk cee9395acd8043be0644b25c34bfa86623f2b935 +scmi 12866c647dc3802b9e4fc55c43b3ff0b12130652 +sophgo 76acfee87c74dc0dc7a68a00b70b9c39d1c8428f +sophgo-soc c8754c7deab4cbfa947fa2d656cbaf83771828ef +spacemit cee9395acd8043be0644b25c34bfa86623f2b935 +stm32 90d4d470d4d478ea011f2ebb6b4ecf199a6d0653 +sunxi 913f7eda9f3e3fb1c67285e144e202494d440049 +tee a3067938fd192b116b6dfb325b654300a6d1b461 +tegra a53c665d36f6121368ed00ada05dd2b905baca57 +tenstorrent-dt 33583baeb1ba7d328e6a9775d889036900b74cdb +fustini-config 020e209272bee0b8add8cba21125e4744f034b26 +thead-dt 15d32aaf6300b832814a9930fdbd20efdbc9adcf +ti 7abd3f29fdf16a6582d79f7e9a2ab3a7604bb82f +xilinx 20df11b6d06664c1333abdc4d7e8123bf62664e3 +socfpga ff98c9832fd430ba7b3158b8521002c8eb8aa16c +clk 551ba775497e93d93bb849f6ee572ca768cbaa3e +clk-imx 39ec460b56b26319d1f31b459e8be7ea34ae9c67 +clk-renesas 26e80ec751c87f437650a2f9eed5ef3ec436bf0b +thead-clk b2ee00d0bf4cca5dd02702fc339d2b523191966f +tenstorrent-clk 23c8ebc952849b3ba47d04d0ec95daf5cc136061 +csky abb81e5ce7d995baa41556b8125fa59e28ba3be8 +loongarch e2a848868441874d416d47a5cba5452b34473504 +m68k cee9395acd8043be0644b25c34bfa86623f2b935 +m68knommu de0dab22cbf6943d0f12f3b1e2eb1bbdd807f039 +microblaze 8518fd17ccef4a7d77877ccb0403b722106ad1a4 +mips cee9395acd8043be0644b25c34bfa86623f2b935 +openrisc 6620f5e8c11c4f7e41222a86f5c97150cc5f84a5 +parisc-hd 8d3ae59288f1e7d58d76558a6ee96d533bc5019f +powerpc cee9395acd8043be0644b25c34bfa86623f2b935 +risc-v 77ae27fd98f3b548797c9f22c10ab5cf1c4ada53 +riscv-dt 82ab962ef6c2a3571e3d590b30c023118c92b776 +riscv-soc 8cdeaa50eae8dad34885515f62559ee83e7e8dda +s390 11acf42968807b5f44d56cd12c225f114dc02c33 +sh b0aa5e4b087b686575f1b31ce54048b4d059b7b8 +sparc 5b2a3b1a98fb47c593144c2770e012d463952b70 +uml 1590cf0329716306e948a8fc29f1d3ee87d3989f +xtensa eb049bdbf2b98d103d93daef44a5e1a0164d01eb +fs-next 054b582169847c0cac8b1d99bb5ac5a4c4fe070d +printk 4f85c1c6efdb69efa809b4baad039a64a2015c36 +pci cee9395acd8043be0644b25c34bfa86623f2b935 +pstore 24b8f8dcb9a139a36cf48bfbe935e8dc1f33ed79 +hid 8dd52b64d08223b2e32092ed510a5f9c074b1dae +i2c 8cd9520d35a6c38db6567e97dd93b1f11f185dc6 +i2c-andi 1f3e66348d2527c31b9bedb0b4d9e28bdad9bc83 +i2c-rust 61ddec70c9bcc3d4b0471c8683e57d3b03cecf0a +i3c cab40cfc9e116acd4d60f95b4b1264cab78f3803 +dmi 1afafbaf749d8e8ec53f8e38efdc731131902b5b +hwmon-staging ba08432bda66a7889d8f3d1581dabf10f59b25eb +jc_docs ab2704c2a884028fd12d455cf27a9585aefe2961 +v4l-dvb ce92674d8d8b29562de397b437e16bee499dd92b +v4l-dvb-next adc218676eef25575469234709c2d87185ca223a +pm 208027d2c8957800e199a9e0f55da5fdb9550207 +cpufreq-arm 40bad7f0aaf0043457a1d54aeb2197fd1b7d729f +cpupower cee9395acd8043be0644b25c34bfa86623f2b935 +devfreq 9a222650d9e70d0027126e4df5b00b1b5b678a97 +pmdomain 685596d9996f2e660e9abab16ded16d54cc0fc99 +opp a5096d4927d1eb607d51a7342a7e7591a3838c19 +thermal b115930d716defc3a7daa2bc2ae2465d864b7114 +rdma 6ea2157ce8c7497a2862bd3579a987afe6b129ed +net-next b35d3d2fae3058265ba937544a3895cedce18d08 +bpf-next d761934c9483ecde93fe99d8705282f716dfee50 +ipsec-next 298bb2b8903323f6ef2eab4819a2e477765f0ff1 +mlx5-next 36b1d3299d0b6aa51485cc9a79b8d94948d5f3d4 +netfilter-next 91ec2035134982b98fab0609a9fd8480e8217dc1 +ipvs-next 91ec2035134982b98fab0609a9fd8480e8217dc1 +bluetooth 6696072ffe07205255cf83621a95a1aa2f9f6e62 +wireless-next 1b78070aaef63512688aebfbc82365ef9d6660f1 +ath-next 1b78070aaef63512688aebfbc82365ef9d6660f1 +iwlwifi-next 4a2610a5a9fcf77eaad82bb030ec77ec5e381db8 +wpan-next a6bfdfcc6711d1d5a92e98644359dedc67c0c858 +wpan-staging a6bfdfcc6711d1d5a92e98644359dedc67c0c858 +mtd 94d32f1ace8ee3ef3537ac6c8e71b76a418a4762 +nand 15a3cbce32994141252bb4ecfe3ff3a5d22d0b4f +spi-nor df415c5e1de0f1aeefacb4e6252ff98d38c04437 +crypto 7537036a2e6fe96f8ed82034f755c54714a0e417 +libcrypto cee9395acd8043be0644b25c34bfa86623f2b935 +drm cee9395acd8043be0644b25c34bfa86623f2b935 +drm-exynos 3a8660878839faadb4f1a6dd72c3179c1df56787 +drm-misc 271e90eb5f9ff34951647e5ed33c1775eebcca50 +amdgpu 8fad652e7235dbf3bebfab1460610b0b0e75b199 +drm-intel 0d63d6fc993b1fb97e314f435bdefbf0ccd947bd +drm-msm 140b13475302601368c0cf4e193e66126a49feb3 +drm-msm-lumag 140b13475302601368c0cf4e193e66126a49feb3 +drm-xe c874bc70c897c554cc460cfe1fa2c715f780d368 +drm-rust 6cb331644c441ff4101a4f8726283a6ed0d5947e +drm-nova 93296e9d9528f0d87f2cf3fee494599060a0f14a +etnaviv 6bde14ba5f7ef59e103ac317df6cc5ac4291ff4a +fbdev 94e6a058b16820e02f25e1221a4c4e713ba23550 +regmap 92e6b920e7f4a64ff483ecbcae339ecc5f7ab9c4 +sound bdeed0642157667bd14912241c83ffd50f9fec22 +ieee1394 68097f9cdd526b81e7832556aaf76acd44dc357e +sound-asoc d687afbc2b9be53d6969a6ac3793c889e776818a +modules cee9395acd8043be0644b25c34bfa86623f2b935 +input fcc1d6eab4ce4ba86ae05a87ecf7ce06cd2ca4d8 +block c300d74d44c27064768a8dea90e67684b00681de +device-mapper 39c5aa3bd8ec3912d2cd0b3fe092642b0d2b0713 +libata bd46a0b22933fb45f4ff009fa4a3836124df34b3 +pcmcia b3c26ea81ccc522e77ed0b1707add61fc9206216 +mmc cee9395acd8043be0644b25c34bfa86623f2b935 +mfd 9d0e4b1ae5b045a2c92b0b9a1c3b268191c219d9 +backlight cf1a12e0804515e0ed8b3f50c1c32f7f425bd660 +battery 2da28b059e0ddcd2e1956eeae383246207965573 +regulator f656e44fc0c84027f56541b64866dcf5297cf805 +security 3a84fc1a21577dbc8da7e4b6a49c9d8b07c07000 +apparmor 3daad923a8685adb66087e0d819559b7eb6ba975 +integrity 8861f6d5c0678a7c5089c7b272509fc5931b8437 +selinux 23be82ec7c9fb4c0c80a0672e0cbd6d1e95a5c16 +smack fedc88e38ce979a720cd2de042578cb5df3dc8de +tomoyo 2d2338c93da79b3bfe4b6099a931d9468d539952 +tpmdd-tpm 22a50c3745c5ca2927300a26ca2d09dd6ce71b0e +tpmdd-keys bf0d7882cd43d6b155d93e482893f685d6f08b89 +watchdog b85ed7f7259b49077955e835ecc32b32b75053e8 +iommu 3c6a2bac15247bc4f28125247ff89165324612fe +audit ca12826b4aac898318e7359bf7a1b1532ae02d2d +devicetree 1e151de2d11ec09766e719427eb0d852fc6db8b8 +dt-krzk 6bd34d56818cc881a75c789ad6ce8ec4a1e32f3a +mailbox 14af7a96afa39f4f3c1972705489b9ba15c01857 +spi 5f01cb141169fe422c42279f7d502479b1ff9fd2 +tip 461735aa6e8e357fb90d2cf827d2b15ce78a1bc7 +kexec c6ed7331aad89976a37469be7be6e2f110378720 +liveupdate c65687600ed075d237a204755d99ad538db34b47 +clockevents 1b8b356b4b06e3a28feb852c04e06caeac07bc87 +edac 9fa765b3ed33d6dd64aca0ba8f6ed7697de50410 +ftrace 3df3ae98c30c249c4bbf3d74559cc6972898052c +rcu 9cc63f8bcd560c760d0b12e15bba8f81c86237cf +paulmck 85f4ab0da78876af7666ce41d429b52d16b56f46 +kvm 76671054f9a1ff6abb976583cd8da37650acdc97 +kvm-arm aa8e5dc6a7a2a1141ab40706a51010adcd0e57d2 +kvms390 dd6f4ef6f8a37412909ad787c837332fb070159c +kvm-ppc cee9395acd8043be0644b25c34bfa86623f2b935 +kvm-riscv b5060a4aa33d7d78a8c1837ab08ef0c81ea96650 +kvm-x86 76671054f9a1ff6abb976583cd8da37650acdc97 +xen-tip d330fb86a7170f845123ae82d95df440fad9b707 +percpu 8f0b4cce4481fb22653697cced8d0d04027cb1e8 +workqueues ab85b68e630e8a4db0271165149589d7702c2532 +sched-ext c43ea8a85f652163dd0c542aabf70cb339ba5813 +drivers-x86 cee9395acd8043be0644b25c34bfa86623f2b935 +chrome-platform c55749415623a58eba3eb527da023d29b43fbf5d +chrome-platform-firmware 44e33a5aaa2de3f5cf2701d6f4b0a7c63d4fe602 +hsi e81250ec6b69248b00d38c523dc6a13efaf38aab +leds-lj f8cca63a0a4f3475199d9ab7f86782bdfeb0744d +ipmi 89a312991dc6e638a36adc43ccb91dbc25504c04 +driver-core cee9395acd8043be0644b25c34bfa86623f2b935 +usb cee9395acd8043be0644b25c34bfa86623f2b935 +thunderbolt 48e989e33b715611438ce4b8d6ff712d4becd84f +usb-serial cee9395acd8043be0644b25c34bfa86623f2b935 +tty cee9395acd8043be0644b25c34bfa86623f2b935 +char-misc cee9395acd8043be0644b25c34bfa86623f2b935 +coresight 9e3604d7369cfc0110100eb1a0acab1865ee2d18 +fastrpc dc59e4fea9d83f03bad6bddf3fa2e52491777482 +fpga 4216e549a265701c88df660f42d74a7eb4819af2 +icc 9621c81fd2eb7db4b647814b7b6e537744831e60 +iio 183f05a300eab41e4578337eac59335730dfebf9 +nfc cee9395acd8043be0644b25c34bfa86623f2b935 +phy-next cee9395acd8043be0644b25c34bfa86623f2b935 +soundwire 1d4d3198ccce18ec1d58ac41c47e7b2aabba2e94 +extcon 8d3ae59288f1e7d58d76558a6ee96d533bc5019f +gnss cee9395acd8043be0644b25c34bfa86623f2b935 +vfio 4e3c1fc8abcb8eff062150b4340fa4569696d645 +w1 cee9395acd8043be0644b25c34bfa86623f2b935 +spmi 8cdeaa50eae8dad34885515f62559ee83e7e8dda +staging 758be99c46265b74476d5e1173d781abe325533e +counter-next 353b2e09f44a91e64c6b8117df46b6999c52e465 +mux ac7bde3c53166656d80e3aacc7d274d3c60a6de4 +dmaengine cee9395acd8043be0644b25c34bfa86623f2b935 +cgroup 75f8b845df69ad3502838dbaff680f968e261d14 +scsi af8c27375733fb6a6df9fa484cda77cc3dd0cb80 +scsi-mkp e83b47309f73313e75c3888d7839666aba5b2b2a +vhost ae864da7d762b4c692e2cdf83cad43110d5d46ee +rpmsg c0bdd460b8fd5c1d825c7ecc3ede2db409da461c +gpio-brgl a2cfc48fee416f0ea78e7dc38acbf2ce072574cc +gpio-intel cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl d067f0f4c96cb5402a111cf649b7bf9c5771ec74 +pinctrl-intel ed1608d5e3c4432c1dcba709668b90f44228c10c +pinctrl-renesas d48a308fabd6dc7676d570f0bc971817e39d824c +pinctrl-samsung cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-qcom d0344c6510d050ee79c6aa3997f2c641168ebf08 +pwm 6b0c6ff76795f62981352cc0be00163077d0961b +ktest 932cdaf3e273a2727e77af97f79f12577174c5a0 +kselftest cee9395acd8043be0644b25c34bfa86623f2b935 +kunit cee9395acd8043be0644b25c34bfa86623f2b935 +kunit-next e38f53f0482468efd04397f66bda4648b70ac9fa +livepatching 26260251022fbc2f248a3d747a9b2b961b18d2d8 +rtc afce9701d6423a63194a349d2f1e34c50ce76482 +nvdimm e99cb3ecd8334ca21e01ff9a79a916693f58f1fb +at24 dc59e4fea9d83f03bad6bddf3fa2e52491777482 +ntb dc59e4fea9d83f03bad6bddf3fa2e52491777482 +seccomp 41fa04327384148b0e2e828c9be9862c5240e9fa +slimbus c922423ce66bcee6c61d1b5aadea09b29ca81400 +nvmem dc59e4fea9d83f03bad6bddf3fa2e52491777482 +hyperv be0cfab740e58b70047ef6e7e3d578f00ed5d258 +auxdisplay c868291c2b3d74d5f1cae748a9452b340b21f1d1 +kgdb fdbdd0ccb30af18d3b29e714ac8d5ab6163279e0 +hmm cee9395acd8043be0644b25c34bfa86623f2b935 +cfi dc59e4fea9d83f03bad6bddf3fa2e52491777482 +mhi 83c29a55b89e0de6e15bcc6c21b16b4b75ec3e89 +cxl 274b30592196c9918af700e5edcb8bce8923213e +zstd 65d1f5507ed2c78c64fce40e44e5574a9419eb09 +efi 42bf9d266e52823c68e416040984edd0f6d62938 +unicode a511442085c140da9cdbe60e3f0fab1c71480801 +random 26f5abb98a9cde6825d3cdb12a690d0ea7ede5b4 +landlock 172b6a6d8463562b0cbebfd66f770b078f81966b +sysctl 8d75c338f0bcecaa6c9af67f86c176b67b6acf3e +execve ab11176bd3a76058ecafd066d0f9c718dc80f389 +bitmap cf72cbb39da84b6f02f90c07f33b102fc10b16f0 +hte 1329abe1bae416ab2cff89c3312d33f745d341c7 +kspp 1b501c6afc7a6d71a76d3f951bf63cb2d2b2e8f3 +nolibc 6ef847683e7d294a4187a7253924b8f0d8911416 +iommufd cee9395acd8043be0644b25c34bfa86623f2b935 +turbostat 1c996a37fd244d19e5dbb715328c1676e28ef607 +pwrseq 3b04e9b8056e868c3e9a04cc74168c7c9a18746a +capabilities-next 9b0ee13b80798feb0e6a83a6d69e915f23fcb6bc +ipe 028ef9c96e96197026887c0f092424679298aae8 +kcsan a8488ecbd7ba44d65b912dfe88a73f438eba2447 +crc cee9395acd8043be0644b25c34bfa86623f2b935 +keys-next 965e9a2cf23b066d8bdeb690dff9cd7089c5f667 +fwctl cee9395acd8043be0644b25c34bfa86623f2b935 +devsec-tsm 3177779ae17db4c66c851f799505fb95c7530c03 +hisilicon a5db65458a911daa6f8927264366dab501715dc0 +device-id 995832b2cebe6969d1b42635db698803ee31294d +kthread fa39ec4f89f2637ed1cdbcde3656825951787668 +pagemap-headers e02cb91d4644dfff593146f89e70d2008cc6ac16 diff --git a/Next/Trees b/Next/Trees new file mode 100644 index 00000000000000..7d0d5ebef20800 --- /dev/null +++ b/Next/Trees @@ -0,0 +1,428 @@ +Trees included into this release: + +Name Url +---- --- +origin https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git#master +ext4-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git#fixes +vfs-brauner-fixes https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git#vfs.fixes +fscrypt-current https://git.kernel.org/pub/scm/fs/fscrypt/linux.git#for-current +fsverity-current https://git.kernel.org/pub/scm/fs/fsverity/linux.git#for-current +btrfs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git#next-fixes +vfs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git#fixes +erofs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git#fixes +nfsd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux#nfsd-fixes +v9fs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git#fixes/next +overlayfs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git#ovl-fixes +fscrypt https://git.kernel.org/pub/scm/fs/fscrypt/linux.git#for-next +btrfs https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git#for-next +ceph https://github.com/ceph/ceph-client.git#master +cifs https://git.manguebit.org/linux.git#cifs-next +configfs https://git.kernel.org/pub/scm/linux/kernel/git/leitao/linux.git#configfs-next +ecryptfs https://git.kernel.org/pub/scm/linux/kernel/git/tyhicks/ecryptfs.git#next +dlm https://git.kernel.org/pub/scm/linux/kernel/git/teigland/linux-dlm.git#next +erofs https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git#dev +exfat https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/exfat.git#dev +ext3 https://git.kernel.org/pub/scm/linux/kernel/git/jack/linux-fs.git#for_next +ext4 https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git#dev +f2fs https://git.kernel.org/pub/scm/linux/kernel/git/jaegeuk/f2fs.git#dev +fsverity https://git.kernel.org/pub/scm/fs/fsverity/linux.git#for-next +fuse https://git.kernel.org/pub/scm/linux/kernel/git/mszeredi/fuse.git#for-next +gfs2 https://git.kernel.org/pub/scm/linux/kernel/git/gfs2/linux-gfs2.git#for-next +jfs https://github.com/kleikamp/linux-shaggy.git#jfs-next +ksmbd https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/smb.git#ksmbd-for-next +nfs git://git.linux-nfs.org/projects/trondmy/nfs-2.6.git#linux-next +nfs-anna git://git.linux-nfs.org/projects/anna/linux-nfs.git#linux-next +nfsd https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux#nfsd-next +ntfs https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/ntfs.git#ntfs-next +ntfs3 https://github.com/Paragon-Software-Group/linux-ntfs3.git#master +orangefs https://git.kernel.org/pub/scm/linux/kernel/git/hubcap/linux.git#for-next +overlayfs https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git#overlayfs-next +ubifs https://git.kernel.org/pub/scm/linux/kernel/git/rw/ubifs.git#next +v9fs https://github.com/martinetd/linux#9p-next +v9fs-ericvh https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git#ericvh/for-next +xfs https://git.kernel.org/pub/scm/fs/xfs/xfs-linux.git#for-next +zonefs https://git.kernel.org/pub/scm/linux/kernel/git/dlemoal/zonefs.git#for-next +vfs-brauner https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git#vfs.all +vfs https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git#for-next +mm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git#for-next-fixes +kbuild-current https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git#kbuild-fixes-for-next +clang-fixes https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git#clang-fixes-for-next +arc-current https://git.kernel.org/pub/scm/linux/kernel/git/vgupta/arc.git#for-curr +arm-current https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git#fixes +arm64-fixes https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux#for-next/fixes +arm-soc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git#arm/fixes +davinci-current https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#davinci/for-current +realtek-fixes https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git#fixes +drivers-memory-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git#fixes +sophgo-fixes https://github.com/sophgo/linux.git#fixes +sophgo-soc-fixes https://github.com/sophgo/linux.git#soc-fixes +m68k-current https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git#for-linus +powerpc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git#fixes +s390-fixes https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git#fixes +net https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net.git#main +bpf https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf.git/#master +ipsec https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec.git#master +netfilter https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf.git#main +ipvs https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs.git#main +wireless https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless.git#for-next +ath https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git#for-current +iwlwifi https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git#fixes +wpan https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan.git#master +rdma-fixes https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git#for-rc +sound-current https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git#for-linus +sound-asoc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git#for-linus +regmap-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git#for-linus +regulator-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git#for-linus +spi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git#for-linus +pci-current https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git#for-linus +driver-core.current https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git#driver-core-linus +tty.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git#tty-linus +usb.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git#usb-linus +usb-serial-fixes https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git#usb-linus +phy https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git#fixes +staging.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git#staging-linus +iio-fixes https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git#fixes-togreg +watchdog-fixes https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#watchdog +counter-current https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git#counter-current +char-misc.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git#char-misc-linus +soundwire-fixes https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git#fixes +thunderbolt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git#fixes +input-current https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git#for-linus +crypto-current https://git.kernel.org/pub/scm/linux/kernel/git/herbert/crypto-2.6.git#master +libcrypto-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git#libcrypto-fixes +vfio-fixes https://github.com/awilliam/linux-vfio.git#for-linus +kselftest-fixes https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#fixes +dmaengine-fixes https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git#fixes +backlight-fixes https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git#for-backlight-fixes +mtd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#mtd/fixes +mfd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git#for-mfd-fixes +v4l-dvb-fixes git://linuxtv.org/media-ci/media-pending.git#fixes +reset-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux#reset/fixes +mips-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git#mips-fixes +at91-fixes https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git#at91-fixes +omap-fixes https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git#fixes +tegra-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git#fixes +kvm-fixes git://git.kernel.org/pub/scm/virt/kvm/kvm.git#master +kvms390-fixes https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git#master +kvm-arm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git#fixes +hwmon-fixes https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#hwmon +nvdimm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git#libnvdimm-fixes +cxl-fixes https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git#fixes +dma-mapping-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git#dma-mapping-fixes +drivers-x86-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git#fixes +samsung-krzk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git#fixes +pinctrl-samsung-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git#fixes +pinctrl-qcom-fixes https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pinctrl-qcom/for-current +devicetree-fixes https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git#dt/linus +dt-krzk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git#fixes +scsi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git#fixes +drm-fixes https://gitlab.freedesktop.org/drm/kernel.git#drm-fixes +drm-intel-fixes https://gitlab.freedesktop.org/drm/i915/kernel.git#for-linux-next-fixes +mmc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git#fixes +rtc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git#rtc-fixes +gnss-fixes https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git#gnss-linus +hyperv-fixes https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git#hyperv-fixes +risc-v-fixes https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git#fixes +riscv-dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-dt-fixes +riscv-soc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-soc-fixes +fpga-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git#fixes +spdx https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/spdx.git#spdx-linus +gpio-brgl-fixes https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#gpio/for-current +gpio-intel-fixes https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git#fixes +pinctrl-intel-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git#fixes +auxdisplay-fixes https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git#fixes +kunit-fixes https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#kunit-fixes +renesas-fixes https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git#fixes +perf-current https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools.git#perf-tools +efi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git#urgent +battery-fixes https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git#fixes +iommufd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git#for-rc +rust-fixes https://github.com/Rust-for-Linux/linux.git#rust-fixes +w1-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git#fixes +pmdomain-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git#fixes +i2c-andi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git#i2c/i2c-fixes +i2c-rust-fixes https://github.com/ikrtn/rust-for-linux#rust-i2c-fixes +sparc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git#for-linus +clk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git#clk-fixes +thead-clk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-clk-fixes +tenstorrent-clk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-clk-fixes +fustini-config-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#riscv-config-fixes +pwrseq-fixes https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pwrseq/for-current +thead-dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-dt-fixes +ftrace-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#ftrace/fixes +ring-buffer-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#ring-buffer/fixes +trace-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#trace/fixes +tracefs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#tracefs/fixes +spacemit-fixes https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux#fixes +tip-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git#tip/urgent +kexec-fixes https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#kexec-fixes +liveupdate-fixes https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#fixes +drm-msm-fixes https://gitlab.freedesktop.org/drm/msm.git#msm-fixes +uml-fixes https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git#fixes +fwctl-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git#for-rc +devsec-tsm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git#fixes +drm-rust-fixes https://gitlab.freedesktop.org/drm/rust/kernel.git#for-linux-next-fixes +tenstorrent-dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-dt-fixes +nfc-fixes https://codeberg.org/linux-nfc/linux.git#for-linus +drm-misc-fixes https://gitlab.freedesktop.org/drm/misc/kernel.git#for-linux-next-fixes +rust https://github.com/Rust-for-Linux/linux.git#rust-next +rust-interop https://github.com/Rust-for-Linux/linux.git#interop-next +rust-alloc https://github.com/Rust-for-Linux/linux.git#alloc-next +rust-io https://github.com/Rust-for-Linux/linux.git#io-next +rust-pin-init https://github.com/Rust-for-Linux/linux.git#pin-init-next +rust-timekeeping https://github.com/Rust-for-Linux/linux.git#timekeeping-next +rust-xarray https://github.com/Rust-for-Linux/linux.git#xarray-next +rust-analyzer https://github.com/Rust-for-Linux/linux.git#rust-analyzer-next +mm https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git#for-next +mm-nonmm-stable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-stable +mm-nonmm-unstable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-unstable +kbuild https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git#kbuild-for-next +clang-format https://github.com/ojeda/linux.git#clang-format +perf https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools-next.git#perf-tools-next +compiler-attributes https://github.com/ojeda/linux.git#compiler-attributes +dma-mapping https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git#dma-mapping-for-next +asm-generic https://git.kernel.org/pub/scm/linux/kernel/git/arnd/asm-generic#master +alpha https://git.kernel.org/pub/scm/linux/kernel/git/mattst88/alpha.git#alpha-next +arm https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git#for-next +arm64 https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux#for-next/core +arm-perf https://git.kernel.org/pub/scm/linux/kernel/git/will/linux.git#for-next/perf +arm-soc https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git#for-next +amlogic https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux.git#for-next +asahi-soc https://github.com/AsahiLinux/linux.git#asahi-soc/for-next +at91 https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git#at91-next +bmc https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux.git#for-next +broadcom https://github.com/Broadcom/stblinux.git#next +cix https://git.kernel.org/pub/scm/linux/kernel/git/peter.chen/cix.git#for-next +davinci https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#davinci/for-next +drivers-memory https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git#for-next +fsl https://git.kernel.org/pub/scm/linux/kernel/git/chleroy/linux.git#soc_fsl +imx-mxs https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux.git#for-next +mediatek https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux.git#for-next +mvebu https://git.kernel.org/pub/scm/linux/kernel/git/gclement/mvebu.git#for-next +omap https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git#for-next +qcom https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux.git#for-next +realtek https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git#for-next +renesas https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git#next +reset https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux#reset/next +rockchip https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip.git#for-next +samsung-krzk https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git#for-next +scmi https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux.git#for-linux-next +sophgo https://github.com/sophgo/linux.git#for-next +sophgo-soc https://github.com/sophgo/linux.git#soc-for-next +spacemit https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux#for-next +stm32 https://git.kernel.org/pub/scm/linux/kernel/git/atorgue/stm32.git#stm32-next +sunxi https://git.kernel.org/pub/scm/linux/kernel/git/sunxi/linux.git#sunxi/for-next +tee https://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee.git#next +tegra https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git#for-next +tenstorrent-dt https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-dt-for-next +fustini-config https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#riscv-config-for-next +thead-dt https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-dt-for-next +ti https://git.kernel.org/pub/scm/linux/kernel/git/ti/linux.git#ti-next +xilinx https://github.com/Xilinx/linux-xlnx.git#for-next +socfpga https://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux.git#for-next +clk https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git#clk-next +clk-imx https://git.kernel.org/pub/scm/linux/kernel/git/abelvesa/linux.git#for-next +clk-renesas https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git#renesas-clk +thead-clk https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-clk-for-next +tenstorrent-clk https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-clk-for-next +csky https://github.com/c-sky/csky-linux.git#linux-next +loongarch https://git.kernel.org/pub/scm/linux/kernel/git/chenhuacai/linux-loongson.git#loongarch-next +m68k https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git#for-next +m68knommu https://git.kernel.org/pub/scm/linux/kernel/git/gerg/m68knommu.git#for-next +microblaze git://git.monstr.eu/linux-2.6-microblaze.git#next +mips https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git#mips-next +openrisc https://github.com/openrisc/linux.git#for-next +parisc-hd https://git.kernel.org/pub/scm/linux/kernel/git/deller/parisc-linux.git#for-next +powerpc https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git#next +risc-v https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git#for-next +riscv-dt https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-dt-for-next +riscv-soc https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-soc-for-next +s390 https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git#for-next +sh https://git.kernel.org/pub/scm/linux/kernel/git/glaubitz/sh-linux.git#for-next +sparc https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git#for-next +uml https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git#next +xtensa https://github.com/jcmvbkbc/linux-xtensa.git#xtensa-for-next +printk https://git.kernel.org/pub/scm/linux/kernel/git/printk/linux.git#for-next +pci https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git#next +pstore https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/pstore +hid https://git.kernel.org/pub/scm/linux/kernel/git/hid/hid.git#for-next +i2c https://git.kernel.org/pub/scm/linux/kernel/git/wsa/linux.git#i2c/for-next +i2c-andi https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git#i2c/i2c-next +i2c-rust https://github.com/ikrtn/rust-for-linux#rust-i2c-next +i3c https://git.kernel.org/pub/scm/linux/kernel/git/i3c/linux.git#i3c/next +dmi https://git.kernel.org/pub/scm/linux/kernel/git/jdelvare/staging.git#dmi-for-next +hwmon-staging https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#hwmon-next +jc_docs git://git.lwn.net/linux.git#docs-next +v4l-dvb git://linuxtv.org/media-ci/media-pending.git#next +v4l-dvb-next git://linuxtv.org/mchehab/media-next.git#master +pm https://git.kernel.org/pub/scm/linux/kernel/git/rafael/linux-pm.git#linux-next +cpufreq-arm https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git#cpufreq/arm/linux-next +cpupower https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux.git#cpupower +devfreq https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/linux.git#devfreq-next +pmdomain https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git#next +opp https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git#opp/linux-next +thermal https://git.kernel.org/pub/scm/linux/kernel/git/thermal/linux.git#thermal/linux-next +rdma https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git#for-next +net-next https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next.git#main +bpf-next https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git#for-next +ipsec-next https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec-next.git#master +mlx5-next https://git.kernel.org/pub/scm/linux/kernel/git/mellanox/linux.git#mlx5-next +netfilter-next https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf-next.git#main +ipvs-next https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs-next.git#main +bluetooth https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth-next.git#master +wireless-next https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless-next.git#for-next +ath-next https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git#for-next +iwlwifi-next https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git#next +wpan-next https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git#master +wpan-staging https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git#staging +mtd https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#mtd/next +nand https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#nand/next +spi-nor https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#spi-nor/next +crypto https://git.kernel.org/pub/scm/linux/kernel/git/herbert/cryptodev-2.6.git#master +libcrypto https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git#libcrypto-next +drm https://gitlab.freedesktop.org/drm/kernel.git#drm-next +drm-exynos https://git.kernel.org/pub/scm/linux/kernel/git/daeinki/drm-exynos.git#for-linux-next +drm-misc https://gitlab.freedesktop.org/drm/misc/kernel.git#for-linux-next +amdgpu https://gitlab.freedesktop.org/agd5f/linux.git#drm-next +drm-intel https://gitlab.freedesktop.org/drm/i915/kernel.git#for-linux-next +drm-msm https://gitlab.freedesktop.org/drm/msm.git#msm-next +drm-msm-lumag https://gitlab.freedesktop.org/lumag/msm.git#msm-next-lumag +drm-xe https://gitlab.freedesktop.org/drm/xe/kernel.git#drm-xe-next +drm-rust https://gitlab.freedesktop.org/drm/rust/kernel.git#for-linux-next +drm-nova https://gitlab.freedesktop.org/drm/nova.git#nova-next +etnaviv https://git.pengutronix.de/git/lst/linux#etnaviv/next +fbdev https://git.kernel.org/pub/scm/linux/kernel/git/deller/linux-fbdev.git#for-next +regmap https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git#for-next +sound https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git#for-next +ieee1394 https://git.kernel.org/pub/scm/linux/kernel/git/ieee1394/linux1394.git#for-next +sound-asoc https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git#for-next +modules https://git.kernel.org/pub/scm/linux/kernel/git/modules/linux.git#modules-next +input https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git#next +block https://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux.git#for-next +device-mapper https://git.kernel.org/pub/scm/linux/kernel/git/device-mapper/linux-dm.git#for-next +libata https://git.kernel.org/pub/scm/linux/kernel/git/libata/linux#for-next +pcmcia https://git.kernel.org/pub/scm/linux/kernel/git/brodo/linux.git#pcmcia-next +mmc https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git#next +mfd https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git#for-mfd-next +backlight https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git#for-backlight-next +battery https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git#for-next +regulator https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git#for-next +security https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git#next +apparmor https://git.kernel.org/pub/scm/linux/kernel/git/jj/linux-apparmor#apparmor-next +integrity https://git.kernel.org/pub/scm/linux/kernel/git/zohar/linux-integrity#next-integrity +selinux https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/selinux.git#next +smack https://github.com/cschaufler/smack-next#next +tomoyo git://git.code.sf.net/p/tomoyo/tomoyo.git#master +tpmdd-tpm https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git#for-next-tpm +tpmdd-keys https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git#for-next-keys +watchdog https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#watchdog-next +iommu https://git.kernel.org/pub/scm/linux/kernel/git/iommu/linux.git#next +audit https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/audit.git#next +devicetree https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git#for-next +dt-krzk https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git#for-next +mailbox https://git.kernel.org/pub/scm/linux/kernel/git/jassibrar/mailbox.git#for-next +spi https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git#for-next +tip https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git#master +kexec https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#kexec-next +liveupdate https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#next +clockevents https://git.kernel.org/pub/scm/linux/kernel/git/daniel.lezcano/linux.git#timers/drivers/next +edac https://git.kernel.org/pub/scm/linux/kernel/git/ras/ras.git#edac-for-next +ftrace https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#for-next +rcu https://git.kernel.org/pub/scm/linux/kernel/git/rcu/linux#next +paulmck https://git.kernel.org/pub/scm/linux/kernel/git/paulmck/linux-rcu.git#non-rcu/next +kvm git://git.kernel.org/pub/scm/virt/kvm/kvm.git#next +kvm-arm https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git#next +kvms390 https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git#next +kvm-ppc https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git#topic/ppc-kvm +kvm-riscv https://github.com/kvm-riscv/linux.git#riscv_kvm_next +kvm-x86 https://github.com/kvm-x86/linux.git#next +xen-tip https://git.kernel.org/pub/scm/linux/kernel/git/xen/tip.git#linux-next +percpu https://git.kernel.org/pub/scm/linux/kernel/git/dennis/percpu.git#for-next +workqueues https://git.kernel.org/pub/scm/linux/kernel/git/tj/wq.git#for-next +sched-ext https://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext.git#for-next +drivers-x86 https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git#for-next +chrome-platform https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git#for-next +chrome-platform-firmware https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git#for-firmware-next +hsi https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-hsi.git#for-next +leds-lj https://git.kernel.org/pub/scm/linux/kernel/git/lee/leds.git#for-leds-next +ipmi https://github.com/cminyard/linux-ipmi.git#for-next +driver-core https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git#driver-core-next +usb https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git#usb-next +thunderbolt https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git#next +usb-serial https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git#usb-next +tty https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git#tty-next +char-misc https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git#char-misc-next +coresight https://git.kernel.org/pub/scm/linux/kernel/git/coresight/linux.git#next +fastrpc https://git.kernel.org/pub/scm/linux/kernel/git/srini/fastrpc.git#for-next +fpga https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git#for-next +icc https://git.kernel.org/pub/scm/linux/kernel/git/djakov/icc.git#icc-next +iio https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git#togreg +nfc https://codeberg.org/linux-nfc/linux.git#for-next +phy-next https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git#next +soundwire https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git#next +extcon https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/extcon.git#extcon-next +gnss https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git#gnss-next +vfio https://github.com/awilliam/linux-vfio.git#next +w1 https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git#for-next +spmi https://git.kernel.org/pub/scm/linux/kernel/git/sboyd/spmi.git#spmi-next +staging https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git#staging-next +counter-next https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git#counter-next +mux https://gitlab.com/peda-linux/mux.git#for-next +dmaengine https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git#next +cgroup https://git.kernel.org/pub/scm/linux/kernel/git/tj/cgroup.git#for-next +scsi https://git.kernel.org/pub/scm/linux/kernel/git/jejb/scsi.git#for-next +scsi-mkp https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git#for-next +vhost https://git.kernel.org/pub/scm/linux/kernel/git/mst/vhost.git#linux-next +rpmsg https://git.kernel.org/pub/scm/linux/kernel/git/remoteproc/linux.git#for-next +gpio-brgl https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#gpio/for-next +gpio-intel https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git#for-next +pinctrl https://git.kernel.org/pub/scm/linux/kernel/git/linusw/linux-pinctrl.git#for-next +pinctrl-intel https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git#for-next +pinctrl-renesas https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git#renesas-pinctrl +pinctrl-samsung https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git#for-next +pinctrl-qcom https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pinctrl-qcom/for-next +pwm https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git#pwm/for-next +ktest https://git.kernel.org/pub/scm/linux/kernel/git/rostedt/linux-ktest.git#for-next +kselftest https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#next +kunit https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#test +kunit-next https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#kunit +livepatching https://git.kernel.org/pub/scm/linux/kernel/git/livepatching/livepatching.git#for-next +rtc https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git#rtc-next +nvdimm https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git#libnvdimm-for-next +at24 https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#at24/for-next +ntb https://github.com/jonmason/ntb.git#ntb-next +seccomp https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/seccomp +slimbus https://git.kernel.org/pub/scm/linux/kernel/git/srini/slimbus.git#for-next +nvmem https://git.kernel.org/pub/scm/linux/kernel/git/srini/nvmem.git#for-next +hyperv https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git#hyperv-next +auxdisplay https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git#for-next +kgdb https://git.kernel.org/pub/scm/linux/kernel/git/danielt/linux.git#kgdb/for-next +hmm https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git#hmm +cfi https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#cfi/next +mhi https://git.kernel.org/pub/scm/linux/kernel/git/mani/mhi.git#mhi-next +cxl https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git#next +zstd https://github.com/terrelln/linux.git#zstd-next +efi https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git#next +unicode https://git.kernel.org/pub/scm/linux/kernel/git/krisman/unicode.git#for-next +random https://git.kernel.org/pub/scm/linux/kernel/git/crng/random.git#master +landlock https://git.kernel.org/pub/scm/linux/kernel/git/mic/linux.git#next +sysctl https://git.kernel.org/pub/scm/linux/kernel/git/sysctl/sysctl.git#sysctl-next +execve https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/execve +bitmap https://github.com/norov/linux.git#bitmap-for-next +hte https://git.kernel.org/pub/scm/linux/kernel/git/pateldipen1984/linux.git#for-next +kspp https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/kspp +nolibc https://git.kernel.org/pub/scm/linux/kernel/git/nolibc/linux-nolibc.git#for-next +iommufd https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git#for-next +turbostat https://git.kernel.org/pub/scm/linux/kernel/git/lenb/linux.git#next +pwrseq https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pwrseq/for-next +capabilities-next https://git.kernel.org/pub/scm/linux/kernel/git/sergeh/linux.git#caps-next +ipe https://git.kernel.org/pub/scm/linux/kernel/git/wufan/ipe.git#next +kcsan https://git.kernel.org/pub/scm/linux/kernel/git/melver/linux.git#next +crc https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git#crc-next +keys-next https://git.kernel.org/pub/scm/linux/kernel/git/dhowells/linux-fs.git#keys-next +fwctl https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git#for-next +devsec-tsm https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git#next +hisilicon https://github.com/hisilicon/linux-hisi.git#for-next +device-id https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git#device-id-rework +kthread https://git.kernel.org/pub/scm/linux/kernel/git/frederic/linux-dynticks.git#for-next +pagemap-headers git://git.infradead.org/users/willy/pagecache.git#headers diff --git a/Next/merge.log b/Next/merge.log new file mode 100644 index 00000000000000..46192fdfc4c7b0 --- /dev/null +++ b/Next/merge.log @@ -0,0 +1,6264 @@ +$ date -R +Thu, 03 Sep 2026 12:45:28 +0100 +$ git checkout master +Already on 'master' +$ git reset --hard stable +HEAD is now at 89a312991dc6e Merge tag 'cifs-fixes-7.3-rc2' of https://git.manguebit.org/linux +Merging origin/master (940de590b839f Merge tag 'hardening-v7.3-rc2' of git://git.kernel.org/pub/scm/linux/kernel/git/kees/linux) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git origin/master +Updating 89a312991dc6e..940de590b839f +Fast-forward (no commit created; -m option ignored) + security/Kconfig.hardening | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) +Merging ext4-fixes/fixes (981fcc5674e67 jbd2: fix deadlock in jbd2_journal_cancel_revoke()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git ext4-fixes/fixes +Already up to date. +Merging vfs-brauner-fixes/vfs.fixes (e14d4302cbd0d Merge patch series "afs: Miscellaneous fixes") +$ git merge -m Merge branch 'vfs.fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git vfs-brauner-fixes/vfs.fixes +Auto-merging include/linux/sched.h +Auto-merging init/main.c +Auto-merging kernel/signal.c +Merge made by the 'ort' strategy. + fs/adfs/super.c | 24 +++--- + fs/afs/addr_list.c | 5 +- + fs/afs/dir_edit.c | 9 +- + fs/afs/dir_search.c | 11 +-- + fs/afs/fs_probe.c | 1 + + fs/afs/internal.h | 8 ++ + fs/afs/server.c | 1 - + fs/cachefiles/xattr.c | 16 ++-- + fs/ext4/inode.c | 11 ++- + fs/netfs/buffered_read.c | 164 +++++++++++++++++++++++++++---------- + fs/netfs/direct_write.c | 31 ++++--- + fs/netfs/internal.h | 3 + + fs/netfs/misc.c | 19 +++++ + fs/netfs/objects.c | 32 +++++--- + fs/netfs/read_collect.c | 128 ++++++++++++++++++++--------- + fs/netfs/read_pgpriv2.c | 15 ++-- + fs/netfs/read_retry.c | 13 ++- + fs/netfs/read_single.c | 2 + + fs/netfs/rolling_buffer.c | 79 +++++++++++------- + fs/netfs/write_issue.c | 2 + + fs/overlayfs/super.c | 2 +- + fs/super.c | 9 +- + fs/ufs/cylinder.c | 10 +++ + fs/ufs/dir.c | 2 +- + fs/ufs/super.c | 17 ++-- + include/linux/netfs.h | 5 +- + include/linux/ns/ns_common_types.h | 6 +- + include/linux/rolling_buffer.h | 6 +- + include/linux/sched.h | 2 +- + include/linux/sched/signal.h | 5 +- + include/trace/events/cachefiles.h | 19 ++++- + include/trace/events/netfs.h | 30 ++++++- + init/main.c | 2 +- + kernel/reboot.c | 19 ++++- + kernel/signal.c | 12 +++ + 35 files changed, 506 insertions(+), 214 deletions(-) +Merging fscrypt-current/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-current' of https://git.kernel.org/pub/scm/fs/fscrypt/linux.git fscrypt-current/for-current +Already up to date. +Merging fsverity-current/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-current' of https://git.kernel.org/pub/scm/fs/fsverity/linux.git fsverity-current/for-current +Already up to date. +Merging btrfs-fixes/next-fixes (4d1d66ba287be Merge branch 'misc-7.3' into next-fixes) +$ git merge -m Merge branch 'next-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git btrfs-fixes/next-fixes +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 2 +- + fs/btrfs/dev-replace.c | 2 +- + fs/btrfs/inode.c | 3 +++ + fs/btrfs/ioctl.c | 21 ++++++++++++++++++--- + fs/btrfs/raid-stripe-tree.c | 25 ++++++++++++++++--------- + fs/btrfs/raid-stripe-tree.h | 1 + + fs/btrfs/scrub.c | 24 ++++++++++++++---------- + fs/btrfs/send.c | 9 ++++++++- + fs/btrfs/tests/extent-io-tests.c | 5 +++-- + fs/btrfs/transaction.c | 19 ++++++++++++++++++- + fs/btrfs/volumes.c | 4 ++++ + fs/btrfs/zoned.c | 16 +++++++--------- + fs/btrfs/zstd.c | 11 ++++++++++- + 13 files changed, 104 insertions(+), 38 deletions(-) +Merging vfs-fixes/fixes (49c5d168a3a8f udf: fix nls leak on udf_fill_super() failure) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git vfs-fixes/fixes +Auto-merging fs/udf/super.c +Merge made by the 'ort' strategy. +Merging erofs-fixes/fixes (617d0d8d199ba erofs: preserve LZMA decoders on resize failure) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git erofs-fixes/fixes +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-fs-erofs | 2 +- + fs/erofs/decompressor_lzma.c | 18 ++++++++++++++---- + fs/erofs/sysfs.c | 2 ++ + 3 files changed, 17 insertions(+), 5 deletions(-) +Merging nfsd-fixes/nfsd-fixes (46db3c8a1be96 nfsd: export NFSv4 callback op stats via netlink) +$ git merge -m Merge branch 'nfsd-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux nfsd-fixes/nfsd-fixes +Already up to date. +Merging v9fs-fixes/fixes/next (028ef9c96e961 Linux 7.0) +$ git merge -m Merge branch 'fixes/next' of https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git v9fs-fixes/fixes/next +Already up to date. +Merging overlayfs-fixes/ovl-fixes (4549871118cf6 Linux 7.1-rc7) +$ git merge -m Merge branch 'ovl-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git overlayfs-fixes/ovl-fixes +Already up to date. +Merging fscrypt/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/fs/fscrypt/linux.git fscrypt/for-next +Already up to date. +Merging btrfs/for-next (966bb86e7420c Merge branch 'for-next-next-v7.3-20260902' into for-next-20260902) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git btrfs/for-next +Merge made by the 'ort' strategy. + fs/btrfs/Kconfig | 1 + + fs/btrfs/bio.c | 141 ++++++++++++++++++---------------------- + fs/btrfs/bio.h | 5 +- + fs/btrfs/block-group.c | 20 +++--- + fs/btrfs/btrfs_inode.h | 12 ++-- + fs/btrfs/dev-replace.c | 10 ++- + fs/btrfs/disk-io.c | 27 ++++++-- + fs/btrfs/extent-tree.c | 10 +++ + fs/btrfs/extent_io.c | 64 +++++++++++------- + fs/btrfs/extent_map.c | 8 +++ + fs/btrfs/file-item.c | 19 ++---- + fs/btrfs/fs.h | 2 +- + fs/btrfs/inode.c | 114 +++++++++++--------------------- + fs/btrfs/ioctl.c | 2 +- + fs/btrfs/qgroup.c | 126 ++++++++++++++++++++--------------- + fs/btrfs/qgroup.h | 16 ++++- + fs/btrfs/raid56.c | 46 +++++++------ + fs/btrfs/sysfs.c | 129 ++---------------------------------- + fs/btrfs/tree-checker.c | 66 +++++++++++++++---- + fs/btrfs/verity.c | 31 ++++----- + fs/btrfs/volumes.c | 39 ++++++++++- + fs/btrfs/volumes.h | 1 + + fs/btrfs/zoned.c | 5 ++ + include/uapi/linux/btrfs_tree.h | 21 +++--- + 24 files changed, 460 insertions(+), 455 deletions(-) +Merging ceph/master (dc173b37415e8 ceph: apply nearfull_sync option on remount) +$ git merge -m Merge branch 'master' of https://github.com/ceph/ceph-client.git ceph/master +Merge made by the 'ort' strategy. + fs/ceph/mds_client.c | 4 ++++ + fs/ceph/mds_client.h | 1 + + fs/ceph/super.c | 5 +++++ + net/ceph/messenger.c | 1 - + 4 files changed, 10 insertions(+), 1 deletion(-) +Merging cifs/cifs-next (4ee5025d18677 smb: client: reject userspace cifs.idmap descriptions) +$ git merge -m Merge branch 'cifs-next' of https://git.manguebit.org/linux.git cifs/cifs-next +Merge made by the 'ort' strategy. + fs/smb/client/cifsacl.c | 15 +++++++++++++++ + fs/smb/client/cifssmb.c | 25 ++++++++++++++++++++----- + fs/smb/client/trace.h | 1 + + 3 files changed, 36 insertions(+), 5 deletions(-) +Merging configfs/configfs-next (2251d0ed97c24 configfs: unhash the dentry before dropping the item in rmdir) +$ git merge -m Merge branch 'configfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/leitao/linux.git configfs/configfs-next +Merge made by the 'ort' strategy. + fs/configfs/dir.c | 9 +++++++++ + fs/configfs/symlink.c | 24 ++++++++++++++++++++---- + 2 files changed, 29 insertions(+), 4 deletions(-) +Merging ecryptfs/next (f81cb44f9a4b8 ecryptfs: ecryptfs_kernel.h: clean up kernel-doc comments) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/tyhicks/ecryptfs.git ecryptfs/next +Already up to date. +Merging dlm/next (ed9b6a1296f10 dlm: wait for outstanding SRCU callbacks to complete in exit paths) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/teigland/linux-dlm.git dlm/next +Merge made by the 'ort' strategy. + fs/dlm/config.c | 30 ++++++++++++++++++++++++++++-- + fs/dlm/dlm_internal.h | 5 +++++ + fs/dlm/lock.c | 18 +++++++++++++----- + fs/dlm/lock.h | 4 ++-- + fs/dlm/lowcomms.c | 1 + + fs/dlm/midcomms.c | 1 + + fs/dlm/plock.c | 14 +++++++++++++- + fs/dlm/user.c | 23 +++++++++++++++++++++++ + 8 files changed, 86 insertions(+), 10 deletions(-) +Merging erofs/dev (a7d28aa0e9b2c erofs: simplify z_erofs_gbuf_growsize()) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git erofs/dev +Already up to date. +Merging exfat/dev (f096faa6bb169 exfat: doc: add documentation) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/exfat.git exfat/dev +Merge made by the 'ort' strategy. + Documentation/filesystems/exfat.rst | 117 ++++++++++++++++++++++++++++++++++++ + Documentation/filesystems/index.rst | 1 + + fs/exfat/dir.c | 49 ++++++++++++++- + fs/exfat/iomap.c | 28 ++++++++- + fs/exfat/namei.c | 3 +- + 5 files changed, 193 insertions(+), 5 deletions(-) + create mode 100644 Documentation/filesystems/exfat.rst +Merging ext3/for_next (a3cf61f60051c ext2: enable context analysis support for ext2 filesystem) +$ git merge -m Merge branch 'for_next' of https://git.kernel.org/pub/scm/linux/kernel/git/jack/linux-fs.git ext3/for_next +Merge made by the 'ort' strategy. + fs/ext2/Makefile | 2 ++ + fs/ext2/balloc.c | 4 ++++ + fs/ext2/ext2.h | 19 +++++++++++-------- + fs/ext2/inode.c | 7 +++++++ + fs/ext2/super.c | 33 +++++++++++++++++++-------------- + fs/udf/inode.c | 52 +++++++++++++++++++++++++++++++++++++++++++++++----- + fs/udf/misc.c | 15 ++++++++------- + fs/udf/super.c | 5 ++++- + 8 files changed, 102 insertions(+), 35 deletions(-) +Merging ext4/dev (9091c97be3408 ext4: fix estimate extent index blocks in ext4_ext_index_trans_blocks()) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git ext4/dev +Already up to date. +Merging f2fs/dev (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/jaegeuk/f2fs.git f2fs/dev +Already up to date. +Merging fsverity/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/fs/fsverity/linux.git fsverity/for-next +Already up to date. +Merging fuse/for-next (10bd6b3296526 fuse: fall back to copy_splice_read() on FOPEN_DIRECT_IO) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mszeredi/fuse.git fuse/for-next +Merge made by the 'ort' strategy. + fs/fuse/dax.c | 1 + + fs/fuse/file.c | 7 ++++-- + fs/fuse/fuse_i.h | 9 +++++++ + fs/fuse/inode.c | 28 ++++++++++++++++++++++ + fs/fuse/ioctl.c | 3 --- + fs/fuse/req.c | 7 +++--- + include/uapi/linux/fuse.h | 12 +++++++++- + .../testing/selftests/filesystems/fuse/.gitignore | 1 + + tools/testing/selftests/filesystems/fuse/Makefile | 2 +- + 9 files changed, 59 insertions(+), 11 deletions(-) +Merging gfs2/for-next (ccaa1524a18aa gfs2: Fix journaled truncate range end offset) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gfs2/linux-gfs2.git gfs2/for-next +Merge made by the 'ort' strategy. + fs/gfs2/acl.c | 11 +++-- + fs/gfs2/aops.c | 17 +++---- + fs/gfs2/bmap.c | 91 ++++++++++++++++++++---------------- + fs/gfs2/dentry.c | 6 +-- + fs/gfs2/dir.c | 61 +++++++++++++++--------- + fs/gfs2/export.c | 7 +-- + fs/gfs2/file.c | 72 +++++++++++++++------------- + fs/gfs2/glock.c | 60 ++++++++++++++++-------- + fs/gfs2/glops.c | 13 ++++-- + fs/gfs2/incore.h | 8 ++-- + fs/gfs2/inode.c | 122 ++++++++++++++++++++++++++---------------------- + fs/gfs2/log.c | 3 +- + fs/gfs2/lops.c | 18 ++++--- + fs/gfs2/meta_io.c | 7 +-- + fs/gfs2/ops_fstype.c | 26 +++++------ + fs/gfs2/quota.c | 54 +++++++++++---------- + fs/gfs2/recovery.c | 17 ++++--- + fs/gfs2/rgrp.c | 129 +++++++++++++++++++++++++++------------------------ + fs/gfs2/super.c | 98 +++++++++++++++++++++++--------------- + fs/gfs2/trace_gfs2.h | 6 +-- + fs/gfs2/util.c | 8 ++-- + fs/gfs2/xattr.c | 69 ++++++++++++++++----------- + 22 files changed, 516 insertions(+), 387 deletions(-) +Merging jfs/jfs-next (dad98c5b2a05e jfs: avoid -Wtautological-constant-out-of-range-compare warning again) +$ git merge -m Merge branch 'jfs-next' of https://github.com/kleikamp/linux-shaggy.git jfs/jfs-next +Already up to date. +Merging ksmbd/ksmbd-for-next (da6066cf54a9c ksmbd: doc: update feature status) +$ git merge -m Merge branch 'ksmbd-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/smb.git ksmbd/ksmbd-for-next +Merge made by the 'ort' strategy. + Documentation/filesystems/smb/ksmbd.rst | 14 +- + fs/smb/server/Kconfig | 2 + + fs/smb/server/Makefile | 1 + + fs/smb/server/ksmbd_work.c | 3 - + fs/smb/server/ksmbd_work.h | 7 +- + fs/smb/server/mgmt/tree_connect.c | 8 + + fs/smb/server/oplock.c | 73 ++++-- + fs/smb/server/smb2pdu.c | 412 +++++++++----------------------- + fs/smb/server/smbacl.c | 2 + + fs/smb/server/tests/Kconfig | 15 ++ + fs/smb/server/tests/Makefile | 4 + + fs/smb/server/tests/smbacl_kunit.c | 301 +++++++++++++++++++++++ + fs/smb/server/vfs.c | 14 +- + fs/smb/server/vfs_cache.c | 48 ---- + fs/smb/server/vfs_cache.h | 6 - + 15 files changed, 525 insertions(+), 385 deletions(-) + create mode 100644 fs/smb/server/tests/Kconfig + create mode 100644 fs/smb/server/tests/Makefile + create mode 100644 fs/smb/server/tests/smbacl_kunit.c +$ git am -3 ../patches/0001-ntfs3-Fix-up-merge-with-Linus.patch +Applying: ntfs3: Fix up merge with Linus +Using index info to reconstruct a base tree... +M fs/ntfs3/file.c +Falling back to patching base and 3-way merge... +Auto-merging fs/ntfs3/file.c +No changes -- Patch already applied. +Merging nfs/linux-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'linux-next' of git://git.linux-nfs.org/projects/trondmy/nfs-2.6.git nfs/linux-next +Already up to date. +Merging nfs-anna/linux-next (e053b624f5d36 NFSv4.2: fix nfs4_listxattr size accounting) +$ git merge -m Merge branch 'linux-next' of git://git.linux-nfs.org/projects/anna/linux-nfs.git nfs-anna/linux-next +Already up to date. +Merging nfsd/nfsd-next (f5dc2038906bb NFSD: Point contributors and sashiko.dev to the nfsd-testing branch) +$ git merge -m Merge branch 'nfsd-next' of https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux nfsd/nfsd-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 3 +- + fs/lockd/svc.c | 1 - + fs/lockd/trace.h | 1 - + fs/lockd/xdr.h | 2 +- + fs/namei.c | 2 + + fs/nfs/nfs4file.c | 1 + + fs/nfs/super.c | 25 --- + fs/nfs_common/nfs_ssc.c | 126 +++++++++------ + fs/nfsd/blocklayout.c | 1 + + fs/nfsd/blocklayoutxdr.c | 11 ++ + fs/nfsd/export.c | 8 +- + fs/nfsd/export.h | 3 +- + fs/nfsd/filecache.c | 1 + + fs/nfsd/flexfilelayout.c | 24 +-- + fs/nfsd/flexfilelayoutxdr.c | 9 +- + fs/nfsd/flexfilelayoutxdr.h | 8 +- + fs/nfsd/localio.c | 10 +- + fs/nfsd/lockd.c | 4 +- + fs/nfsd/nfs2acl.c | 1 + + fs/nfsd/nfs3acl.c | 1 + + fs/nfsd/nfs3proc.c | 37 ++++- + fs/nfsd/nfs3xdr.c | 3 + + fs/nfsd/nfs4acl.c | 1 + + fs/nfsd/nfs4callback.c | 8 +- + fs/nfsd/nfs4ctl.h | 83 ++++++++++ + fs/nfsd/nfs4idmap.c | 1 + + fs/nfsd/nfs4layouts.c | 1 + + fs/nfsd/nfs4proc.c | 362 ++++++++++++++++++++++++-------------------- + fs/nfsd/nfs4recover.c | 1 + + fs/nfsd/nfs4state.c | 244 ++++++++++++++++++++++++----- + fs/nfsd/nfs4xdr.c | 104 +++++++++++-- + fs/nfsd/nfscache.c | 1 + + fs/nfsd/nfsctl.c | 2 + + fs/nfsd/nfsd.h | 236 +---------------------------- + fs/nfsd/nfserr.h | 158 +++++++++++++++++++ + fs/nfsd/nfsfh.c | 72 ++++----- + fs/nfsd/nfsfh.h | 28 +++- + fs/nfsd/nfsproc.c | 20 ++- + fs/nfsd/nfssvc.c | 8 + + fs/nfsd/nfsxdr.c | 1 + + fs/nfsd/state.h | 43 +++++- + fs/nfsd/vfs.c | 91 ++++++----- + fs/nfsd/vfs.h | 10 +- + fs/nfsd/xdr3.h | 2 +- + fs/nfsd/xdr4.h | 152 ++----------------- + fs/nfsd/xdr4cb.h | 20 +-- + include/linux/nfs.h | 55 +------ + include/linux/nfs3.h | 8 + + include/linux/nfs4.h | 6 + + include/linux/nfs_fh.h | 63 ++++++++ + include/linux/nfs_ssc.h | 69 ++------- + include/linux/nfsd_ssc.h | 38 +++++ + include/linux/nfslocalio.h | 11 +- + include/trace/misc/nfs.h | 1 + + 54 files changed, 1276 insertions(+), 906 deletions(-) + create mode 100644 fs/nfsd/nfs4ctl.h + create mode 100644 fs/nfsd/nfserr.h + create mode 100644 include/linux/nfs_fh.h + create mode 100644 include/linux/nfsd_ssc.h +$ git am -3 ../patches/0001-Revert-smb-client-implement-fileattr_get-to-support-.patch +Applying: Revert "smb: client: implement fileattr_get to support FS_IOC_GETFLAGS" +Using index info to reconstruct a base tree... +M fs/smb/client/cifsfs.c +M fs/smb/client/cifsfs.h +M fs/smb/client/inode.c +Falling back to patching base and 3-way merge... +Auto-merging fs/smb/client/inode.c +Auto-merging fs/smb/client/cifsfs.h +Auto-merging fs/smb/client/cifsfs.c +No changes -- Patch already applied. +Merging ntfs/ntfs-next (0fecc393f2060 ntfs: take invalidate_lock in ntfs_filemap_page_mkwrite()) +$ git merge -m Merge branch 'ntfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/ntfs.git ntfs/ntfs-next +Merge made by the 'ort' strategy. + fs/ntfs/attrib.c | 18 +++++--- + fs/ntfs/bdev-io.c | 2 +- + fs/ntfs/bitmap.c | 8 ++-- + fs/ntfs/compress.c | 2 +- + fs/ntfs/ea.c | 49 +++++++++++++++------ + fs/ntfs/file.c | 50 +++++++++++++-------- + fs/ntfs/inode.c | 10 +---- + fs/ntfs/lcnalloc.c | 9 ++-- + fs/ntfs/mft.c | 16 ++++--- + fs/ntfs/ntfs.h | 10 ++--- + fs/ntfs/reparse.c | 7 +-- + fs/ntfs/super.c | 15 ++++--- + fs/ntfs/wof.c | 127 +++++++++++++++++++++++++++++++++-------------------- + 13 files changed, 196 insertions(+), 127 deletions(-) +Merging ntfs3/master (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'master' of https://github.com/Paragon-Software-Group/linux-ntfs3.git ntfs3/master +Already up to date. +Merging orangefs/for-next (d410cd5303ec5 orangefs: skip leading spaces before parsing client debug masks) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/hubcap/linux.git orangefs/for-next +Already up to date. +Merging overlayfs/overlayfs-next (1f6ee9be92f8d ovl: make fsync after metadata copy-up opt-in mount option) +$ git merge -m Merge branch 'overlayfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git overlayfs/overlayfs-next +Already up to date. +Merging ubifs/next (a5e0055eac837 UBI: support per-device wear-leveling threshold) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/rw/ubifs.git ubifs/next +Already up to date. +Merging v9fs/9p-next (aa88278693cbf 9p: Add missing read barrier in virtio zero-copy path) +$ git merge -m Merge branch '9p-next' of https://github.com/martinetd/linux v9fs/9p-next +Already up to date. +Merging v9fs-ericvh/ericvh/for-next (028ef9c96e961 Linux 7.0) +$ git merge -m Merge branch 'ericvh/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git v9fs-ericvh/ericvh/for-next +Already up to date. +Merging xfs/for-next (c44f3db4f4c00 xfs: remove an extra cast in xfs_file_compat_ioctl) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/fs/xfs/xfs-linux.git xfs/for-next +Merge made by the 'ort' strategy. + fs/xfs/libxfs/xfs_da_btree.c | 1 + + fs/xfs/libxfs/xfs_defer.c | 18 +- + fs/xfs/libxfs/xfs_exchmaps.c | 10 ++ + fs/xfs/libxfs/xfs_parent.c | 12 +- + fs/xfs/libxfs/xfs_trans_space.c | 19 +- + fs/xfs/scrub/dir_repair.c | 27 ++- + fs/xfs/scrub/metapath.c | 87 ++++++++- + fs/xfs/scrub/symlink_repair.c | 2 +- + fs/xfs/xfs_buf.c | 4 +- + fs/xfs/xfs_extent_busy.c | 4 +- + fs/xfs/xfs_healthmon.c | 6 + + fs/xfs/xfs_icache.c | 3 +- + fs/xfs/xfs_ioctl.c | 381 ++++++++++++++++++++++++---------------- + fs/xfs/xfs_ioctl.h | 4 +- + fs/xfs/xfs_ioctl32.c | 188 +++++++++++--------- + fs/xfs/xfs_log.c | 33 ++-- + fs/xfs/xfs_log_cil.c | 3 +- + fs/xfs/xfs_log_priv.h | 4 +- + fs/xfs/xfs_mru_cache.c | 2 +- + fs/xfs/xfs_trans_ail.c | 4 +- + fs/xfs/xfs_verify_media.c | 24 ++- + fs/xfs/xfs_zone_alloc.c | 2 + + fs/xfs/xfs_zone_space_resv.c | 8 +- + 23 files changed, 561 insertions(+), 285 deletions(-) +Merging zonefs/for-next (3a8389d42bdf4 zonefs: handle integer overflow in zonefs_fname_to_fno) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dlemoal/zonefs.git zonefs/for-next +Already up to date. +Merging vfs-brauner/vfs.all (21860bb220225 Merge branch 'vfs-7.4.signal' into vfs.all) +$ git merge -m Merge branch 'vfs.all' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git vfs-brauner/vfs.all +Auto-merging fs/gfs2/inode.c +Auto-merging fs/gfs2/log.c +Auto-merging fs/gfs2/lops.c +Auto-merging fs/namei.c +Auto-merging fs/smb/client/transport.c +Merge made by the 'ort' strategy. + Documentation/admin-guide/binfmt-misc.rst | 3 + + Documentation/filesystems/proc.rst | 4 + + arch/powerpc/Kconfig | 1 - + arch/powerpc/include/asm/elf.h | 6 - + arch/powerpc/include/asm/spu.h | 3 - + arch/powerpc/platforms/cell/Kconfig | 1 - + arch/powerpc/platforms/cell/spu_syscalls.c | 20 - + arch/powerpc/platforms/cell/spufs/Makefile | 1 - + arch/powerpc/platforms/cell/spufs/coredump.c | 183 -- + arch/powerpc/platforms/cell/spufs/file.c | 114 -- + arch/powerpc/platforms/cell/spufs/spufs.h | 12 - + arch/powerpc/platforms/cell/spufs/syscalls.c | 4 - + drivers/base/devtmpfs.c | 108 +- + fs/9p/vfs_inode.c | 3 + + fs/9p/vfs_inode_dotl.c | 3 + + fs/adfs/dir.c | 2 +- + fs/aio.c | 11 +- + fs/binfmt_elf.c | 16 +- + fs/binfmt_elf_fdpic.c | 12 +- + fs/binfmt_misc.c | 12 +- + fs/bpf_fs_kfuncs.c | 4 +- + fs/buffer.c | 37 +- + fs/ceph/file.c | 3 + + fs/coredump.c | 438 +++-- + fs/exfat/misc.c | 2 +- + fs/ext2/xattr.c | 2 +- + fs/ext4/ext4_jbd2.c | 2 +- + fs/ext4/mmp.c | 2 +- + fs/fat/misc.c | 2 +- + fs/fs-writeback.c | 2 +- + fs/fuse/dir.c | 3 + + fs/gfs2/inode.c | 3 + + fs/gfs2/log.c | 4 +- + fs/gfs2/lops.c | 4 +- + fs/inode.c | 2 +- + fs/internal.h | 1 + + fs/iomap/buffered-io.c | 6 +- + fs/jbd2/commit.c | 22 +- + fs/jbd2/journal.c | 31 +- + fs/jbd2/transaction.c | 2 +- + fs/kernfs/dir.c | 76 +- + fs/kernfs/kernfs-internal.h | 9 +- + fs/namei.c | 227 ++- + fs/nfs/dir.c | 6 + + fs/ocfs2/buffer_head_io.c | 12 +- + fs/ocfs2/journal.c | 25 +- + fs/omfs/inode.c | 4 +- + fs/open.c | 59 +- + fs/smb/client/dir.c | 3 + + fs/smb/client/transport.c | 13 +- + fs/vboxsf/dir.c | 3 + + include/linux/binfmts.h | 3 +- + include/linux/buffer_head.h | 36 +- + include/linux/coredump.h | 37 +- + include/linux/fcntl.h | 6 + + include/linux/sched.h | 2 +- + include/linux/sched/signal.h | 27 +- + include/uapi/linux/coredump.h | 149 +- + ipc/mqueue.c | 7 + + kernel/pid_namespace.c | 3 +- + kernel/user_namespace.c | 3 - + tools/include/uapi/linux/coredump.h | 149 +- + tools/testing/selftests/Makefile | 1 + + tools/testing/selftests/coredump/Makefile | 7 +- + .../selftests/coredump/coredump_notify_signal.h | 29 + + .../coredump/coredump_notify_signal_helper.c | 46 + + .../coredump/coredump_notify_signal_test.c | 245 +++ + .../coredump/coredump_socket_protocol_test.c | 1983 +++++++++++++++----- + tools/testing/selftests/coredump/coredump_test.h | 32 +- + .../selftests/coredump/coredump_test_helpers.c | 1742 ++++++++++++++++- + .../selftests/coredump/coredump_test_helpers.h | 79 + + tools/testing/selftests/exec/Makefile | 4 + + tools/testing/selftests/exec/binfmt_misc_delim.c | 127 ++ + tools/testing/selftests/filesystems/.gitignore | 2 +- + tools/testing/selftests/filesystems/Makefile | 2 +- + .../selftests/filesystems/file_stressor/.gitignore | 2 + + .../selftests/filesystems/file_stressor/Makefile | 6 + + .../{ => file_stressor}/file_stressor.c | 0 + .../selftests/filesystems/file_stressor/settings | 3 + + .../selftests/filesystems/fscontext_ns/.gitignore | 2 + + tools/testing/selftests/filesystems/kernfs_test.c | 16 +- + .../selftests/filesystems/open_o_creat_o_dir.c | 201 ++ + tools/testing/selftests/filesystems/wrappers.h | 11 + + 83 files changed, 5169 insertions(+), 1321 deletions(-) + delete mode 100644 arch/powerpc/platforms/cell/spufs/coredump.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal.h + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_helper.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_test_helpers.h + create mode 100644 tools/testing/selftests/exec/binfmt_misc_delim.c + create mode 100644 tools/testing/selftests/filesystems/file_stressor/.gitignore + create mode 100644 tools/testing/selftests/filesystems/file_stressor/Makefile + rename tools/testing/selftests/filesystems/{ => file_stressor}/file_stressor.c (100%) + create mode 100644 tools/testing/selftests/filesystems/file_stressor/settings + create mode 100644 tools/testing/selftests/filesystems/fscontext_ns/.gitignore + create mode 100644 tools/testing/selftests/filesystems/open_o_creat_o_dir.c +$ git am -3 ../patches/0001-ksmbd-Fix-removal-of-type-parameter-from-vfs_path_pa.patch +Applying: ksmbd: Fix removal of type parameter from vfs_path_parent_lookup() +Using index info to reconstruct a base tree... +M fs/smb/server/vfs.c +Falling back to patching base and 3-way merge... +Auto-merging fs/smb/server/vfs.c +No changes -- Patch already applied. +Merging vfs/for-next (4dda01b67c866 Merge branches 'work.dcache', 'work.dcache-d_add' and 'work.configfs' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git vfs/for-next +Merge made by the 'ort' strategy. +Merging mm-fixes/for-next-fixes (60492c2aabf53 Merge https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm.git mm-hotfixes-unstable into for-next-fixes) +$ git merge -m Merge branch 'for-next-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git mm-fixes/for-next-fixes +Updating 940de590b839f..60492c2aabf53 +Fast-forward (no commit created; -m option ignored) + .mailmap | 7 +- + Documentation/admin-guide/cgroup-v1/memory.rst | 49 +- + MAINTAINERS | 10 +- + fs/dax.c | 9 +- + fs/super.c | 18 +- + include/linux/sched/user.h | 3 +- + kernel/module/main.c | 8 +- + lib/alloc_tag.c | 1029 ------------------------ + lib/once.c | 2 +- + mm/filemap.c | 2 +- + mm/folio.c | 2 + + mm/huge_memory.c | 88 +- + mm/hugetlb.c | 29 +- + mm/hugetlb_cgroup.c | 7 +- + mm/hugetlb_cma.c | 21 +- + mm/khugepaged.c | 6 - + mm/madvise.c | 8 + + mm/memcontrol-v1.c | 43 +- + mm/memcontrol.c | 2 +- + mm/mempolicy.c | 2 +- + mm/migrate_device.c | 18 + + mm/mlock.c | 2 +- + mm/mremap.c | 29 +- + mm/secretmem.c | 116 ++- + mm/shmem.c | 5 + + mm/shrinker.c | 2 + + mm/slab_common.c | 18 +- + mm/slub.c | 5 +- + mm/swapfile.c | 2 +- + mm/userfaultfd.c | 4 +- + mm/vma.c | 6 +- + tools/testing/selftests/mm/memfd_secret.c | 30 +- + 32 files changed, 354 insertions(+), 1228 deletions(-) + delete mode 100644 lib/alloc_tag.c +Merging fs-current (9ac4915b36faa Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git) +$ git merge -m Merge branch 'fs-current' of linux-next fs-current +Auto-merging MAINTAINERS +Auto-merging fs/super.c +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-fs-erofs | 2 +- + MAINTAINERS | 2 +- + fs/adfs/super.c | 24 ++--- + fs/afs/addr_list.c | 5 +- + fs/afs/dir_edit.c | 9 +- + fs/afs/dir_search.c | 11 +-- + fs/afs/fs_probe.c | 1 + + fs/afs/internal.h | 8 ++ + fs/afs/server.c | 1 - + fs/btrfs/dev-replace.c | 2 +- + fs/btrfs/inode.c | 3 + + fs/btrfs/ioctl.c | 21 +++- + fs/btrfs/raid-stripe-tree.c | 25 +++-- + fs/btrfs/raid-stripe-tree.h | 1 + + fs/btrfs/scrub.c | 24 +++-- + fs/btrfs/send.c | 9 +- + fs/btrfs/tests/extent-io-tests.c | 5 +- + fs/btrfs/transaction.c | 19 +++- + fs/btrfs/volumes.c | 4 + + fs/btrfs/zoned.c | 16 ++- + fs/btrfs/zstd.c | 11 ++- + fs/cachefiles/xattr.c | 16 +-- + fs/erofs/decompressor_lzma.c | 18 +++- + fs/erofs/sysfs.c | 2 + + fs/ext4/inode.c | 11 ++- + fs/netfs/buffered_read.c | 164 +++++++++++++++++++++++-------- + fs/netfs/direct_write.c | 31 +++--- + fs/netfs/internal.h | 3 + + fs/netfs/misc.c | 19 ++++ + fs/netfs/objects.c | 32 +++--- + fs/netfs/read_collect.c | 128 +++++++++++++++++------- + fs/netfs/read_pgpriv2.c | 15 +-- + fs/netfs/read_retry.c | 13 ++- + fs/netfs/read_single.c | 2 + + fs/netfs/rolling_buffer.c | 79 +++++++++------ + fs/netfs/write_issue.c | 2 + + fs/overlayfs/super.c | 2 +- + fs/super.c | 9 +- + fs/ufs/cylinder.c | 10 ++ + fs/ufs/dir.c | 2 +- + fs/ufs/super.c | 17 ++-- + include/linux/netfs.h | 5 +- + include/linux/ns/ns_common_types.h | 6 +- + include/linux/rolling_buffer.h | 6 +- + include/linux/sched.h | 2 +- + include/linux/sched/signal.h | 5 +- + include/trace/events/cachefiles.h | 19 +++- + include/trace/events/netfs.h | 30 +++++- + init/main.c | 2 +- + kernel/reboot.c | 19 +++- + kernel/signal.c | 12 +++ + 51 files changed, 627 insertions(+), 257 deletions(-) +Merging kbuild-current/kbuild-fixes-for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'kbuild-fixes-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git kbuild-current/kbuild-fixes-for-next +Already up to date. +Merging clang-fixes/clang-fixes-for-next (175db11786bde Disable -Wattribute-alias for clang-23 and newer) +$ git merge -m Merge branch 'clang-fixes-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git clang-fixes/clang-fixes-for-next +Already up to date. +Merging arc-current/for-curr (1590cf0329716 Linux 7.2-rc4) +$ git merge -m Merge branch 'for-curr' of https://git.kernel.org/pub/scm/linux/kernel/git/vgupta/arc.git arc-current/for-curr +Already up to date. +Merging arm-current/fixes (1039bffd6ae9c ARM: 9485/1: mm: acquire mmap write lock around show_pte() for user faults) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git arm-current/fixes +Already up to date. +Merging arm64-fixes/for-next/fixes (f73a8edc2ccc6 arm64: make huge_ptep_get handled unaligned addresses) +$ git merge -m Merge branch 'for-next/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux arm64-fixes/for-next/fixes +Already up to date. +Merging arm-soc-fixes/arm/fixes (e36c0670d5ebe Merge tag 'tegra-for-7.2-arm64-dt-fixes-v2' of git://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux into arm/fixes) +$ git merge -m Merge branch 'arm/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git arm-soc-fixes/arm/fixes +Already up to date. +Merging davinci-current/davinci/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'davinci/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git davinci-current/davinci/for-current +Already up to date. +Merging realtek-fixes/fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git realtek-fixes/fixes +Already up to date. +Merging drivers-memory-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git drivers-memory-fixes/fixes +Already up to date. +Merging sophgo-fixes/fixes (19272b37aa4f8 Linux 6.16-rc1) +$ git merge -m Merge branch 'fixes' of https://github.com/sophgo/linux.git sophgo-fixes/fixes +Already up to date. +Merging sophgo-soc-fixes/soc-fixes (0af2f6be1b428 Linux 6.15-rc1) +$ git merge -m Merge branch 'soc-fixes' of https://github.com/sophgo/linux.git sophgo-soc-fixes/soc-fixes +Already up to date. +Merging m68k-current/for-linus (2f8e3cad53b5c m68k: nfcon: Do not call console_is_registered() in nfcon_device()) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git m68k-current/for-linus +Already up to date. +Merging powerpc-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git powerpc-fixes/fixes +Already up to date. +Merging s390-fixes/fixes (98d23edcd4143 s390/zcrypt: Fix uninitialized padding in CRT key structure) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git s390-fixes/fixes +Merge made by the 'ort' strategy. + arch/s390/boot/alternative.c | 5 +- + arch/s390/boot/ipl_parm.c | 28 ++++++------ + arch/s390/boot/physmem_info.c | 2 +- + arch/s390/include/asm/cpacf.h | 6 +++ + arch/s390/include/asm/processor.h | 1 - + arch/s390/include/asm/smp.h | 4 +- + arch/s390/kernel/diag/diag324.c | 3 +- + arch/s390/kernel/ipl.c | 12 +++-- + arch/s390/kernel/perf_pai.c | 27 +++++++++-- + arch/s390/kernel/smp.c | 16 ++----- + arch/s390/kernel/topology.c | 2 +- + arch/s390/kernel/vtime.c | 6 +-- + arch/s390/mm/pgalloc.c | 89 +++++++++++++++--------------------- + arch/s390/pci/pci_sysfs.c | 3 ++ + drivers/s390/crypto/zcrypt_cca_key.h | 1 + + drivers/s390/crypto/zcrypt_ccamisc.c | 15 +++++- + include/linux/device-id/ap.h | 2 - + 17 files changed, 119 insertions(+), 103 deletions(-) +Merging net/main (66817a9794263 net: gro: Fix nesting of TCP GSO SKBs in skb_gro_receive_list()) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net.git net/main +Merge made by the 'ort' strategy. + Documentation/netlink/specs/conntrack.yaml | 21 +- + drivers/bluetooth/btintel.c | 44 +- + drivers/bluetooth/btintel_pcie.c | 3 + + drivers/bluetooth/hci_mrvl.c | 3 +- + drivers/net/bonding/bond_alb.c | 15 +- + drivers/net/bonding/bond_main.c | 18 +- + drivers/net/bonding/bond_options.c | 4 +- + drivers/net/ethernet/airoha/airoha_eth.h | 2 +- + drivers/net/ethernet/amd/xgbe/xgbe-dev.c | 3 +- + drivers/net/ethernet/cadence/macb.h | 3 + + drivers/net/ethernet/cadence/macb_main.c | 21 +- + drivers/net/ethernet/marvell/octeontx2/af/rvu.c | 33 +- + .../net/ethernet/marvell/octeontx2/af/rvu_npc.c | 5 +- + .../net/ethernet/mellanox/mlx5/core/en/xsk/rx.c | 2 + + drivers/net/ethernet/mellanox/mlx5/core/en_rx.c | 7 +- + drivers/net/ethernet/oa_tc6.c | 260 +++++++--- + drivers/net/ethernet/stmicro/stmmac/stmmac_main.c | 61 ++- + drivers/net/gtp.c | 5 + + drivers/net/ipvlan/ipvlan_main.c | 4 +- + drivers/net/ntb_netdev.c | 47 +- + drivers/net/ppp/ppp_async.c | 82 +--- + drivers/net/ppp/ppp_synctty.c | 83 +--- + drivers/net/usb/qmi_wwan.c | 1 + + drivers/net/vxlan/vxlan_mdb.c | 8 + + drivers/s390/net/ctcm_mpc.c | 3 +- + include/linux/igmp.h | 7 +- + include/linux/skbuff.h | 5 + + include/net/af_vsock.h | 3 + + include/net/if_inet6.h | 2 - + include/net/ip.h | 3 +- + include/net/tcp.h | 3 +- + include/trace/events/icmp.h | 13 +- + net/bluetooth/hci_core.c | 4 +- + net/bluetooth/l2cap_core.c | 33 +- + net/bluetooth/msft.c | 2 +- + net/bridge/br_multicast.c | 13 +- + net/core/dev.c | 25 +- + net/core/gro_cells.c | 2 + + net/core/page_pool.c | 3 +- + net/core/sock.c | 3 + + net/ipv4/fib_semantics.c | 2 +- + net/ipv4/igmp.c | 210 +++++---- + net/ipv4/tcp.c | 32 +- + net/ipv4/tcp_cong.c | 4 +- + net/ipv4/tcp_dctcp.c | 4 +- + net/ipv4/tcp_minisocks.c | 2 +- + net/ipv4/tcp_offload.c | 22 +- + net/ipv4/tcp_output.c | 6 +- + net/ipv4/tcp_timer.c | 6 +- + net/ipv4/udp.c | 15 +- + net/ipv6/exthdrs.c | 4 +- + net/ipv6/ip6_gre.c | 6 +- + net/ipv6/mcast.c | 148 +++--- + net/ipv6/route.c | 2 +- + net/ipv6/tcpv6_offload.c | 15 +- + net/ipv6/udp.c | 13 + + net/iucv/af_iucv.c | 42 +- + net/mac802154/ieee802154_i.h | 7 +- + net/mac802154/main.c | 1 + + net/mac802154/scan.c | 51 +- + net/mptcp/protocol.c | 3 +- + net/mptcp/protocol.h | 2 +- + net/packet/af_packet.c | 5 +- + net/qrtr/af_qrtr.c | 66 ++- + net/qrtr/ns.c | 35 +- + net/rds/connection.c | 89 +++- + net/rds/ib_recv.c | 9 +- + net/rds/send.c | 14 +- + net/rds/tcp.c | 101 ++-- + net/rds/tcp_listen.c | 6 +- + net/sched/act_api.c | 37 +- + net/sched/cls_flower.c | 5 + + net/sched/cls_u32.c | 32 +- + net/sctp/inqueue.c | 6 +- + net/sctp/sm_make_chunk.c | 14 +- + net/sctp/sm_sideeffect.c | 11 +- + net/tipc/link.c | 6 +- + net/tipc/name_table.c | 32 +- + net/tipc/node.c | 2 + + net/vmw_vsock/af_vsock.c | 32 ++ + net/vmw_vsock/virtio_transport_common.c | 3 +- + net/vmw_vsock/vmci_transport.c | 34 +- + tools/testing/selftests/net/Makefile | 1 + + tools/testing/selftests/net/exception_cache.sh | 521 +++++++++++++++++++++ + tools/testing/selftests/net/test_vxlan_mdb.sh | 6 + + .../selftests/tc-testing/tc-tests/filters/u32.json | 23 + + 86 files changed, 1841 insertions(+), 705 deletions(-) + create mode 100755 tools/testing/selftests/net/exception_cache.sh +Merging bpf/master (ac0aaef0aa997 selftests/bpf: BPF_PSEUDO_FUNC reference to the main program) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf.git/ bpf/master +Merge made by the 'ort' strategy. + arch/x86/net/bpf_jit_comp.c | 8 ++- + include/linux/bpf.h | 2 +- + include/linux/filter.h | 1 - + kernel/bpf/arraymap.c | 5 +- + kernel/bpf/backtrack.c | 64 +++++++++++++--------- + kernel/bpf/core.c | 5 -- + kernel/bpf/disasm.c | 3 +- + kernel/bpf/hashtab.c | 5 +- + kernel/bpf/local_storage.c | 5 +- + kernel/bpf/percpu_freelist.c | 35 +++++++++--- + kernel/bpf/percpu_freelist.h | 1 + + kernel/bpf/states.c | 11 ++-- + kernel/bpf/verifier.c | 29 ++++++++-- + tools/testing/selftests/bpf/progs/iters.c | 39 +++++++++++++ + .../selftests/bpf/progs/task_local_data.bpf.h | 3 + + .../testing/selftests/bpf/progs/verifier_bounds.c | 41 ++++++++++++++ + tools/testing/selftests/bpf/progs/verifier_cfg.c | 14 +++++ + .../bpf/progs/verifier_jeq_infer_not_null.c | 52 ++++++++++++++++++ + .../selftests/bpf/progs/verifier_spill_fill.c | 40 ++++++++++++++ + .../bpf/progs/verifier_subprog_precision.c | 51 +++++++++++++++++ + tools/testing/selftests/bpf/verifier/pseudo_func.c | 45 +++++++++++++++ + 21 files changed, 394 insertions(+), 65 deletions(-) + create mode 100644 tools/testing/selftests/bpf/verifier/pseudo_func.c +Merging ipsec/master (96f01b53c2d05 net: xfrm: reject unrepresentable espintcp transport headers) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec.git ipsec/master +Auto-merging net/ipv4/esp4.c +Auto-merging net/ipv6/esp6.c +Merge made by the 'ort' strategy. + net/ipv4/esp4.c | 6 ++++++ + net/ipv6/esp6.c | 6 ++++++ + net/ipv6/xfrm6_output.c | 10 ++++++++-- + net/xfrm/espintcp.c | 6 +++++- + net/xfrm/xfrm_input.c | 22 +++++++++++++++++++--- + net/xfrm/xfrm_iptfs.c | 12 ++++++++++-- + net/xfrm/xfrm_policy.c | 20 +++++++++++++++----- + net/xfrm/xfrm_state.c | 9 +++++++-- + net/xfrm/xfrm_user.c | 18 +++++------------- + 9 files changed, 81 insertions(+), 28 deletions(-) +Merging netfilter/main (1b78070aaef63 Merge tag 'net-7.3-rc1' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf.git netfilter/main +Already up to date. +Merging ipvs/main (1b78070aaef63 Merge tag 'net-7.3-rc1' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs.git ipvs/main +Already up to date. +Merging wireless/for-next (1b78070aaef63 Merge tag 'net-7.3-rc1' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless.git wireless/for-next +Already up to date. +Merging ath/for-current (1b78070aaef63 Merge tag 'net-7.3-rc1' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net) +$ git merge -m Merge branch 'for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git ath/for-current +Already up to date. +Merging iwlwifi/fixes (d13d5d299c11b wifi: iwlwifi: validate SEC_RT TLV minimum size) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git iwlwifi/fixes +Already up to date. +Merging wpan/master (2f37fba846c9f mac802154: fix use-after-free of sdata via queued RX frames) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan.git wpan/master +Merge made by the 'ort' strategy. + drivers/net/ieee802154/cc2520.c | 5 +- + drivers/net/ieee802154/mac802154_hwsim.c | 10 ++- + include/net/cfg802154.h | 1 + + net/6lowpan/nhc.h | 3 +- + net/ieee802154/6lowpan/core.c | 2 +- + net/mac802154/ieee802154_i.h | 8 +++ + net/mac802154/iface.c | 6 ++ + net/mac802154/main.c | 1 + + net/mac802154/rx.c | 120 +++++++++++++++++++++++++------ + net/mac802154/scan.c | 10 +-- + 10 files changed, 128 insertions(+), 38 deletions(-) +Merging rdma-fixes/for-rc (3fb905f07ea45 RDMA/irdma: Enforce local fence for IB_WR_REG_MR) +$ git merge -m Merge branch 'for-rc' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git rdma-fixes/for-rc +Merge made by the 'ort' strategy. + drivers/infiniband/core/mad.c | 3 +- + drivers/infiniband/core/rdma_core.c | 1 - + drivers/infiniband/core/uverbs_flow.c | 1 + + drivers/infiniband/core/uverbs_main.c | 25 ++++++-------- + drivers/infiniband/core/verbs.c | 12 ++++--- + drivers/infiniband/hw/bnxt_re/main.c | 10 ++++-- + drivers/infiniband/hw/bnxt_re/uapi.c | 6 ++-- + drivers/infiniband/hw/erdma/erdma_main.c | 4 +-- + drivers/infiniband/hw/erdma/erdma_verbs.c | 18 +++++----- + drivers/infiniband/hw/hfi1/file_ops.c | 23 +++++++++---- + drivers/infiniband/hw/irdma/verbs.c | 2 +- + drivers/infiniband/hw/mlx4/sysfs.c | 4 +++ + drivers/infiniband/hw/mlx5/main.c | 7 ++-- + drivers/infiniband/sw/rxe/rxe_mcast.c | 50 ++++++++++++++++++++-------- + drivers/infiniband/sw/rxe/rxe_mr.c | 3 +- + drivers/infiniband/sw/rxe/rxe_odp.c | 16 ++++++--- + drivers/infiniband/sw/rxe/rxe_verbs.c | 13 ++++---- + drivers/infiniband/sw/siw/siw_cm.c | 7 ++-- + drivers/infiniband/ulp/iser/iser_initiator.c | 16 ++++++--- + drivers/infiniband/ulp/isert/ib_isert.c | 22 ++++++++++++ + drivers/infiniband/ulp/isert/ib_isert.h | 2 ++ + drivers/infiniband/ulp/rtrs/rtrs-clt-trace.h | 2 +- + drivers/infiniband/ulp/rtrs/rtrs-srv-trace.h | 2 +- + drivers/infiniband/ulp/srp/ib_srp.c | 14 +++++--- + include/rdma/uverbs_types.h | 2 -- + include/uapi/rdma/bnxt_re-abi.h | 2 +- + 26 files changed, 176 insertions(+), 91 deletions(-) +Merging sound-current/for-linus (8ba27b90095a4 ALSA: hda/realtek: Fix cold-boot headset misdetection on Acer Aspire A515-57G) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git sound-current/for-linus +Merge made by the 'ort' strategy. + sound/core/pcm_native.c | 37 +++++++++++++++++++++++++++---------- + sound/core/rawmidi.c | 2 +- + sound/core/ump.c | 4 ++++ + sound/drivers/dummy.c | 2 +- + sound/hda/codecs/cirrus/cs420x.c | 2 ++ + sound/hda/codecs/conexant.c | 20 ++++++++++++++++++++ + sound/hda/codecs/realtek/alc269.c | 39 ++++++++++++++++++++++++++++++++++++++- + sound/hda/core/device.c | 5 +++-- + sound/parisc/harmony.c | 6 +++--- + sound/usb/fcp.c | 8 ++++++++ + sound/usb/midi.c | 2 ++ + sound/usb/mixer_maps.c | 18 ++++++++++++++++++ + sound/usb/mixer_quirks.c | 8 ++++++++ + sound/usb/mixer_s1810c.c | 8 ++++++++ + sound/usb/mixer_scarlett.c | 4 ++++ + sound/usb/mixer_scarlett2.c | 36 ++++++++++++++++++++++++++++++------ + sound/usb/mixer_us16x08.c | 7 +++++++ + 17 files changed, 184 insertions(+), 24 deletions(-) +Merging sound-asoc-fixes/for-linus (edc5d19f9ffea ASoC: ux500: Fix MSP lifecycle, clocking and resources) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git sound-asoc-fixes/for-linus +Merge made by the 'ort' strategy. + sound/soc/amd/renoir/acp3x-pdm-dma.c | 2 +- + sound/soc/amd/yc/acp6x-mach.c | 7 + + sound/soc/amd/yc/acp6x-pdm-dma.c | 2 + + sound/soc/codecs/ab8500-codec.c | 688 +++++++++------------- + sound/soc/codecs/cs35l56-sdw.c | 7 +- + sound/soc/codecs/cs35l56.c | 71 ++- + sound/soc/codecs/cs35l56.h | 2 + + sound/soc/codecs/es8326.c | 19 +- + sound/soc/codecs/rt721-sdca.c | 1 + + sound/soc/codecs/tas2783-sdw.c | 25 + + sound/soc/fsl/fsl_micfil.c | 21 +- + sound/soc/intel/boards/Kconfig | 4 +- + sound/soc/intel/boards/sof_rt5682.c | 8 + + sound/soc/intel/common/soc-acpi-intel-nvl-match.c | 8 +- + sound/soc/sprd/sprd-pcm-compress.c | 15 +- + sound/soc/sti/uniperif_reader.c | 4 +- + sound/soc/ux500/ux500_msp_dai.c | 189 +++--- + sound/soc/ux500/ux500_msp_dai.h | 14 +- + sound/soc/ux500/ux500_msp_i2s.c | 260 +++++--- + sound/soc/ux500/ux500_msp_i2s.h | 12 +- + 20 files changed, 704 insertions(+), 655 deletions(-) +Merging regmap-fixes/for-linus (2e42cade8ff1f regmap: irq: Free the irqdomain we create) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git regmap-fixes/for-linus +Merge made by the 'ort' strategy. + drivers/base/regmap/regmap-irq.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) +Merging regulator-fixes/for-linus (ca12149896ed0 regulator: dt-bindings: fan53555: add tcs,tcs4526) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git regulator-fixes/for-linus +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/regulator/fcs,fan53555.yaml | 1 + + 1 file changed, 1 insertion(+) +Merging spi-fixes/for-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git spi-fixes/for-linus +Already up to date. +Merging pci-current/for-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git pci-current/for-linus +Already up to date. +Merging driver-core.current/driver-core-linus (f6d752278c138 MAINTAINERS: Remove Russ Weight from Firmware Loader) +$ git merge -m Merge branch 'driver-core-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git driver-core.current/driver-core-linus +Auto-merging CREDITS +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + CREDITS | 4 ++++ + Documentation/ABI/testing/sysfs-class-firmware | 14 +++++++------- + MAINTAINERS | 1 - + drivers/base/test/Kconfig | 1 - + drivers/base/test/property-entry-test.c | 3 +++ + fs/kernfs/inode.c | 4 +--- + rust/kernel/pci/irq.rs | 4 +++- + 7 files changed, 18 insertions(+), 13 deletions(-) +Merging tty.current/tty-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'tty-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git tty.current/tty-linus +Already up to date. +Merging usb.current/usb-linus (c9273c8388583 usb: typec: qcom-pmic-typec: drain cc_debounce_dwork if port_start() fails) +$ git merge -m Merge branch 'usb-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git usb.current/usb-linus +Merge made by the 'ort' strategy. + drivers/usb/cdns3/cdnsp-gadget.c | 111 ++++++++++++++++++++- + drivers/usb/cdns3/cdnsp-gadget.h | 1 + + drivers/usb/cdns3/cdnsp-mem.c | 98 +++++++----------- + drivers/usb/dwc3/dwc3-google.c | 1 + + drivers/usb/dwc3/ep0.c | 2 +- + drivers/usb/dwc3/gadget.c | 21 ++-- + drivers/usb/gadget/function/f_mass_storage.c | 5 +- + drivers/usb/gadget/function/f_midi.c | 2 +- + drivers/usb/gadget/function/f_midi2.c | 17 ++-- + drivers/usb/gadget/functions.c | 2 +- + drivers/usb/gadget/legacy/inode.c | 3 +- + drivers/usb/host/xhci-mem.c | 2 +- + drivers/usb/host/xhci-ring.c | 43 ++++++-- + drivers/usb/image/mdc800.c | 4 +- + drivers/usb/storage/ene_ub6250.c | 2 + + drivers/usb/storage/realtek_cr.c | 9 +- + drivers/usb/typec/hd3ss3220.c | 9 +- + drivers/usb/typec/mux.c | 21 +++- + .../usb/typec/tcpm/qcom/qcom_pmic_typec_pdphy.c | 2 + + drivers/usb/typec/tcpm/qcom/qcom_pmic_typec_port.c | 5 + + drivers/usb/typec/tcpm/tcpm.c | 28 ++++-- + drivers/usb/typec/tipd/core.c | 17 +++- + drivers/usb/typec/tipd/tps6598x.h | 4 +- + drivers/usb/typec/ucsi/displayport.c | 2 +- + 24 files changed, 284 insertions(+), 127 deletions(-) +Merging usb-serial-fixes/usb-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'usb-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git usb-serial-fixes/usb-linus +Already up to date. +Merging phy/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git phy/fixes +Already up to date. +Merging staging.current/staging-linus (cc7cd2a922817 staging: sm750fb: fix mono image source stride mismatch in lynxfb_ops_imageblit()) +$ git merge -m Merge branch 'staging-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git staging.current/staging-linus +Merge made by the 'ort' strategy. + drivers/staging/fbtft/fbtft-core.c | 9 +++++---- + drivers/staging/rtl8723bs/core/rtw_ieee80211.c | 7 +++++++ + drivers/staging/rtl8723bs/core/rtw_mlme.c | 3 +++ + drivers/staging/sm750fb/sm750.c | 2 +- + drivers/staging/sm750fb/sm750.h | 2 +- + drivers/staging/sm750fb/sm750_accel.c | 6 ++---- + drivers/staging/sm750fb/sm750_accel.h | 4 +--- + 7 files changed, 20 insertions(+), 13 deletions(-) +Merging iio-fixes/fixes-togreg (da7c937e0abc8 dt-bindings: iio: adc: rockchip-saradc: Fix RV1106 compatible) +$ git merge -m Merge branch 'fixes-togreg' of https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git iio-fixes/fixes-togreg +Merge made by the 'ort' strategy. + .../bindings/iio/adc/rockchip-saradc.yaml | 18 ++++++------- + drivers/iio/accel/kionix-kx022a.c | 2 +- + drivers/iio/accel/sca3000.c | 2 +- + drivers/iio/adc/ade9000.c | 31 +++++++++++----------- + drivers/iio/adc/adi-axi-adc.c | 4 +++ + drivers/iio/adc/aspeed_adc.c | 4 ++- + drivers/iio/adc/max1363.c | 8 ++++++ + drivers/iio/adc/rohm-bd79124.c | 20 +++++++++----- + drivers/iio/adc/xilinx-xadc-core.c | 9 ++++--- + drivers/iio/buffer/industrialio-buffer-dmaengine.c | 16 +++++++---- + .../iio/common/inv_sensors/inv_sensors_timestamp.c | 11 +++++--- + drivers/iio/dac/rohm-bd79703.c | 3 +++ + drivers/iio/frequency/adf4377.c | 4 +-- + drivers/iio/frequency/admv1013.c | 9 ++++--- + drivers/iio/gyro/adis16136.c | 2 +- + drivers/iio/health/max30102.c | 12 ++++++--- + drivers/iio/imu/adis16400.c | 2 +- + drivers/iio/imu/adis16480.c | 4 +-- + drivers/iio/industrialio-buffer.c | 3 ++- + drivers/iio/industrialio-trigger.c | 2 ++ + drivers/iio/light/gp2ap020a00f.c | 2 ++ + drivers/iio/light/rohm-bu27034.c | 2 +- + drivers/iio/pressure/bmp280-core.c | 2 +- + drivers/iio/pressure/rohm-bm1390.c | 2 +- + drivers/iio/proximity/aw96103.c | 30 ++++++++++++++++----- + drivers/iio/proximity/pulsedlight-lidar-lite-v2.c | 4 ++- + drivers/iio/proximity/sx9324.c | 2 +- + 27 files changed, 141 insertions(+), 69 deletions(-) +Merging watchdog-fixes/watchdog (d83b7502bb087 MAINTAINERS: Update URI for watchdog tree) +$ git merge -m Merge branch 'watchdog' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git watchdog-fixes/watchdog +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 2 +- + drivers/watchdog/msc313e_wdt.c | 1 + + drivers/watchdog/sunxi_wdt.c | 45 +++++++++++++++++++++++++++++++++++++++++- + 3 files changed, 46 insertions(+), 2 deletions(-) +Merging counter-current/counter-current (f1a3a9946aab6 counter: microchip-tcb-capture: Fix DT channel validation) +$ git merge -m Merge branch 'counter-current' of https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git counter-current/counter-current +Already up to date. +Merging char-misc.current/char-misc-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'char-misc-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git char-misc.current/char-misc-linus +Already up to date. +Merging soundwire-fixes/fixes (6d49beec658f6 soundwire: dmi-quirks: Disable ghost Realtek on Asus GX651AX) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git soundwire-fixes/fixes +Merge made by the 'ort' strategy. + drivers/soundwire/dmi-quirks.c | 7 +++++++ + 1 file changed, 7 insertions(+) +Merging thunderbolt-fixes/fixes (4310c6b8e75d6 thunderbolt: Fix NULL dereference in tb_remove_work()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git thunderbolt-fixes/fixes +Merge made by the 'ort' strategy. + drivers/thunderbolt/switch.c | 9 +++++- + drivers/thunderbolt/tb.c | 37 ++++++++++++++-------- + drivers/thunderbolt/test.c | 58 ++++++++++++++++++++++++++++------- + drivers/thunderbolt/tunnel.c | 73 ++++++++++++++++++++++++++++---------------- + drivers/thunderbolt/tunnel.h | 8 +++-- + 5 files changed, 131 insertions(+), 54 deletions(-) +Merging input-current/for-linus (85f080fb87ed5 Input: cyttsp5 - clamp the HID report size before memcpy) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git input-current/for-linus +Merge made by the 'ort' strategy. + .../devicetree/bindings/input/mediatek,mt6779-keypad.yaml | 1 + + drivers/input/evdev.c | 2 ++ + drivers/input/input-compat.c | 2 ++ + drivers/input/keyboard/adp5588-keys.c | 12 ++++++------ + drivers/input/keyboard/atkbd.c | 8 ++++++++ + drivers/input/rmi4/rmi_driver.c | 13 +++++++++++++ + drivers/input/rmi4/rmi_smbus.c | 10 +++++----- + drivers/input/touchscreen/cyttsp5.c | 1 + + 8 files changed, 38 insertions(+), 11 deletions(-) +Merging crypto-current/master (ee440d4fc0d2f crypto: acomp - allocate async request context when cloning) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/herbert/crypto-2.6.git crypto-current/master +Already up to date. +Merging libcrypto-fixes/libcrypto-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'libcrypto-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git libcrypto-fixes/libcrypto-fixes +Already up to date. +Merging vfio-fixes/for-linus (e242e974e812e vfio: selftests: Add luuid to libvfio.mk's list of libraries, not to the Makefile) +$ git merge -m Merge branch 'for-linus' of https://github.com/awilliam/linux-vfio.git vfio-fixes/for-linus +Already up to date. +Merging kselftest-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kselftest-fixes/fixes +Already up to date. +Merging dmaengine-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git dmaengine-fixes/fixes +Already up to date. +Merging backlight-fixes/for-backlight-fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'for-backlight-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git backlight-fixes/for-backlight-fixes +Already up to date. +Merging mtd-fixes/mtd/fixes (2b533e775aec5 Revert "mtd: maps: remove uclinux map driver") +$ git merge -m Merge branch 'mtd/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git mtd-fixes/mtd/fixes +Already up to date. +Merging mfd-fixes/for-mfd-fixes (d5d2d7a8d8be1 MAINTAINERS: Add a mailing list entry to MFD) +$ git merge -m Merge branch 'for-mfd-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git mfd-fixes/for-mfd-fixes +Already up to date. +Merging v4l-dvb-fixes/fixes (e04ffff543db0 media: rppx1: bls: read the raw pattern from the PRE2 acquisition module) +$ git merge -m Merge branch 'fixes' of git://linuxtv.org/media-ci/media-pending.git v4l-dvb-fixes/fixes +Merge made by the 'ort' strategy. + drivers/media/platform/dreamchip/rppx1/rpp_params.c | 1 + + drivers/media/platform/dreamchip/rppx1/rppx1_bls.c | 2 +- + 2 files changed, 2 insertions(+), 1 deletion(-) +Merging reset-fixes/reset/fixes (71827776667f4 reset: imx7: Correct polarity of MIPI CSI resets on i.MX8MQ) +$ git merge -m Merge branch 'reset/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux reset-fixes/reset/fixes +Already up to date. +Merging mips-fixes/mips-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'mips-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git mips-fixes/mips-fixes +Already up to date. +Merging at91-fixes/at91-fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'at91-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git at91-fixes/at91-fixes +Already up to date. +Merging omap-fixes/fixes (2fabd2f406d0c ARM: dts: ti/omap: dra7: fix PCIe PHY clock divider definition) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git omap-fixes/fixes +Merge made by the 'ort' strategy. + arch/arm/boot/dts/ti/omap/dra7xx-clocks.dtsi | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) +Merging tegra-fixes/fixes (0bc9d07096f17 Merge branch for-7.3/arm64/dt into fixes) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git tegra-fixes/fixes +Merge made by the 'ort' strategy. + .../boot/dts/nvidia/tegra264-p4071-0000+p3834.dtsi | 13 ++ + arch/arm64/boot/dts/nvidia/tegra264.dtsi | 180 ++++++++++++++++++--- + drivers/soc/tegra/fuse/tegra-apbmisc.c | 2 +- + 3 files changed, 170 insertions(+), 25 deletions(-) +Merging kvm-fixes/master (8d3ae59288f1e Linux 7.2) +$ git merge -m Merge branch 'master' of git://git.kernel.org/pub/scm/virt/kvm/kvm.git kvm-fixes/master +Already up to date. +Merging kvms390-fixes/master (f47190b08b71e s390/uv: Prevent potential out-of-bounds read) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git kvms390-fixes/master +Merge made by the 'ort' strategy. + arch/s390/kernel/uv.c | 8 ++-- + arch/s390/kvm/gmap/dat.c | 65 ++++++++++++++++------------ + arch/s390/kvm/gmap/dat.h | 10 ++--- + arch/s390/kvm/gmap/gmap.c | 6 ++- + arch/s390/kvm/gmap/kvm_mmu.c | 91 ++++++++++++++++----------------------- + arch/s390/kvm/gmap/kvm_mmu.h | 4 -- + arch/s390/kvm/s390/gaccess.c | 13 ++++++ + arch/s390/kvm/s390/interrupt.c | 81 +++++++++++++++++----------------- + arch/s390/kvm/s390/s390.c | 3 +- + arch/s390/kvm/s390/s390.h | 2 +- + drivers/s390/crypto/vfio_ap_ops.c | 18 ++++++-- + 11 files changed, 157 insertions(+), 144 deletions(-) +Merging kvm-arm-fixes/fixes (679d7201c1f09 KVM: arm64: Reject guest_memfd memslots when the VM has MTE) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git kvm-arm-fixes/fixes +Already up to date. +Merging hwmon-fixes/hwmon (a15f90964998e hwmon: (corsair-cpro) Remove debugfs entries when probe fails) +$ git merge -m Merge branch 'hwmon' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git hwmon-fixes/hwmon +Merge made by the 'ort' strategy. + Documentation/hwmon/gpd-fan.rst | 2 +- + Documentation/hwmon/hwmon-kernel-api.rst | 15 +++ + drivers/hwmon/applesmc.c | 7 +- + drivers/hwmon/aspeed-pwm-tacho.c | 4 +- + drivers/hwmon/chipcap2.c | 9 +- + drivers/hwmon/corsair-cpro.c | 24 +++-- + drivers/hwmon/gpio-fan.c | 13 ++- + drivers/hwmon/hwmon.c | 31 +++--- + drivers/hwmon/ina2xx.c | 163 ++++++++++++++++++++++++++----- + drivers/hwmon/ltc4282.c | 2 +- + drivers/hwmon/mcp9982.c | 2 + + drivers/hwmon/pmbus/pmbus_core.c | 4 +- + drivers/hwmon/sht4x.c | 8 +- + drivers/hwmon/yogafan.c | 2 +- + 14 files changed, 223 insertions(+), 63 deletions(-) +Merging nvdimm-fixes/libnvdimm-fixes (a8aec14230322 nvdimm/bus: Fix potential use after free in asynchronous initialization) +$ git merge -m Merge branch 'libnvdimm-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git nvdimm-fixes/libnvdimm-fixes +Already up to date. +Merging cxl-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git cxl-fixes/fixes +Already up to date. +Merging dma-mapping-fixes/dma-mapping-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'dma-mapping-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git dma-mapping-fixes/dma-mapping-fixes +Already up to date. +Merging drivers-x86-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git drivers-x86-fixes/fixes +Already up to date. +Merging samsung-krzk-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git samsung-krzk-fixes/fixes +Already up to date. +Merging pinctrl-samsung-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git pinctrl-samsung-fixes/fixes +Already up to date. +Merging pinctrl-qcom-fixes/pinctrl-qcom/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'pinctrl-qcom/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pinctrl-qcom-fixes/pinctrl-qcom/for-current +Already up to date. +Merging devicetree-fixes/dt/linus (5bb01c657ff9f of: fix out-of-bounds read in of_alias_scan() stem parser) +$ git merge -m Merge branch 'dt/linus' of https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git devicetree-fixes/dt/linus +Already up to date. +Merging dt-krzk-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git dt-krzk-fixes/fixes +Already up to date. +Merging scsi-fixes/fixes (af8c27375733f scsi: megaraid_sas: Limit NVMe request size to the PRP chain frame) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git scsi-fixes/fixes +Merge made by the 'ort' strategy. + drivers/scsi/fnic/fnic_nvme.c | 2 +- + drivers/scsi/ibmvscsi/ibmvfc-core.c | 3 +- + drivers/scsi/megaraid/megaraid_sas_base.c | 13 ++++++- + drivers/scsi/mpi3mr/mpi3mr_os.c | 45 +++++++++++++++++------ + drivers/scsi/mpi3mr/mpi3mr_transport.c | 8 +++++ + drivers/scsi/mpt3sas/mpt3sas_base.c | 5 ++- + drivers/scsi/pm8001/pm8001_init.c | 4 +-- + drivers/scsi/scsi_bsg.c | 47 ++++++++++++++---------- + drivers/target/iscsi/iscsi_target.c | 4 ++- + drivers/target/iscsi/iscsi_target_login.c | 2 +- + drivers/ufs/host/ufs-qcom.c | 17 ++++++--- + drivers/ufs/host/ufs-qcom.h | 1 + + drivers/ufs/host/ufshcd-pci.c | 59 +++++++++++++++++++++++++++++++ + 13 files changed, 170 insertions(+), 40 deletions(-) +Merging drm-fixes/drm-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'drm-fixes' of https://gitlab.freedesktop.org/drm/kernel.git drm-fixes/drm-fixes +Already up to date. +Merging drm-intel-fixes/for-linux-next-fixes (3785d40831ba5 drm/i915: Guard against NULL driver_data in i915_pci_probe()) +$ git merge -m Merge branch 'for-linux-next-fixes' of https://gitlab.freedesktop.org/drm/i915/kernel.git drm-intel-fixes/for-linux-next-fixes +Merge made by the 'ort' strategy. + drivers/gpu/drm/i915/display/intel_cdclk.c | 10 ++++++---- + drivers/gpu/drm/i915/display/intel_cursor.c | 15 ++++++++++----- + drivers/gpu/drm/i915/display/intel_cx0_phy.c | 5 +++-- + drivers/gpu/drm/i915/display/intel_ddi.c | 11 +++++++++++ + drivers/gpu/drm/i915/display/intel_ddi.h | 1 + + drivers/gpu/drm/i915/display/intel_dp_mst.c | 4 ---- + drivers/gpu/drm/i915/display/intel_lt_phy.c | 6 ++++-- + drivers/gpu/drm/i915/display/skl_universal_plane.c | 15 ++++++++++----- + drivers/gpu/drm/i915/i915_pci.c | 3 +++ + 9 files changed, 48 insertions(+), 22 deletions(-) +Merging mmc-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git mmc-fixes/fixes +Already up to date. +Merging rtc-fixes/rtc-fixes (254f49634ee16 Linux 7.1-rc1) +$ git merge -m Merge branch 'rtc-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git rtc-fixes/rtc-fixes +Already up to date. +Merging gnss-fixes/gnss-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'gnss-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git gnss-fixes/gnss-linus +Already up to date. +Merging hyperv-fixes/hyperv-fixes (0fd49f7bfb8f1 mshv_vtl: bounds-check cpu index in vtl mmap fault handler) +$ git merge -m Merge branch 'hyperv-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git hyperv-fixes/hyperv-fixes +Auto-merging drivers/hv/connection.c +Merge made by the 'ort' strategy. +Merging risc-v-fixes/fixes (b94cec5761d22 riscv: skip software algning code for HAVE_EFFICIENT_UNALIGNED_ACCESS) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git risc-v-fixes/fixes +Merge made by the 'ort' strategy. + Documentation/arch/riscv/hwprobe.rst | 8 +- + Documentation/devicetree/bindings/riscv/cpus.yaml | 8 +- + .../zh_CN/arch/riscv/patch-acceptance.rst | 44 +++++++--- + arch/riscv/Kconfig | 2 +- + arch/riscv/include/asm/bug.h | 5 -- + arch/riscv/include/asm/switch_to.h | 4 +- + arch/riscv/kernel/cpufeature.c | 20 ++++- + arch/riscv/kernel/patch.c | 2 + + arch/riscv/kernel/process.c | 4 +- + arch/riscv/kernel/sys_hwprobe.c | 5 +- + arch/riscv/kernel/usercfi.c | 5 +- + arch/riscv/lib/uaccess.S | 5 +- + arch/riscv/mm/init.c | 4 +- + drivers/perf/riscv_pmu_legacy.c | 5 +- + drivers/perf/riscv_pmu_sbi.c | 98 +++++++++++++++------- + include/linux/perf/riscv_pmu.h | 2 +- + tools/testing/selftests/riscv/cfi/cfi_rv_test.h | 2 +- + tools/testing/selftests/riscv/hwprobe/hwprobe.c | 20 ++++- + 18 files changed, 169 insertions(+), 74 deletions(-) +Merging riscv-dt-fixes/riscv-dt-fixes (42c57c049054d riscv: dts: starfive: jh7110-common: fix jh7110 SoC boot from SD-card.) +$ git merge -m Merge branch 'riscv-dt-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-dt-fixes/riscv-dt-fixes +Merge made by the 'ort' strategy. + arch/riscv/boot/dts/starfive/jh7110-common.dtsi | 1 + + 1 file changed, 1 insertion(+) +Merging riscv-soc-fixes/riscv-soc-fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'riscv-soc-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-soc-fixes/riscv-soc-fixes +Already up to date. +Merging fpga-fixes/fixes (19272b37aa4f8 Linux 6.16-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git fpga-fixes/fixes +Already up to date. +Merging spdx/spdx-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'spdx-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/spdx.git spdx/spdx-linus +Already up to date. +Merging gpio-brgl-fixes/gpio/for-current (1f1d0812f6a8a gpiolib: of: don't mark hog nodes OF_POPULATED before a chip is found) +$ git merge -m Merge branch 'gpio/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git gpio-brgl-fixes/gpio/for-current +Merge made by the 'ort' strategy. + drivers/gpio/gpiolib-of.c | 6 +++--- + drivers/gpio/gpiolib-shared.c | 9 ++++++--- + 2 files changed, 9 insertions(+), 6 deletions(-) +Merging gpio-intel-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git gpio-intel-fixes/fixes +Already up to date. +Merging pinctrl-intel-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git pinctrl-intel-fixes/fixes +Already up to date. +Merging auxdisplay-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git auxdisplay-fixes/fixes +Already up to date. +Merging kunit-fixes/kunit-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'kunit-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kunit-fixes/kunit-fixes +Already up to date. +Merging renesas-fixes/fixes (2ac7bad110be6 arm64: dts: renesas: r9a09g087: Switch GBETH TX queue scheduling to WRR) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git renesas-fixes/fixes +Merge made by the 'ort' strategy. + arch/arm64/boot/dts/renesas/r9a09g047.dtsi | 10 ++++++++++ + arch/arm64/boot/dts/renesas/r9a09g056.dtsi | 10 ++++++++++ + arch/arm64/boot/dts/renesas/r9a09g057.dtsi | 10 ++++++++++ + arch/arm64/boot/dts/renesas/r9a09g077.dtsi | 27 +++++++++++++++++++++++++++ + arch/arm64/boot/dts/renesas/r9a09g087.dtsi | 27 +++++++++++++++++++++++++++ + 5 files changed, 84 insertions(+) +Merging perf-current/perf-tools (aadea57f53288 perf powerpc-vpadtl: Fix raw_size of DTL samples) +$ git merge -m Merge branch 'perf-tools' of https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools.git perf-current/perf-tools +Merge made by the 'ort' strategy. + tools/perf/util/powerpc-vpadtl.c | 2 +- + tools/perf/util/symbol.c | 11 ++++++++++- + 2 files changed, 11 insertions(+), 2 deletions(-) +Merging efi-fixes/urgent (d8809f6931065 efi: sysfb_efi: Extend quirk to cover IdeaPad Duet 3 10IGL5-LTE) +$ git merge -m Merge branch 'urgent' of https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git efi-fixes/urgent +Already up to date. +Merging battery-fixes/fixes (160a783aa65b7 power: supply: bq25890: fix the -10 C NTC lookup entry) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git battery-fixes/fixes +Already up to date. +Merging iommufd-fixes/for-rc (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-rc' of https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git iommufd-fixes/for-rc +Already up to date. +Merging rust-fixes/rust-fixes (e510334fbaeaa rust: samples: add missing newlines in rust_print_main) +$ git merge -m Merge branch 'rust-fixes' of https://github.com/Rust-for-Linux/linux.git rust-fixes/rust-fixes +Merge made by the 'ort' strategy. + rust/pin-init/src/lib.rs | 8 +------- + samples/rust/rust_print_main.rs | 8 ++++---- + 2 files changed, 5 insertions(+), 11 deletions(-) +Merging w1-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git w1-fixes/fixes +Already up to date. +Merging pmdomain-fixes/fixes (2b0ac85512b7f cpuidle: dt_idle_genpd: kfree() the original name allocation) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git pmdomain-fixes/fixes +Merge made by the 'ort' strategy. + drivers/cpuidle/cpuidle-psci.c | 42 +++++++++++++++------------------------ + drivers/cpuidle/dt_idle_genpd.c | 3 +-- + drivers/pmdomain/mediatek/Kconfig | 5 +++-- + drivers/pmdomain/qcom/rpmhpd.c | 4 ---- + 4 files changed, 20 insertions(+), 34 deletions(-) +Merging i2c-andi-fixes/i2c/i2c-fixes (b15b548d52b43 i2c: core: fix debugfs UAF on adapter removal) +$ git merge -m Merge branch 'i2c/i2c-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git i2c-andi-fixes/i2c/i2c-fixes +Already up to date. +Merging i2c-rust-fixes/rust-i2c-fixes (4eb422482ca5d rust: i2c: fix I2cAdapter refcounts double increment) +$ git merge -m Merge branch 'rust-i2c-fixes' of https://github.com/ikrtn/rust-for-linux i2c-rust-fixes/rust-i2c-fixes +Already up to date. +Merging sparc-fixes/for-linus (254f49634ee16 Linux 7.1-rc1) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git sparc-fixes/for-linus +Already up to date. +Merging clk-fixes/clk-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'clk-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git clk-fixes/clk-fixes +Already up to date. +Merging thead-clk-fixes/thead-clk-fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'thead-clk-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-clk-fixes/thead-clk-fixes +Already up to date. +Merging tenstorrent-clk-fixes/tenstorrent-clk-fixes (6de23f81a5e08 Linux 7.0-rc1) +$ git merge -m Merge branch 'tenstorrent-clk-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-clk-fixes/tenstorrent-clk-fixes +Already up to date. +Merging fustini-config-fixes/riscv-config-fixes (254f49634ee16 Linux 7.1-rc1) +$ git merge -m Merge branch 'riscv-config-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git fustini-config-fixes/riscv-config-fixes +Already up to date. +Merging pwrseq-fixes/pwrseq/for-current (3b54dbd119805 power: sequencing: Fix build issue with COMPILE_TEST) +$ git merge -m Merge branch 'pwrseq/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pwrseq-fixes/pwrseq/for-current +Merge made by the 'ort' strategy. + drivers/power/sequencing/Kconfig | 3 ++- + 1 file changed, 2 insertions(+), 1 deletion(-) +Merging thead-dt-fixes/thead-dt-fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'thead-dt-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-dt-fixes/thead-dt-fixes +Already up to date. +Merging ftrace-fixes/ftrace/fixes (1650a1b6cb1ae fgraph: Check ftrace_pids_enabled on registration for early filtering) +$ git merge -m Merge branch 'ftrace/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git ftrace-fixes/ftrace/fixes +Already up to date. +Merging ring-buffer-fixes/ring-buffer/fixes (057caace5214d tracing: Create output file from cmd_check_undefined) +$ git merge -m Merge branch 'ring-buffer/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git ring-buffer-fixes/ring-buffer/fixes +Already up to date. +Merging trace-fixes/trace/fixes (5eab74874d111 ring-buffer: Stop remote reader update when page swap fails) +$ git merge -m Merge branch 'trace/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git trace-fixes/trace/fixes +Already up to date. +Merging tracefs-fixes/tracefs/fixes (07004a8c4b572 eventfs: Hold eventfs_mutex and SRCU when remount walks events) +$ git merge -m Merge branch 'tracefs/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git tracefs-fixes/tracefs/fixes +Already up to date. +Merging spacemit-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux spacemit-fixes/fixes +Already up to date. +Merging tip-fixes/tip/urgent (9cbd68d265373 Merge branch into tip/master: 'x86/urgent') +$ git merge -m Merge branch 'tip/urgent' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git tip-fixes/tip/urgent +Merge made by the 'ort' strategy. + arch/x86/kernel/amd_node.c | 5 ++++ + arch/x86/kernel/itmt.c | 8 ++---- + include/linux/interrupt_rc.h | 22 ++++++++-------- + include/linux/preempt.h | 4 --- + kernel/events/core.c | 20 +++++++-------- + kernel/events/ring_buffer.c | 9 +++++-- + kernel/locking/lockdep.c | 50 ++++++++++++++++++++++++++++++------ + kernel/sched/core.c | 10 ++++++-- + kernel/sched/deadline.c | 4 +-- + kernel/sched/fair.c | 60 ++++++++++++++++++++++++++++++++++++-------- + kernel/sched/rt.c | 4 +-- + kernel/softirq.c | 17 +++---------- + 12 files changed, 143 insertions(+), 70 deletions(-) +Merging kexec-fixes/kexec-fixes (a901b0778ae82 Merge patch series "kexec: fix probe error codes and error propagation") +$ git merge -m Merge branch 'kexec-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git kexec-fixes/kexec-fixes +Merge made by the 'ort' strategy. + arch/arm64/kernel/kexec_image.c | 4 ++-- + arch/loongarch/kernel/kexec_efi.c | 4 ++-- + arch/riscv/kernel/kexec_image.c | 4 ++-- + kernel/kexec_elf.c | 4 ++-- + kernel/kexec_file.c | 12 +++++++----- + 5 files changed, 15 insertions(+), 13 deletions(-) +Merging liveupdate-fixes/fixes (3a0b8fa2eb36a kho: fix size calculation in kho_preserved_memory_reserve()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git liveupdate-fixes/fixes +Already up to date. +Merging drm-msm-fixes/msm-fixes (e2332abed2a4d drm/msm: Only fini scheduler after successful init) +$ git merge -m Merge branch 'msm-fixes' of https://gitlab.freedesktop.org/drm/msm.git drm-msm-fixes/msm-fixes +Already up to date. +Merging uml-fixes/fixes (af421e9aed392 um: vector: fix use-after-free in vector_mmsg_rx()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git uml-fixes/fixes +Already up to date. +Merging fwctl-fixes/for-rc (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-rc' of https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git fwctl-fixes/for-rc +Already up to date. +Merging devsec-tsm-fixes/fixes (c3fd16c3b98ed virt: tdx-guest: Fix handling of host controlled 'quote' buffer length) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git devsec-tsm-fixes/fixes +Already up to date. +Merging drm-rust-fixes/for-linux-next-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-linux-next-fixes' of https://gitlab.freedesktop.org/drm/rust/kernel.git drm-rust-fixes/for-linux-next-fixes +Already up to date. +Merging tenstorrent-dt-fixes/tenstorrent-dt-fixes (6de23f81a5e08 Linux 7.0-rc1) +$ git merge -m Merge branch 'tenstorrent-dt-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-dt-fixes/tenstorrent-dt-fixes +Already up to date. +Merging nfc-fixes/for-linus (50a0fd2402776 selftests: nci: Correct pthread_create return value check) +$ git merge -m Merge branch 'for-linus' of https://codeberg.org/linux-nfc/linux.git nfc-fixes/for-linus +Merge made by the 'ort' strategy. + drivers/nfc/nfcmrvl/fw_dnld.c | 11 ++++++++--- + drivers/nfc/port100.c | 7 +++++++ + drivers/nfc/st21nfca/i2c.c | 29 +++++++++++++++++++---------- + net/nfc/llcp_core.c | 25 +++++++++++++++++++++++++ + tools/testing/selftests/nci/nci_dev.c | 12 +++++++----- + 5 files changed, 66 insertions(+), 18 deletions(-) +Merging drm-misc-fixes/for-linux-next-fixes (d3609b5408389 MAINTAINERS, mailmap: use Aditya Garg's linux.dev account) +$ git merge -m Merge branch 'for-linux-next-fixes' of https://gitlab.freedesktop.org/drm/misc/kernel.git drm-misc-fixes/for-linux-next-fixes +Auto-merging .mailmap +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .mailmap | 3 +- + MAINTAINERS | 2 +- + drivers/accel/amdxdna/aie2_message.c | 2 +- + drivers/accel/amdxdna/amdxdna_ctx.c | 4 +- + drivers/accel/amdxdna/amdxdna_ctx.h | 2 +- + drivers/accel/amdxdna/amdxdna_gem.c | 10 +- + drivers/accel/ethosu/ethosu_drv.c | 2 + + drivers/accel/ethosu/ethosu_gem.c | 2 +- + drivers/accel/ethosu/ethosu_job.c | 10 +- + drivers/accel/qaic/qaic_control.c | 46 ++-- + drivers/dma-buf/dma-buf.c | 20 ++ + drivers/dma-buf/dma-heap.c | 80 +++--- + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c | 6 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_plane.c | 31 ++- + drivers/gpu/drm/drm_atomic_state_helper.c | 7 + + drivers/gpu/drm/drm_atomic_uapi.c | 5 +- + drivers/gpu/drm/drm_pagemap.c | 270 ++++++++++++++++++--- + drivers/gpu/drm/drm_prime.c | 2 +- + drivers/gpu/drm/gud/gud_connector.c | 12 +- + drivers/gpu/drm/gud/gud_drv.c | 2 + + drivers/gpu/drm/nouveau/include/nvkm/engine/disp.h | 1 + + drivers/gpu/drm/nouveau/nouveau_chan.c | 9 +- + drivers/gpu/drm/nouveau/nouveau_dmem.c | 18 +- + drivers/gpu/drm/nouveau/nouveau_sgdma.c | 4 +- + drivers/gpu/drm/nouveau/nouveau_uvmm.c | 6 +- + drivers/gpu/drm/nouveau/nvkm/engine/device/base.c | 10 +- + drivers/gpu/drm/nouveau/nvkm/engine/disp/Kbuild | 1 + + drivers/gpu/drm/nouveau/nvkm/engine/disp/ga102.c | 13 +- + drivers/gpu/drm/nouveau/nvkm/engine/disp/gb202.c | 191 +++++++++++++++ + drivers/gpu/drm/nouveau/nvkm/engine/disp/head.h | 2 + + drivers/gpu/drm/nouveau/nvkm/engine/disp/ior.h | 1 + + drivers/gpu/drm/nouveau/nvkm/engine/disp/priv.h | 17 ++ + drivers/gpu/drm/nouveau/nvkm/engine/disp/tu102.c | 86 ++++++- + .../gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/disp.c | 125 ++++------ + .../gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/disp.c | 64 +++++ + .../gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/gsp.c | 9 + + .../nouveau/nvkm/subdev/gsp/rm/r570/nvrm/disp.h | 2 + + drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/rm.h | 5 + + drivers/gpu/drm/nouveau/nvkm/subdev/instmem/nv50.c | 3 + + drivers/gpu/drm/sysfb/ofdrm.c | 8 +- + drivers/gpu/drm/tegra/dc.c | 6 + + drivers/gpu/drm/tegra/hub.c | 2 + + drivers/gpu/drm/tiny/cirrus-qemu.c | 3 + + drivers/gpu/drm/virtio/virtgpu_display.c | 9 +- + drivers/gpu/drm/virtio/virtgpu_drv.h | 21 ++ + drivers/gpu/drm/virtio/virtgpu_kms.c | 1 + + drivers/gpu/drm/virtio/virtgpu_object.c | 2 +- + drivers/gpu/drm/virtio/virtgpu_vq.c | 21 +- + drivers/misc/fastrpc.c | 16 +- + include/drm/drm_pagemap.h | 8 +- + include/linux/dma-buf.h | 1 + + include/linux/dma-fence-array.h | 1 - + include/linux/dma-fence-chain.h | 9 +- + tools/testing/selftests/dmabuf-heaps/dmabuf-heap.c | 113 ++++++++- + 54 files changed, 1072 insertions(+), 234 deletions(-) + create mode 100644 drivers/gpu/drm/nouveau/nvkm/engine/disp/gb202.c +Merging rust/rust-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'rust-next' of https://github.com/Rust-for-Linux/linux.git rust/rust-next +Already up to date. +Merging rust-interop/interop-next (05f7e89ab9731 Linux 6.19) +$ git merge -m Merge branch 'interop-next' of https://github.com/Rust-for-Linux/linux.git rust-interop/interop-next +Already up to date. +Merging rust-alloc/alloc-next (82134823a07b4 MAINTAINERS: add Alice Ryhl as co-maintainer for Rust [ALLOC]) +$ git merge -m Merge branch 'alloc-next' of https://github.com/Rust-for-Linux/linux.git rust-alloc/alloc-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 1 + + 1 file changed, 1 insertion(+) +Merging rust-io/io-next (86731a2a651e5 Linux 6.16-rc3) +$ git merge -m Merge branch 'io-next' of https://github.com/Rust-for-Linux/linux.git rust-io/io-next +Already up to date. +Merging rust-pin-init/pin-init-next (1e26aea0355ad rust: pin-init: add `#[inline]` to small functions) +$ git merge -m Merge branch 'pin-init-next' of https://github.com/Rust-for-Linux/linux.git rust-pin-init/pin-init-next +Already up to date. +Merging rust-timekeeping/timekeeping-next (ddb1444d33351 hrtimer: add usage examples to documentation) +$ git merge -m Merge branch 'timekeeping-next' of https://github.com/Rust-for-Linux/linux.git rust-timekeeping/timekeeping-next +Already up to date. +Merging rust-xarray/xarray-next (c455f19bbe610 rust: xarray: add __rust_helper to helpers) +$ git merge -m Merge branch 'xarray-next' of https://github.com/Rust-for-Linux/linux.git rust-xarray/xarray-next +Already up to date. +Merging rust-analyzer/rust-analyzer-next (5f45afb8ab04d scripts: generate_rust_analyzer.py: pass cfg to macros crate) +$ git merge -m Merge branch 'rust-analyzer-next' of https://github.com/Rust-for-Linux/linux.git rust-analyzer/rust-analyzer-next +Auto-merging scripts/generate_rust_analyzer.py +Merge made by the 'ort' strategy. + scripts/generate_rust_analyzer.py | 1 + + 1 file changed, 1 insertion(+) +Merging mm/for-next (19e5fec518a91 Merge https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm.git mm-unstable into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git mm/for-next +Auto-merging arch/riscv/Kconfig +Auto-merging net/ipv4/tcp.c +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-kernel-mm-damon | 13 + + Documentation/admin-guide/cgroup-v2.rst | 4 + + Documentation/admin-guide/mm/damon/usage.rst | 18 +- + Documentation/admin-guide/mm/ksm.rst | 8 +- + Documentation/mm/damon/design.rst | 26 +- + Documentation/mm/page_owner.rst | 8 +- + arch/alpha/include/asm/pgtable.h | 7 - + arch/arc/include/asm/pgtable-levels.h | 11 - + arch/arm/Kconfig | 1 - + arch/arm/include/asm/pgtable.h | 7 - + arch/arm/kernel/traps.c | 17 - + arch/arm64/Kconfig | 1 - + arch/arm64/include/asm/pgtable.h | 15 - + arch/arm64/include/asm/set_memory.h | 5 +- + arch/arm64/mm/pageattr.c | 24 +- + arch/csky/include/asm/pgtable.h | 4 - + arch/hexagon/include/asm/pgtable.h | 3 - + arch/loongarch/Kconfig | 1 - + arch/loongarch/include/asm/pgtable.h | 13 - + arch/loongarch/include/asm/set_memory.h | 5 +- + arch/loongarch/mm/init.c | 4 +- + arch/loongarch/mm/pageattr.c | 27 +- + arch/m68k/include/asm/mcf_pgtable.h | 6 - + arch/m68k/include/asm/motorola_pgtable.h | 8 - + arch/m68k/include/asm/sun3_pgtable.h | 7 - + arch/microblaze/include/asm/pgtable.h | 7 - + arch/mips/include/asm/pgtable-32.h | 10 - + arch/mips/include/asm/pgtable-64.h | 13 - + arch/nios2/include/asm/pgtable.h | 7 - + arch/openrisc/include/asm/pgtable.h | 7 - + arch/parisc/include/asm/pgtable.h | 9 - + arch/parisc/kernel/pci-dma.c | 6 +- + arch/powerpc/include/asm/book3s/32/pgtable.h | 2 - + arch/powerpc/include/asm/book3s/64/pgtable.h | 7 - + arch/powerpc/include/asm/nohash/32/pgtable.h | 2 - + arch/powerpc/include/asm/nohash/64/pgtable-4k.h | 3 - + arch/powerpc/include/asm/nohash/64/pgtable.h | 5 - + arch/powerpc/platforms/powernv/Kconfig | 1 - + arch/powerpc/platforms/pseries/Kconfig | 1 - + arch/riscv/Kconfig | 1 - + arch/riscv/include/asm/page.h | 6 - + arch/riscv/include/asm/pgtable-64.h | 9 - + arch/riscv/include/asm/pgtable.h | 4 - + arch/riscv/include/asm/set_memory.h | 5 +- + arch/riscv/mm/pageattr.c | 23 +- + arch/s390/Kconfig | 1 - + arch/s390/include/asm/pgtable.h | 11 - + arch/s390/include/asm/set_memory.h | 5 +- + arch/s390/mm/pageattr.c | 20 +- + arch/sh/include/asm/pgtable-3level.h | 3 - + arch/sh/include/asm/pgtable_32.h | 13 - + arch/sh/mm/init.c | 16 +- + arch/sparc/include/asm/pgtable_32.h | 3 - + arch/sparc/include/asm/pgtable_64.h | 10 - + arch/um/include/asm/pgtable-2level.h | 7 - + arch/um/include/asm/pgtable-4level.h | 13 - + arch/x86/Kconfig | 3 - + arch/x86/include/asm/pgtable-2level.h | 5 - + arch/x86/include/asm/pgtable-3level.h | 11 - + arch/x86/include/asm/pgtable_64.h | 18 - + arch/x86/include/asm/set_memory.h | 5 +- + arch/x86/include/asm/string_64.h | 85 ++++- + arch/x86/mm/pat/set_memory.c | 20 +- + arch/xtensa/include/asm/pgtable.h | 4 - + drivers/android/binder/page_range.rs | 19 +- + drivers/android/binder_alloc.c | 63 ++-- + drivers/block/zram/zram_drv.c | 21 +- + fs/Kconfig | 1 - + fs/proc/base.c | 3 +- + fs/proc/internal.h | 2 - + fs/proc/task_mmu.c | 93 ----- + include/asm-generic/pgtable-nop4d.h | 1 - + include/asm-generic/pgtable-nopmd.h | 1 - + include/asm-generic/pgtable-nopud.h | 1 - + include/linux/damon.h | 53 ++- + include/linux/gfp_types.h | 2 +- + include/linux/hugetlb.h | 5 - + include/linux/memblock.h | 1 - + include/linux/memcontrol.h | 39 +- + include/linux/mm.h | 47 +-- + include/linux/mm_types.h | 8 +- + include/linux/mmap_lock.h | 94 +++-- + include/linux/mmzone.h | 96 ++--- + include/linux/page-flags.h | 3 +- + include/linux/page_counter.h | 1 + + include/linux/pgtable.h | 14 + + include/linux/set_memory.h | 12 +- + include/linux/shmem_fs.h | 12 +- + include/linux/string.h | 13 + + include/linux/swap.h | 6 + + include/net/mana/mana.h | 4 +- + include/trace/events/huge_memory.h | 18 +- + include/trace/events/vmscan.h | 14 - + kernel/bpf/stackmap.c | 17 +- + kernel/bpf/task_iter.c | 2 +- + kernel/fork.c | 2 - + kernel/power/snapshot.c | 4 +- + lib/codetag.c | 10 +- + mm/Kconfig | 17 - + mm/Kconfig.debug | 1 - + mm/alloc_tag.c | 14 +- + mm/cma.h | 23 +- + mm/damon/core.c | 169 +++++++-- + mm/damon/ops-common.c | 30 +- + mm/damon/paddr.c | 49 ++- + mm/damon/sysfs-schemes.c | 4 + + mm/damon/sysfs.c | 289 ++++++++++++++- + mm/damon/tests/core-kunit.h | 85 ++++- + mm/damon/vaddr.c | 10 +- + mm/debug.c | 4 - + mm/execmem.c | 42 +-- + mm/gup.c | 185 +++++----- + mm/gup_test.c | 2 +- + mm/huge_memory.c | 67 ++-- + mm/hugetlb.c | 139 +++++--- + mm/hugetlb_sysfs.c | 10 +- + mm/hugetlb_vmemmap.c | 103 +----- + mm/hugetlb_vmemmap.h | 15 +- + mm/init-mm.c | 2 - + mm/internal.h | 16 - + mm/khugepaged.c | 39 +- + mm/ksm.c | 7 +- + mm/list_lru.c | 9 +- + mm/madvise.c | 3 + + mm/memblock.c | 16 +- + mm/memcontrol-v1.c | 392 +-------------------- + mm/memcontrol-v1.h | 12 +- + mm/memcontrol.c | 161 ++++++--- + mm/memory.c | 93 ++--- + mm/mempolicy.c | 170 +++++---- + mm/mincore.c | 2 +- + mm/mm_init.c | 165 +++++---- + mm/mm_init.h | 1 + + mm/mmap_lock.c | 61 ++-- + mm/oom_kill.c | 15 +- + mm/page_alloc.c | 32 +- + mm/page_counter.c | 20 ++ + mm/page_io.c | 38 +- + mm/page_isolation.c | 40 ++- + mm/page_owner.c | 2 +- + mm/page_table_check.c | 6 +- + mm/pagewalk.c | 2 - + mm/percpu.c | 9 +- + mm/pgtable-generic.c | 20 +- + mm/rmap.c | 9 +- + mm/secretmem.c | 6 +- + mm/shmem.c | 391 +++++++++++++++----- + mm/slab_common.c | 2 +- + mm/slub.c | 7 +- + mm/sparse-vmemmap.c | 224 ++++++------ + mm/sparse.c | 94 +---- + mm/sparse.h | 85 +++++ + mm/swap.h | 3 +- + mm/swap_state.c | 4 +- + mm/swapfile.c | 43 ++- + mm/userfaultfd.c | 65 +--- + mm/vmalloc.c | 48 ++- + mm/vmalloc.h | 2 +- + mm/vmpressure.c | 3 - + mm/vmscan.c | 190 ++++------ + mm/vmstat.c | 17 +- + mm/zswap.c | 21 +- + net/ipv4/tcp.c | 31 +- + rust/kernel/mm.rs | 58 +-- + samples/damon/mtier.c | 5 + + samples/damon/prcl.c | 5 +- + samples/damon/wsse.c | 5 +- + scripts/gdb/linux/mm.py | 14 +- + tools/include/linux/mm.h | 4 - + tools/mm/page_owner_sort.c | 134 ++++++- + tools/testing/selftests/cgroup/test_zswap.c | 13 +- + tools/testing/selftests/damon/_damon_sysfs.py | 38 +- + tools/testing/selftests/damon/sysfs.py | 4 + + tools/testing/selftests/damon/sysfs.sh | 27 ++ + tools/testing/selftests/mm/.gitignore | 2 + + tools/testing/selftests/mm/hugetlb-soft-offline.c | 49 +-- + tools/testing/selftests/mm/khugepaged.c | 22 +- + tools/testing/selftests/mm/mremap_test.c | 44 ++- + tools/testing/selftests/mm/pkey-helpers.h | 7 - + tools/testing/selftests/mm/pkey_sighandler_tests.c | 3 +- + tools/testing/selftests/mm/uffd-wp-mremap.c | 2 + + tools/testing/vma/include/dup.h | 5 +- + tools/testing/vma/vma_internal.h | 1 - + 183 files changed, 2803 insertions(+), 2781 deletions(-) +Merging mm-nonmm-stable/mm-nonmm-stable (786262be6048d Merge tag 'edac_updates_for_v7.3_rc2' of git://git.kernel.org/pub/scm/linux/kernel/git/ras/ras) +$ git merge -m Merge branch 'mm-nonmm-stable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-stable/mm-nonmm-stable +Already up to date. +Merging mm-nonmm-unstable/mm-nonmm-unstable (aa30aa315f5f8 panic: remove the unneeded panic_print_get()) +$ git merge -m Merge branch 'mm-nonmm-unstable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-unstable/mm-nonmm-unstable +Auto-merging .mailmap +Auto-merging MAINTAINERS +Auto-merging arch/s390/Kconfig +Auto-merging drivers/usb/gadget/legacy/inode.c +Auto-merging fs/proc/base.c +Auto-merging fs/ufs/dir.c +Auto-merging init/main.c +Auto-merging kernel/fork.c +Merge made by the 'ort' strategy. + .mailmap | 1 + + Documentation/admin-guide/sysctl/kernel.rst | 5 +- + MAINTAINERS | 1 + + arch/Kconfig | 8 - + arch/alpha/include/uapi/asm/setup.h | 4 + + arch/arc/include/asm/setup.h | 2 +- + arch/arm/include/uapi/asm/setup.h | 6 +- + arch/arm64/include/uapi/asm/setup.h | 4 + + arch/loongarch/include/uapi/asm/setup.h | 4 + + arch/m68k/include/uapi/asm/setup.h | 6 +- + arch/microblaze/include/uapi/asm/setup.h | 4 + + arch/mips/include/uapi/asm/setup.h | 4 + + arch/parisc/include/uapi/asm/setup.h | 4 + + arch/powerpc/include/uapi/asm/setup.h | 4 + + arch/riscv/include/uapi/asm/setup.h | 4 + + arch/s390/Kconfig | 8 - + arch/s390/kernel/traps.c | 7 + + arch/sparc/include/uapi/asm/setup.h | 10 +- + arch/sparc/kernel/setup.c | 9 + + arch/um/include/asm/setup.h | 2 +- + arch/x86/include/asm/setup.h | 2 +- + arch/xtensa/include/uapi/asm/setup.h | 4 + + drivers/infiniband/hw/hfi1/fault.c | 1 - + drivers/media/v4l2-core/v4l2-vp9.c | 30 +-- + drivers/usb/gadget/legacy/inode.c | 4 +- + fs/fat/fat.h | 2 +- + fs/fat/fatent.c | 21 +- + fs/fat/file.c | 3 +- + fs/fat/misc.c | 6 +- + fs/ocfs2/alloc.c | 27 +++ + fs/ocfs2/dir.c | 35 +++ + fs/ocfs2/inode.c | 53 ++++- + fs/ocfs2/journal.c | 3 +- + fs/ocfs2/journal.h | 1 - + fs/ocfs2/ocfs2.h | 17 ++ + fs/ocfs2/refcounttree.c | 27 +++ + fs/ocfs2/suballoc.c | 235 +++++++++++++++++--- + fs/ocfs2/suballoc.h | 2 +- + fs/ocfs2/super.c | 32 ++- + fs/ocfs2/xattr.c | 157 +++++++++++--- + fs/proc/base.c | 8 +- + fs/squashfs/fragment.c | 6 +- + fs/squashfs/squashfs_fs.h | 2 +- + fs/ufs/dir.c | 2 +- + include/linux/fault-inject.h | 10 +- + include/linux/minmax.h | 6 +- + include/linux/panic.h | 2 + + include/uapi/asm-generic/setup.h | 4 + + init/Kconfig | 19 +- + init/main.c | 8 +- + init/version.c | 11 +- + ipc/mqueue.c | 7 + + kernel/fork.c | 3 +- + kernel/gcov/fs.c | 2 +- + kernel/hung_task.c | 55 ++++- + kernel/panic.c | 22 +- + kernel/resource.c | 2 +- + kernel/taskstats.c | 51 ++--- + lib/Kconfig | 14 +- + lib/Kconfig.debug | 15 ++ + lib/decompress_unxz.c | 12 +- + lib/dynamic_debug.c | 15 +- + lib/fault-inject.c | 7 +- + lib/group_cpus.c | 87 +++++++- + lib/klist.c | 16 +- + lib/raid/Kconfig | 2 + + lib/raid/raid6/x86/avx2.c | 6 + + lib/raid/raid6/x86/avx512.c | 6 + + lib/raid/raid6/x86/recov_avx2.c | 2 + + lib/raid/raid6/x86/recov_avx512.c | 2 + + lib/raid/xor/x86/xor-avx.c | 1 + + lib/tests/Makefile | 1 + + lib/tests/errseq_kunit.c | 237 +++++++++++++++++++++ + scripts/checkpatch.pl | 2 + + tools/testing/selftests/core/unshare_test.c | 19 +- + .../filesystems/epoll/epoll_wakeup_test.c | 32 ++- + .../selftests/membarrier/membarrier_test_impl.h | 17 +- + 77 files changed, 1211 insertions(+), 261 deletions(-) + create mode 100644 lib/tests/errseq_kunit.c +Merging kbuild/kbuild-for-next (51794b107d54b scripts/gcc-plugins: suppress recorded GCC switches for extmod builds) +$ git merge -m Merge branch 'kbuild-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git kbuild/kbuild-for-next +Merge made by the 'ort' strategy. + Makefile | 5 ++++- + scripts/Makefile.gcc-plugins | 4 ++++ + scripts/kconfig/symbol.c | 5 ++++- + scripts/kconfig/tests/err_recursive_dep/Kconfig | 24 ++++++++++++++++++++++ + .../tests/err_recursive_dep/expected_stderr | 13 ++++++++++++ + scripts/mkcompile_h | 2 +- + scripts/setlocalversion | 3 +-- + 7 files changed, 51 insertions(+), 5 deletions(-) +Merging clang-format/clang-format (8f0b4cce4481f Linux 6.19-rc1) +$ git merge -m Merge branch 'clang-format' of https://github.com/ojeda/linux.git clang-format/clang-format +Already up to date. +Merging perf/perf-tools-next (92d50319b4f0c perf jitdump: Validate unwinding sizes against record payload) +$ git merge -m Merge branch 'perf-tools-next' of https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools-next.git perf/perf-tools-next +Merge made by the 'ort' strategy. + tools/perf/util/jitdump.c | 126 +++++++++++++++++++++++++++++++++++++++------- + 1 file changed, 108 insertions(+), 18 deletions(-) +Merging compiler-attributes/compiler-attributes (8f0b4cce4481f Linux 6.19-rc1) +$ git merge -m Merge branch 'compiler-attributes' of https://github.com/ojeda/linux.git compiler-attributes/compiler-attributes +Already up to date. +Merging dma-mapping/dma-mapping-for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'dma-mapping-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git dma-mapping/dma-mapping-for-next +Already up to date. +Merging asm-generic/master (adbbd9714f805 scripts: headers_install.sh: Remove config leak ignore machinery) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/arnd/asm-generic asm-generic/master +Already up to date. +Merging alpha/alpha-next (d58041d2c63e0 MAINTAINERS: Add Magnus Lindholm as maintainer for alpha port) +$ git merge -m Merge branch 'alpha-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mattst88/alpha.git alpha/alpha-next +Already up to date. +Merging arm/for-next (1a89abc009cb5 Merge branches 'fixes' and 'misc' into for-linus) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git arm/for-next +Already up to date. +Merging arm64/for-next/core (2bd533739234d selftests/arm64: Add MTE test config fragment) +$ git merge -m Merge branch 'for-next/core' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux arm64/for-next/core +Already up to date. +Merging arm-perf/for-next/perf (5936245125f78 perf/arm-cmn: Fix DVM node events) +$ git merge -m Merge branch 'for-next/perf' of https://git.kernel.org/pub/scm/linux/kernel/git/will/linux.git arm-perf/for-next/perf +Already up to date. +Merging arm-soc/for-next (ae77d827fb0e6 soc: document merges) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git arm-soc/for-next +Merge made by the 'ort' strategy. + arch/arm/arm-soc-for-next-contents.txt | 175 +++++++++++++++++++++++++++++++++ + 1 file changed, 175 insertions(+) + create mode 100644 arch/arm/arm-soc-for-next-contents.txt +Merging amlogic/for-next (87e20720aaadf Merge branch 'v7.4/arm64-dt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux.git amlogic/for-next +Merge made by the 'ort' strategy. + .../boot/dts/amlogic/amlogic-t7-a311d2-an400.dts | 2 +- + .../dts/amlogic/amlogic-t7-a311d2-khadas-vim4.dts | 113 ++++++++++++++++++++- + arch/arm64/boot/dts/amlogic/amlogic-t7.dtsi | 44 ++++++-- + .../dts/amlogic/meson-g12b-odroid-go-ultra.dts | 2 +- + .../boot/dts/amlogic/meson-gxbb-nanopi-k2.dts | 2 +- + arch/arm64/boot/dts/amlogic/meson-gxbb.dtsi | 2 +- + arch/arm64/boot/dts/amlogic/meson-gxl.dtsi | 2 +- + 7 files changed, 153 insertions(+), 14 deletions(-) +Merging asahi-soc/asahi-soc/for-next (3c709d42fc7a4 Merge branch 'apple-soc/drivers-7.4' into asahi-soc/for-next) +$ git merge -m Merge branch 'asahi-soc/for-next' of https://github.com/AsahiLinux/linux.git asahi-soc/asahi-soc/for-next +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/apple.yaml | 21 + + .../devicetree/bindings/arm/apple/apple,pmgr.yaml | 1 + + Documentation/devicetree/bindings/arm/cpus.yaml | 2 + + .../devicetree/bindings/i2c/apple,i2c.yaml | 1 + + .../bindings/interrupt-controller/apple,aic2.yaml | 1 + + .../devicetree/bindings/pinctrl/apple,pinctrl.yaml | 1 + + .../bindings/power/apple,pmgr-pwrstate.yaml | 1 + + .../devicetree/bindings/pwm/apple,s5l-fpwm.yaml | 1 + + arch/arm64/Kconfig.platforms | 1 + + arch/arm64/boot/dts/apple/Makefile | 6 + + arch/arm64/boot/dts/apple/t602x-common.dtsi | 2 +- + arch/arm64/boot/dts/apple/t6030.dtsi | 2 +- + arch/arm64/boot/dts/apple/t6032.dtsi | 2 +- + arch/arm64/boot/dts/apple/t8132-j604.dts | 35 + + arch/arm64/boot/dts/apple/t8132-j623.dts | 18 + + arch/arm64/boot/dts/apple/t8132-j624.dts | 18 + + arch/arm64/boot/dts/apple/t8132-j713.dts | 35 + + arch/arm64/boot/dts/apple/t8132-j715.dts | 35 + + arch/arm64/boot/dts/apple/t8132-j773g.dts | 25 + + arch/arm64/boot/dts/apple/t8132-jxxx.dtsi | 48 + + arch/arm64/boot/dts/apple/t8132-pmgr.dtsi | 1130 ++++++++++++++++++++ + arch/arm64/boot/dts/apple/t8132.dtsi | 467 ++++++++ + 22 files changed, 1850 insertions(+), 3 deletions(-) + create mode 100644 arch/arm64/boot/dts/apple/t8132-j604.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j623.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j624.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j713.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j715.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j773g.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-jxxx.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8132-pmgr.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8132.dtsi +Merging at91/at91-next (d6e7bce8d4392 Merge branch 'microchip-dt64' into at91-next) +$ git merge -m Merge branch 'at91-next' of https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git at91/at91-next +Merge made by the 'ort' strategy. +Merging bmc/for-next (cd7d1ef7d74ed Merge branches 'aspeed/drivers', 'aspeed/arm/dt', 'aspeed/fixes/drivers', 'aspeed/maintainers', 'nuvoton/arm/dt', 'nuvoton/arm/fixes' and 'nuvoton/arm64/dt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux.git bmc/for-next +Merge made by the 'ort' strategy. +Merging broadcom/next (fca5276165993 ARM: dts: BCM5301X: panamera: add phy-mode to switch) +$ git merge -m Merge branch 'next' of https://github.com/Broadcom/stblinux.git broadcom/next +Auto-merging arch/arm/boot/dts/broadcom/bcm-ns.dtsi +Auto-merging arch/arm/boot/dts/broadcom/bcm4709-linksys-ea9200.dts +Auto-merging arch/arm/boot/dts/broadcom/bcm47094-linksys-panamera.dts +Merge made by the 'ort' strategy. + arch/arm/boot/dts/broadcom/bcm-ns.dtsi | 13 +++++ + .../dts/broadcom/bcm4708-linksys-ea6300-v1.dts | 4 ++ + .../boot/dts/broadcom/bcm4708-smartrg-sr400ac.dts | 18 +++++++ + .../boot/dts/broadcom/bcm4709-linksys-ea9200.dts | 57 ++++++++++++++++++++++ + .../boot/dts/broadcom/bcm4709-netgear-r8000.dts | 12 +++++ + .../boot/dts/broadcom/bcm47094-dlink-dir-890l.dts | 6 ++- + .../dts/broadcom/bcm47094-linksys-panamera.dts | 1 + + arch/arm/boot/dts/broadcom/bcm47189-tenda-ac9.dts | 20 ++++++++ + arch/arm/boot/dts/broadcom/bcm7445.dtsi | 2 +- + 9 files changed, 131 insertions(+), 2 deletions(-) +Merging cix/for-next (a0cffbd8878c5 Merge remote-tracking branch 'cix/dt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/peter.chen/cix.git cix/for-next +Merge made by the 'ort' strategy. +Merging davinci/davinci/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'davinci/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git davinci/davinci/for-next +Already up to date. +Merging drivers-memory/for-next (c8d0e87bc2413 memory: renesas-rpc-if: Use runtime PM autosuspend after transfers) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git drivers-memory/for-next +Merge made by the 'ort' strategy. + drivers/memory/brcmstb_dpfe.c | 24 +++++++++++- + drivers/memory/renesas-rpc-if.c | 87 +++++++++++++++++------------------------ + 2 files changed, 58 insertions(+), 53 deletions(-) +Merging fsl/soc_fsl (e14b8d3938809 bus: fsl-mc: drop unused assignment of acpi_device_id::driver_data) +$ git merge -m Merge branch 'soc_fsl' of https://git.kernel.org/pub/scm/linux/kernel/git/chleroy/linux.git fsl/soc_fsl +Already up to date. +Merging imx-mxs/for-next (ff1bb9ccce43c Merge branches 'imx/dt', 'imx/dt64', 'imx/fixes' and 'imx/soc' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux.git imx-mxs/for-next +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/fsl.yaml | 14 +- + .../bindings/display/bridge/fsl,ldb.yaml | 23 +- + .../devicetree/bindings/display/fsl,lcdif.yaml | 1 + + .../bindings/soc/imx/fsl,imx-iomuxc-gpr.yaml | 62 ++ + arch/arm/boot/dts/nxp/imx/Makefile | 22 + + arch/arm/boot/dts/nxp/imx/e60k02.dtsi | 8 +- + arch/arm/boot/dts/nxp/imx/e70k02.dtsi | 8 +- + arch/arm/boot/dts/nxp/imx/imx6dl-mamoj.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-plym2m.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-prtvt7.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-victgo.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6q-bosch-acc.dts | 4 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-gw560x.dtsi | 20 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-skov-cpu.dtsi | 2 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-zii-rdu2.dtsi | 10 +- + arch/arm/boot/dts/nxp/imx/imx6qdl.dtsi | 10 +- + arch/arm/boot/dts/nxp/imx/imx6sl-kobo-aura2.dts | 8 +- + .../boot/dts/nxp/imx/imx6sl-tolino-shine2hd.dts | 8 +- + arch/arm/boot/dts/nxp/imx/imx6sll-evk.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul.dtsi | 2 +- + ...nel-cap-touch-7inch-parallel-touch-adapter.dtso | 58 ++ + ...olibri-emmc-panel-cap-touch-7inch-parallel.dtso | 43 ++ + ...olibri-emmc-panel-res-touch-7inch-parallel.dtso | 32 + + arch/arm/boot/dts/nxp/imx/imx7s-warp.dts | 2 +- + arch/arm/boot/dts/nxp/imx/mba6ulx.dtsi | 36 +- + arch/arm/boot/dts/nxp/ls/ls1021a.dtsi | 10 +- + arch/arm/boot/dts/nxp/vf/Makefile | 2 + + arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi | 126 ++++ + arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi | 56 ++ + arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts | 14 + + arch/arm/boot/dts/nxp/vf/vf500.dtsi | 6 + + arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts | 14 + + arch/arm/boot/dts/nxp/vf/vf610.dtsi | 3 + + arch/arm/include/asm/linkage.h | 29 + + arch/arm/mach-imx/hardware.h | 2 +- + arch/arm/mach-imx/mach-imx6q.c | 30 - + arch/arm/mach-imx/mxc.h | 2 +- + arch/arm/mach-imx/pm-imx6.c | 26 +- + arch/arm/mach-imx/suspend-imx6.S | 6 +- + arch/arm64/boot/dts/freescale/Makefile | 243 ++++++- + arch/arm64/boot/dts/freescale/fsl-ls1028a.dtsi | 2 +- + arch/arm64/boot/dts/freescale/fsl-lx216x.dtsi | 19 +- + .../arm64/boot/dts/freescale/imx8-apalis-eval.dtsi | 6 +- + .../boot/dts/freescale/imx8-apalis-ixora-v1.1.dtsi | 6 +- + .../boot/dts/freescale/imx8-apalis-ixora-v1.2.dtsi | 6 +- + .../arm64/boot/dts/freescale/imx8-apalis-v1.1.dtsi | 16 +- + .../dts/freescale/imx8mm-beacon-baseboard.dtsi | 2 +- + .../imx8mm-data-modul-edm-sbc-overlay-cm4.dtso | 56 ++ + ...odul-edm-sbc-overlay-edm-mod-imx8mm-common.dtsi | 59 ++ + ...-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtsi | 99 +++ + ...-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso | 79 +++ + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtsi | 69 ++ + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso | 59 ++ + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtsi | 94 +++ + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso | 17 + + ...edm-sbc-overlay-edm-mod-imx8mm-lvds-common.dtsi | 118 ++++ + ...l-edm-sbc-overlay-edm-mod-imx8mm-lvds-dual.dtsi | 32 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtsi | 12 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtsi | 12 + + ...bc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtsi | 12 + + ...bc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtsi | 12 + + ...bc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso | 7 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtsi | 12 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso | 7 + + ...c-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtsi | 12 + + ...c-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso | 7 + + ...verlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtsi | 12 + + ...verlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso | 7 + + ...edm-sbc-overlay-edm-mod-imx8mm-lvds-single.dtsi | 20 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi | 22 + + ...odul-edm-sbc-overlay-edm-sbc-imx8mm-rev900.dtso | 18 + + ...imx8mm-data-modul-edm-sbc-overlay-lvds-3v3.dtsi | 19 + + ...imx8mm-data-modul-edm-sbc-overlay-lvds-5v0.dtsi | 19 + + ...mx8mm-data-modul-edm-sbc-overlay-lvds-dual.dtsi | 29 + + ...data-modul-edm-sbc-overlay-lvds-g070y2-l01.dtsi | 31 + + ...ata-modul-edm-sbc-overlay-lvds-g101ice-l01.dtsi | 31 + + ...ata-modul-edm-sbc-overlay-lvds-g121xce-l01.dtsi | 31 + + ...ata-modul-edm-sbc-overlay-lvds-g156hce-l01.dtsi | 31 + + ...data-modul-edm-sbc-overlay-lvds-g215hvn011.dtsi | 30 + + ...ta-modul-edm-sbc-overlay-lvds-mi0700a2t-30.dtsi | 31 + + ...modul-edm-sbc-overlay-lvds-mi1010z1t-1cp11.dtsi | 31 + + ...8mm-data-modul-edm-sbc-overlay-lvds-single.dtsi | 13 + + .../dts/freescale/imx8mm-data-modul-edm-sbc.dts | 8 +- + .../freescale/imx8mn-vhip4-evalboard-common.dtsi | 3 +- + .../dts/freescale/imx8mn-vhip4-evalboard-v1.dts | 8 + + .../dts/freescale/imx8mn-vhip4-evalboard-v2.dts | 28 +- + .../imx8mn-vhip4-overlay-eeprom-1000.dtso | 103 +++ + .../imx8mn-vhip4-overlay-eeprom-1100.dtso | 103 +++ + .../imx8mn-vhip4-overlay-eeprom-2000.dtso | 107 +++ + .../imx8mp-data-modul-edm-sbc-overlay-cm7.dtso | 57 ++ + ...-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso | 67 ++ + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso | 46 ++ + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso | 17 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso | 7 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso | 11 + + ...c-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso | 7 + + ...verlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso | 7 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi | 35 + + ...sbc-overlay-edm-sbc-imx8mp-lvds-g070y2-l01.dtso | 24 + + ...bc-overlay-edm-sbc-imx8mp-lvds-g101ice-l01.dtso | 24 + + ...bc-overlay-edm-sbc-imx8mp-lvds-g121xce-l01.dtso | 24 + + ...bc-overlay-edm-sbc-imx8mp-lvds-g156hce-l01.dtso | 32 + + ...sbc-overlay-edm-sbc-imx8mp-lvds-g215hvn011.dtso | 36 + + ...c-overlay-edm-sbc-imx8mp-lvds-mi0700a2t-30.dtso | 24 + + ...verlay-edm-sbc-imx8mp-lvds-mi1010z1t-1cp11.dtso | 24 + + ...edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev900.dtso | 41 ++ + ...edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev902.dtso | 14 + + ...-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds.dtsi | 79 +++ + ...odul-edm-sbc-overlay-edm-sbc-imx8mp-rev900.dtso | 97 +++ + ...odul-edm-sbc-overlay-edm-sbc-imx8mp-rev902.dtso | 69 ++ + .../dts/freescale/imx8mp-data-modul-edm-sbc.dts | 48 +- + .../dts/freescale/imx8mp-icore-mx8mp-edimm2.2.dts | 2 +- + .../boot/dts/freescale/imx8mp-icore-mx8mp.dtsi | 2 +- + .../imx8mp-phyboard-pollux-peb-av-10.dtsi | 5 +- + .../imx8mp-phyboard-pollux-peb-wlbt-07.dtso | 76 +++ + ...8mp-tqma8mpqs-mb-smarc-2-lvds0-tm070jvhg33.dtso | 2 +- + ...8mp-tqma8mpqs-mb-smarc-2-lvds1-tm070jvhg33.dtso | 2 +- + arch/arm64/boot/dts/freescale/imx8mp.dtsi | 2 +- + arch/arm64/boot/dts/freescale/imx8mq-evk.dts | 6 + + .../boot/dts/freescale/imx8mq-librem5-devkit.dts | 4 +- + arch/arm64/boot/dts/freescale/imx8mq-librem5.dtsi | 4 +- + arch/arm64/boot/dts/freescale/imx8mq-nitrogen.dts | 2 +- + .../arm64/boot/dts/freescale/imx8mq-zii-ultra.dtsi | 16 +- + arch/arm64/boot/dts/freescale/imx8mq.dtsi | 168 ++--- + arch/arm64/boot/dts/freescale/imx8qm-ss-conn.dtsi | 38 ++ + .../boot/dts/freescale/imx8qm-var-som-symphony.dts | 2 +- + arch/arm64/boot/dts/freescale/imx8qm-var-som.dtsi | 6 +- + .../freescale/imx91-9x9-qsb-tianma-tm050rdh03.dtso | 104 +++ + arch/arm64/boot/dts/freescale/imx91-9x9-qsb.dts | 51 ++ + arch/arm64/boot/dts/freescale/imx91.dtsi | 11 + + .../dts/freescale/imx93-11x11-frdm-common.dtsi | 680 +++++++++++++++++++ + arch/arm64/boot/dts/freescale/imx93-11x11-frdm.dts | 679 +------------------ + arch/arm64/boot/dts/freescale/imx93w-frdm.dts | 23 + + arch/arm64/boot/dts/freescale/imx94.dtsi | 2 +- + .../boot/dts/freescale/imx943-evk-sdwifi.dtso | 25 + + arch/arm64/boot/dts/freescale/imx943-evk.dts | 55 +- + arch/arm64/boot/dts/freescale/imx943.dtsi | 2 +- + arch/arm64/boot/dts/freescale/imx95-15x15-frdm.dts | 24 +- + arch/arm64/boot/dts/freescale/imx95-19x19-evk.dts | 55 +- + arch/arm64/boot/dts/freescale/imx95.dtsi | 12 +- + arch/arm64/boot/dts/freescale/imx952-frdm.dts | 9 + + arch/arm64/boot/dts/freescale/imx952-frdm.dtsi | 726 +++++++++++++++++++++ + arch/arm64/boot/dts/freescale/imx952.dtsi | 8 +- + arch/arm64/boot/dts/freescale/s32g2.dtsi | 2 +- + arch/arm64/boot/dts/freescale/s32g3.dtsi | 2 +- + include/linux/micrel_phy.h | 5 - + include/soc/imx/cpu.h | 2 +- + 153 files changed, 5155 insertions(+), 1010 deletions(-) + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel-touch-adapter.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-res-touch-7inch-parallel.dtso + create mode 100644 arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi + create mode 100644 arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts + create mode 100644 arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-cm4.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-dual.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-single.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-sbc-imx8mm-rev900.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-3v3.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-5v0.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-dual.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g070y2-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g101ice-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g121xce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g156hce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g215hvn011.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-mi0700a2t-30.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-mi1010z1t-1cp11.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-single.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mn-vhip4-overlay-eeprom-1000.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mn-vhip4-overlay-eeprom-1100.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mn-vhip4-overlay-eeprom-2000.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-cm7.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g070y2-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g101ice-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g121xce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g156hce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g215hvn011.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-mi0700a2t-30.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-mi1010z1t-1cp11.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev900.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev902.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-rev900.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-rev902.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-phyboard-pollux-peb-wlbt-07.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx91-9x9-qsb-tianma-tm050rdh03.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx93-11x11-frdm-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93w-frdm.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx952-frdm.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx952-frdm.dtsi +Merging mediatek/for-next (f5be25e697e03 Merge branch 'v7.2-next/dts32' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux.git mediatek/for-next +Auto-merging drivers/soc/mediatek/mt8167-mmsys.h +CONFLICT (content): Merge conflict in drivers/soc/mediatek/mt8167-mmsys.h +Resolved 'drivers/soc/mediatek/mt8167-mmsys.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 90105d2868eb9] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux.git +$ git diff -M --stat --summary HEAD^.. + drivers/soc/mediatek/mt8167-mmsys.h | 5 ----- + 1 file changed, 5 deletions(-) +$ git am -3 ../patches/0001-mediatek-Take-the-other-branch.patch +Applying: mediatek: Take the other branch +$ git reset HEAD^ +Unstaged changes after reset: +M drivers/soc/mediatek/mt8167-mmsys.h +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master fe9458722834f] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux.git + Date: Thu Sep 3 13:54:41 2026 +0100 +Merging mvebu/for-next (1dd243147c2fa ARM: dts: marvell: Replace spaces indentation with tabs) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gclement/mvebu.git mvebu/for-next +Merge made by the 'ort' strategy. + .../boot/dts/marvell/armada-385-clearfog-gtr.dtsi | 2 +- + arch/arm/boot/dts/marvell/armada-388-helios4.dts | 23 +++++++++------------- + arch/arm/boot/dts/marvell/armada-38x.dtsi | 1 + + arch/arm/boot/dts/marvell/armada-395-gp.dts | 1 - + arch/arm/boot/dts/marvell/armada-39x.dtsi | 1 + + arch/arm/boot/dts/marvell/dove.dtsi | 22 ++++++++++----------- + arch/arm/boot/dts/marvell/kirkwood-6282.dtsi | 15 +++++++------- + .../dts/marvell/kirkwood-guruplug-server-plus.dts | 4 ++-- + arch/arm/boot/dts/marvell/kirkwood-lsxl.dtsi | 4 ++-- + arch/arm/boot/dts/marvell/kirkwood-synology.dtsi | 12 +++++------ + .../arm/boot/dts/marvell/orion5x-rd88f5182-nas.dts | 2 +- + 11 files changed, 41 insertions(+), 46 deletions(-) +Merging omap/for-next (e1743a4a9b0b1 Merge branch 'omap-for-v7.4/soc' into tmp/omap-next-20260831.110627) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git omap/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/input/omap-keypad.txt | 28 ---------- + .../devicetree/bindings/input/ti,omap4-keypad.yaml | 63 +++++++++++++++++++++ + .../boot/dts/ti/omap/am335x-boneblack-hdmi.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/am335x-shc.dts | 4 -- + arch/arm/boot/dts/ti/omap/am437x-l4.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/am57-pruss.dtsi | 11 ++++ + arch/arm/boot/dts/ti/omap/am571x-idk.dts | 65 +++++++++++++++++++++- + arch/arm/boot/dts/ti/omap/am5729-beagleboneai.dts | 2 +- + .../boot/dts/ti/omap/am57xx-beagle-x15-common.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/am57xx-idk-common.dtsi | 3 +- + arch/arm/boot/dts/ti/omap/dra7-l4.dtsi | 2 +- + .../boot/dts/ti/omap/motorola-mapphone-common.dtsi | 8 +-- + .../dts/ti/omap/motorola-mapphone-mz607-mz617.dtsi | 2 + + arch/arm/boot/dts/ti/omap/omap3-igep.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/omap3-n9.dts | 5 +- + arch/arm/boot/dts/ti/omap/omap3-n900.dts | 3 +- + arch/arm/boot/dts/ti/omap/omap3-n950.dts | 5 +- + arch/arm/boot/dts/ti/omap/omap3-tao3530.dtsi | 8 +-- + .../boot/dts/ti/omap/omap4-droid-bionic-xt875.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4-droid4-xt894.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4-epson-embt2ws.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4-l4.dtsi | 1 + + .../dts/ti/omap/omap4-samsung-espresso-common.dtsi | 6 +- + arch/arm/boot/dts/ti/omap/omap4-sdp.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4.dtsi | 6 +- + arch/arm/boot/dts/ti/omap/omap5-l4.dtsi | 1 + + arch/arm/boot/dts/ti/omap/omap5.dtsi | 2 +- + arch/arm/configs/omap2plus_defconfig | 3 + + arch/arm/mach-omap2/control.h | 8 +-- + arch/arm/mach-omap2/soc.h | 4 +- + arch/arm/mach-omap2/sram.h | 4 +- + 31 files changed, 193 insertions(+), 67 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/input/omap-keypad.txt + create mode 100644 Documentation/devicetree/bindings/input/ti,omap4-keypad.yaml +Merging qcom/for-next (e61cfc881090c Merge branches 'arm64-defconfig-for-7.4', 'arm64-fixes-for-7.3', 'arm64-for-7.4', 'clk-fixes-for-7.3', 'clk-for-7.4', 'drivers-fixes-for-7.3' and 'drivers-for-7.4' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux.git qcom/for-next +Auto-merging Documentation/devicetree/bindings/arm/cpus.yaml +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/cpus.yaml | 1 + + .../bindings/arm/qcom,coresight-ctcu.yaml | 1 + + Documentation/devicetree/bindings/arm/qcom.yaml | 2 + + .../devicetree/bindings/clock/qcom,hawi-gpucc.yaml | 80 + + .../bindings/clock/qcom,kaanapali-gxclkctl.yaml | 23 +- + .../devicetree/bindings/clock/qcom,kuno-gcc.yaml | 52 + + .../bindings/clock/qcom,milos-camcc.yaml | 39 +- + .../bindings/clock/qcom,milos-videocc.yaml | 29 +- + .../devicetree/bindings/clock/qcom,rpmhcc.yaml | 1 + + .../devicetree/bindings/clock/qcom,sm6375-gcc.yaml | 9 +- + .../devicetree/bindings/clock/qcom,sm7150-gcc.yaml | 53 - + .../bindings/clock/qcom,sm8450-camcc.yaml | 2 + + .../bindings/clock/qcom,sm8450-gpucc.yaml | 3 + + .../bindings/clock/qcom,sm8450-videocc.yaml | 2 + + .../bindings/clock/qcom,sm8550-tcsr.yaml | 1 - + .../bindings/clock/qcom,x1e80100-tcsr.yaml | 114 + + .../devicetree/bindings/firmware/qcom,scm.yaml | 1 + + .../bindings/interconnect/qcom,msm8998-bwmon.yaml | 1 + + .../bindings/net/wireless/qcom,ath11k.yaml | 16 +- + arch/arm64/boot/dts/qcom/Makefile | 16 + + arch/arm64/boot/dts/qcom/agatti.dtsi | 76 +- + arch/arm64/boot/dts/qcom/eliza-evk.dtsi | 221 ++ + arch/arm64/boot/dts/qcom/eliza-mtp.dts | 8 + + arch/arm64/boot/dts/qcom/eliza.dtsi | 697 ++++- + arch/arm64/boot/dts/qcom/glymur-crd.dts | 8 + + arch/arm64/boot/dts/qcom/glymur-crd.dtsi | 3 +- + .../boot/dts/qcom/glymur-hp-elitebook-x-g2q.dts | 940 +++++++ + arch/arm64/boot/dts/qcom/glymur.dtsi | 385 ++- + arch/arm64/boot/dts/qcom/hamoa-iot-evk.dts | 134 +- + arch/arm64/boot/dts/qcom/hamoa-iot-som.dtsi | 21 + + .../qcom/hamoa-lenovo-ideacentre-mini-01q8x10.dts | 21 + + arch/arm64/boot/dts/qcom/hamoa.dtsi | 21 +- + arch/arm64/boot/dts/qcom/kodiak.dtsi | 1413 +++++----- + arch/arm64/boot/dts/qcom/lemans-pmics.dtsi | 116 + + arch/arm64/boot/dts/qcom/mahua-crd.dts | 5 + + arch/arm64/boot/dts/qcom/mahua.dtsi | 89 + + arch/arm64/boot/dts/qcom/milos.dtsi | 10 + + arch/arm64/boot/dts/qcom/monaco-pmics.dtsi | 62 + + arch/arm64/boot/dts/qcom/msm8917-xiaomi-riva.dts | 35 + + arch/arm64/boot/dts/qcom/purwa-iot-evk.dts | 133 +- + arch/arm64/boot/dts/qcom/purwa-iot-som.dtsi | 21 + + .../dts/qcom/qcs6490-rb3gen2-vision-mezzanine.dtso | 13 +- + .../qcs6490-thundercomm-rubikpi3-cam1-imx219.dtso | 115 + + .../qcs6490-thundercomm-rubikpi3-cam2-imx219.dtso | 115 + + .../boot/dts/qcom/qcs6490-thundercomm-rubikpi3.dts | 23 + + arch/arm64/boot/dts/qcom/sc8280xp-crd.dts | 3 + + .../boot/dts/qcom/sc8280xp-huawei-gaokun3.dts | 3 + + .../dts/qcom/sc8280xp-lenovo-thinkpad-x13s.dts | 37 + + .../boot/dts/qcom/sc8280xp-microsoft-arcata.dts | 3 + + .../boot/dts/qcom/sc8280xp-microsoft-blackrock.dts | 3 + + arch/arm64/boot/dts/qcom/sc8280xp.dtsi | 3 - + .../arm64/boot/dts/qcom/sdm845-oneplus-common.dtsi | 1 - + .../boot/dts/qcom/sdm845-sony-xperia-tama.dtsi | 2 + + arch/arm64/boot/dts/qcom/shikra-cqm-som.dtsi | 1 - + arch/arm64/boot/dts/qcom/shikra-evk.dtsi | 8 + + arch/arm64/boot/dts/qcom/shikra-iqs-som.dtsi | 1 - + arch/arm64/boot/dts/qcom/shikra.dtsi | 1490 +++++++++- + .../boot/dts/qcom/sm7325-nothing-spacewar.dts | 2 + + .../boot/dts/qcom/sm8650-ayaneo-pocket-s2.dts | 232 +- + arch/arm64/boot/dts/qcom/sm8750.dtsi | 68 + + arch/arm64/boot/dts/qcom/x1-asus-vivobook-s15.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-asus-zenbook-a14.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-crd.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-dell-thena.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-hp-omnibook-x14.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-microsoft-denali.dtsi | 57 +- + arch/arm64/boot/dts/qcom/x1e001de-devkit.dts | 21 + + .../dts/qcom/x1e78100-lenovo-thinkpad-t14s.dtsi | 21 + + .../boot/dts/qcom/x1e80100-dell-xps13-9345.dts | 21 + + .../dts/qcom/x1e80100-honor-magicbook-art-14.dts | 21 + + .../boot/dts/qcom/x1e80100-lenovo-yoga-slim7x.dts | 21 + + .../dts/qcom/x1e80100-medion-sprchrgd-14-s1.dts | 21 + + .../boot/dts/qcom/x1e80100-microsoft-romulus.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1e80100-qcp.dts | 21 + + .../boot/dts/qcom/x1p42100-lenovo-thinkbook-16.dts | 21 + + .../boot/dts/qcom/x1p42100-microsoft-sp12in.dts | 21 + + arch/arm64/configs/defconfig | 3 + + drivers/clk/qcom/Kconfig | 144 +- + drivers/clk/qcom/Makefile | 7 + + drivers/clk/qcom/cambistmclkcc-eliza.c | 464 +++ + drivers/clk/qcom/camcc-eliza.c | 2803 +++++++++++++++++++ + drivers/clk/qcom/camcc-nord.c | 2941 ++++++++++++++++++++ + drivers/clk/qcom/clk-alpha-pll.c | 9 + + drivers/clk/qcom/clk-rpmh.c | 15 + + drivers/clk/qcom/dispcc-sm8450.c | 40 +- + drivers/clk/qcom/gcc-eliza.c | 6 +- + drivers/clk/qcom/gcc-hawi.c | 6 +- + drivers/clk/qcom/gcc-kaanapali.c | 6 +- + drivers/clk/qcom/gcc-kuno.c | 1483 ++++++++++ + drivers/clk/qcom/gcc-msm8939.c | 4 + + drivers/clk/qcom/gcc-qcs8300.c | 103 + + drivers/clk/qcom/gpucc-eliza.c | 606 ++++ + drivers/clk/qcom/gpucc-glymur.c | 2 +- + drivers/clk/qcom/gpucc-hawi.c | 477 ++++ + drivers/clk/qcom/gpucc-kaanapali.c | 2 +- + drivers/clk/qcom/nwgcc-nord.c | 3 + + drivers/clk/qcom/tcsrcc-x1e80100.c | 337 +-- + drivers/clk/qcom/videocc-eliza.c | 404 +++ + drivers/clk/qcom/videocc-glymur.c | 38 + + drivers/firmware/qcom/qcom_scm.c | 144 +- + drivers/soc/qcom/Kconfig | 1 - + drivers/soc/qcom/apr.c | 75 +- + drivers/soc/qcom/llcc-qcom.c | 60 +- + drivers/soc/qcom/pmic_glink_altmode.c | 4 +- + drivers/soc/qcom/qcom-geni-se.c | 49 +- + drivers/soc/qcom/qcom_aoss.c | 4 +- + drivers/soc/qcom/rpmh-internal.h | 16 +- + drivers/soc/qcom/rpmh-rsc.c | 4 +- + drivers/soc/qcom/rpmh.c | 40 +- + drivers/soc/qcom/smem_dramc.c | 81 +- + drivers/soc/qcom/smp2p.c | 4 +- + drivers/soc/qcom/smsm.c | 20 +- + drivers/soc/qcom/socinfo.c | 1 + + drivers/soc/qcom/ubwc_config.c | 9 + + include/dt-bindings/arm/qcom,ids.h | 1 + + .../dt-bindings/clock/qcom,eliza-cambistmclkcc.h | 32 + + include/dt-bindings/clock/qcom,eliza-camcc.h | 151 + + include/dt-bindings/clock/qcom,eliza-gpucc.h | 51 + + include/dt-bindings/clock/qcom,eliza-videocc.h | 37 + + include/dt-bindings/clock/qcom,hawi-gpucc.h | 47 + + include/dt-bindings/clock/qcom,kuno-gcc.h | 100 + + include/dt-bindings/clock/qcom,nord-camcc.h | 167 ++ + include/dt-bindings/clock/qcom,nord-nwgcc.h | 3 + + include/dt-bindings/clock/qcom,qcs8300-gcc.h | 6 + + include/linux/device-id/apr.h | 21 - + include/linux/firmware/qcom/qcom_scm.h | 29 - + include/linux/mod_devicetable.h | 1 - + include/linux/soc/qcom/apr.h | 4 +- + include/linux/soc/qcom/llcc-qcom.h | 4 + + include/linux/soc/qcom/ubwc.h | 3 + + 130 files changed, 16842 insertions(+), 1590 deletions(-) + create mode 100644 Documentation/devicetree/bindings/clock/qcom,hawi-gpucc.yaml + create mode 100644 Documentation/devicetree/bindings/clock/qcom,kuno-gcc.yaml + delete mode 100644 Documentation/devicetree/bindings/clock/qcom,sm7150-gcc.yaml + create mode 100644 Documentation/devicetree/bindings/clock/qcom,x1e80100-tcsr.yaml + create mode 100644 arch/arm64/boot/dts/qcom/glymur-hp-elitebook-x-g2q.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs6490-thundercomm-rubikpi3-cam1-imx219.dtso + create mode 100644 arch/arm64/boot/dts/qcom/qcs6490-thundercomm-rubikpi3-cam2-imx219.dtso + create mode 100644 drivers/clk/qcom/cambistmclkcc-eliza.c + create mode 100644 drivers/clk/qcom/camcc-eliza.c + create mode 100644 drivers/clk/qcom/camcc-nord.c + create mode 100644 drivers/clk/qcom/gcc-kuno.c + create mode 100644 drivers/clk/qcom/gpucc-eliza.c + create mode 100644 drivers/clk/qcom/gpucc-hawi.c + create mode 100644 drivers/clk/qcom/videocc-eliza.c + create mode 100644 include/dt-bindings/clock/qcom,eliza-cambistmclkcc.h + create mode 100644 include/dt-bindings/clock/qcom,eliza-camcc.h + create mode 100644 include/dt-bindings/clock/qcom,eliza-gpucc.h + create mode 100644 include/dt-bindings/clock/qcom,eliza-videocc.h + create mode 100644 include/dt-bindings/clock/qcom,hawi-gpucc.h + create mode 100644 include/dt-bindings/clock/qcom,kuno-gcc.h + create mode 100644 include/dt-bindings/clock/qcom,nord-camcc.h + delete mode 100644 include/linux/device-id/apr.h +Merging realtek/for-next (3c778f0c9fa36 arm64: dts: realtek: Add GPIO support for RTD1625) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git realtek/for-next +Already up to date. +Merging renesas/next (288e04e37c645 Merge branches 'renesas-drivers-for-v7.4' and 'renesas-dts-for-v7.4' into renesas-next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git renesas/next +Merge made by the 'ort' strategy. + arch/arm/boot/dts/renesas/r7s72100-gr-peach.dts | 2 + + .../boot/dts/renesas/r8a7740-armadillo800eva.dts | 2 + + arch/arm/boot/dts/renesas/r8a7743-sk-rzg1m.dts | 2 + + arch/arm/boot/dts/renesas/r8a7743.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a7744.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a7745-sk-rzg1e.dts | 2 + + arch/arm/boot/dts/renesas/r8a7745.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a77470.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a7790-lager.dts | 28 +- + arch/arm/boot/dts/renesas/r8a7790-stout.dts | 4 +- + arch/arm/boot/dts/renesas/r8a7790.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7791-koelsch.dts | 2 + + arch/arm/boot/dts/renesas/r8a7791-porter.dts | 2 + + arch/arm/boot/dts/renesas/r8a7791.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7792.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7793-gose.dts | 2 + + arch/arm/boot/dts/renesas/r8a7793.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7794-alt.dts | 2 + + arch/arm/boot/dts/renesas/r8a7794-silk.dts | 2 + + arch/arm/boot/dts/renesas/r8a7794.dtsi | 4 +- + .../arm/boot/dts/renesas/r9a06g032-rzn1d400-db.dts | 1 - + .../arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts | 14 +- + arch/arm/boot/dts/renesas/r9a06g032.dtsi | 2 - + arch/arm64/boot/dts/renesas/Makefile | 6 + + .../dts/renesas/aistarvision-mipi-adapter-2.1.dtsi | 2 +- + .../boot/dts/renesas/beacon-renesom-baseboard.dtsi | 6 +- + .../arm64/boot/dts/renesas/beacon-renesom-som.dtsi | 12 +- + arch/arm64/boot/dts/renesas/cat875.dtsi | 2 + + arch/arm64/boot/dts/renesas/condor-common.dtsi | 2 +- + arch/arm64/boot/dts/renesas/draak.dtsi | 2 +- + arch/arm64/boot/dts/renesas/ebisu.dtsi | 6 +- + arch/arm64/boot/dts/renesas/gray-hawk-single.dtsi | 8 +- + arch/arm64/boot/dts/renesas/hihope-common.dtsi | 4 +- + arch/arm64/boot/dts/renesas/hihope-rev2.dtsi | 8 +- + arch/arm64/boot/dts/renesas/hihope-rev4.dtsi | 4 +- + arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi | 6 +- + arch/arm64/boot/dts/renesas/r8a774a1.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a774b1.dtsi | 12 +- + arch/arm64/boot/dts/renesas/r8a774c0-cat874.dts | 4 +- + arch/arm64/boot/dts/renesas/r8a774c0.dtsi | 12 +- + arch/arm64/boot/dts/renesas/r8a774e1.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77951.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77960.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77961.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77965.dtsi | 12 +- + .../renesas/r8a77970-eagle-function-expansion.dtso | 2 +- + arch/arm64/boot/dts/renesas/r8a77970-eagle.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a77970-v3msk.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a77970.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a77980-v3hsk.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a77980.dtsi | 4 +- + arch/arm64/boot/dts/renesas/r8a77990.dtsi | 10 +- + arch/arm64/boot/dts/renesas/r8a77995.dtsi | 6 +- + .../boot/dts/renesas/r8a779a0-falcon-cpu.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a779a0-falcon.dts | 4 +- + arch/arm64/boot/dts/renesas/r8a779a0.dtsi | 2 +- + .../boot/dts/renesas/r8a779f0-spider-cpu.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a779f0.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a779f4-s4sk.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a779g0.dtsi | 35 ++- + .../renesas/r8a779g3-sparrow-hawk-fan-argon40.dtso | 1 - + .../boot/dts/renesas/r8a779g3-sparrow-hawk.dts | 2 + + arch/arm64/boot/dts/renesas/r8a779h0.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a779md-geist.dts | 12 +- + arch/arm64/boot/dts/renesas/r8a78000.dtsi | 300 +++++++++++++++++++-- + .../arm64/boot/dts/renesas/r9a07g044l2-remi-pi.dts | 2 +- + arch/arm64/boot/dts/renesas/r9a08g045.dtsi | 5 +- + arch/arm64/boot/dts/renesas/r9a08g046.dtsi | 126 +++++++++ + arch/arm64/boot/dts/renesas/r9a09g011.dtsi | 4 +- + arch/arm64/boot/dts/renesas/r9a09g047.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r9a09g047e57-smarc.dts | 2 + + arch/arm64/boot/dts/renesas/r9a09g056.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r9a09g057.dtsi | 2 +- + .../boot/dts/renesas/r9a09g057h44-rzv2h-evk.dts | 6 +- + arch/arm64/boot/dts/renesas/r9a09g077.dtsi | 61 ++++- + .../dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso | 53 ++++ + .../boot/dts/renesas/r9a09g077m44-rzt2h-evk.dts | 18 +- + arch/arm64/boot/dts/renesas/r9a09g087.dtsi | 61 ++++- + .../dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso | 63 +++++ + .../boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts | 32 ++- + .../boot/dts/renesas/rzg2l-smarc-pinfunction.dtsi | 16 +- + arch/arm64/boot/dts/renesas/rzg2l-smarc-som.dtsi | 20 +- + arch/arm64/boot/dts/renesas/rzg2l-smarc.dtsi | 2 +- + .../boot/dts/renesas/rzg2lc-smarc-pinfunction.dtsi | 16 +- + arch/arm64/boot/dts/renesas/rzg2lc-smarc-som.dtsi | 20 +- + arch/arm64/boot/dts/renesas/rzg2lc-smarc.dtsi | 2 +- + .../boot/dts/renesas/rzg2ul-smarc-pinfunction.dtsi | 16 +- + arch/arm64/boot/dts/renesas/rzg2ul-smarc-som.dtsi | 20 +- + arch/arm64/boot/dts/renesas/rzg3l-smarc-som.dtsi | 14 + + arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi | 25 +- + arch/arm64/boot/dts/renesas/rzg3s-smarc-switches.h | 4 + + arch/arm64/boot/dts/renesas/rzg3s-smarc.dtsi | 1 - + .../boot/dts/renesas/rzt2h-n2h-evk-common.dtsi | 7 + + .../boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi | 72 +++++ + arch/arm64/boot/dts/renesas/salvator-common.dtsi | 14 +- + arch/arm64/boot/dts/renesas/salvator-xs.dtsi | 2 +- + .../ulcb-kf-audio-graph-card2-mix+split.dtsi | 18 +- + arch/arm64/boot/dts/renesas/ulcb-kf.dtsi | 2 +- + arch/arm64/boot/dts/renesas/ulcb.dtsi | 8 +- + .../boot/dts/renesas/white-hawk-cpu-common.dtsi | 6 +- + drivers/soc/renesas/Kconfig | 10 +- + drivers/soc/renesas/r9a08g046-sysc.c | 1 + + drivers/soc/renesas/rcar-mfis.c | 8 +- + drivers/soc/renesas/rz-sysc.c | 5 + + drivers/soc/renesas/rz-sysc.h | 2 + + 105 files changed, 1116 insertions(+), 319 deletions(-) + create mode 100644 arch/arm64/boot/dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso + create mode 100644 arch/arm64/boot/dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi +Merging reset/reset/next (d373605cd5148 Merge tag 'reset-fixes-for-v7.0-2' into reset/next) +$ git merge -m Merge branch 'reset/next' of https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux reset/reset/next +Already up to date. +Merging rockchip/for-next (32e0f64640d55 Merge branch 'v7.4-armsoc/dts64' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip.git rockchip/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/rockchip.yaml | 7 + + arch/arm/boot/dts/rockchip/rk3229-xms6.dts | 4 +- + arch/arm64/boot/dts/rockchip/Makefile | 6 + + .../boot/dts/rockchip/rk3399pro-vmarc-som.dtsi | 2 + + .../boot/dts/rockchip/rk3576-armsom-cm5-io.dts | 4 +- + arch/arm64/boot/dts/rockchip/rk3576.dtsi | 2 +- + arch/arm64/boot/dts/rockchip/rk3588-base.dtsi | 39 + + .../rockchip/rk3588-jaguar-can1-can2-uart4.dtso | 26 + + arch/arm64/boot/dts/rockchip/rk3588-jaguar.dts | 16 + + .../boot/dts/rockchip/rk3588-lubancat-5-btb.dtsi | 534 ++++++++++ + .../boot/dts/rockchip/rk3588-lubancat-5io.dts | 1032 ++++++++++++++++++++ + .../boot/dts/rockchip/rk3588-nanopc-t6-lts.dts | 17 - + arch/arm64/boot/dts/rockchip/rk3588-nanopc-t6.dtsi | 175 ++-- + .../boot/dts/rockchip/rk3588-tiger-haikou.dts | 4 + + arch/arm64/boot/dts/rockchip/rk3588-tiger.dtsi | 5 + + .../boot/dts/rockchip/rk3588s-gameforce-ace.dts | 7 +- + 16 files changed, 1755 insertions(+), 125 deletions(-) + create mode 100644 arch/arm64/boot/dts/rockchip/rk3588-jaguar-can1-can2-uart4.dtso + create mode 100644 arch/arm64/boot/dts/rockchip/rk3588-lubancat-5-btb.dtsi + create mode 100644 arch/arm64/boot/dts/rockchip/rk3588-lubancat-5io.dts +Merging samsung-krzk/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git samsung-krzk/for-next +Already up to date. +Merging scmi/for-linux-next (12866c647dc38 Merge branches 'for-next/scmi/fixes' and 'for-next/ffa/fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux) +$ git merge -m Merge branch 'for-linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux.git scmi/for-linux-next +Merge made by the 'ort' strategy. + drivers/clk/clk-scpi.c | 11 ++++++++--- + drivers/firmware/arm_ffa/driver.c | 1 + + drivers/firmware/arm_scpi.c | 9 ++++++--- + 3 files changed, 15 insertions(+), 6 deletions(-) +Merging sophgo/for-next (76acfee87c74d Merge branch 'dt/arm' into for-next) +$ git merge -m Merge branch 'for-next' of https://github.com/sophgo/linux.git sophgo/for-next +Merge made by the 'ort' strategy. +Merging sophgo-soc/soc-for-next (c8754c7deab4c soc: sophgo: cv1800: rtcsys: New driver (handling RTC only)) +$ git merge -m Merge branch 'soc-for-next' of https://github.com/sophgo/linux.git sophgo-soc/soc-for-next +Already up to date. +Merging spacemit/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux spacemit/for-next +Already up to date. +Merging stm32/stm32-next (90d4d470d4d47 arm64: dts: st: Add I/O sync to eth1 pinctrl in stm32mp25-pinctrl.dtsi) +$ git merge -m Merge branch 'stm32-next' of https://git.kernel.org/pub/scm/linux/kernel/git/atorgue/stm32.git stm32/stm32-next +Auto-merging arch/arm/boot/dts/st/stm32mp135f-dk.dts +Auto-merging arch/arm64/boot/dts/st/stm32mp257f-ev1.dts +Merge made by the 'ort' strategy. + arch/arm/boot/dts/st/stm32mp131.dtsi | 24 ++++++++++ + arch/arm/boot/dts/st/stm32mp135.dtsi | 44 ++++++++--------- + arch/arm/boot/dts/st/stm32mp135f-dk.dts | 16 ++++++- + arch/arm/boot/dts/st/stm32mp15-pinctrl.dtsi | 34 +++++++++++++ + arch/arm/boot/dts/st/stm32mp151.dtsi | 20 ++++---- + .../arm/boot/dts/st/stm32mp153c-lxa-fairytux2.dtsi | 2 +- + arch/arm/boot/dts/st/stm32mp157a-dk1-scmi.dts | 8 ++-- + arch/arm/boot/dts/st/stm32mp157c-dk2-scmi.dts | 8 ++-- + arch/arm/boot/dts/st/stm32mp157c-ed1-scmi.dts | 8 ++-- + arch/arm/boot/dts/st/stm32mp157c-ev1-scmi.dts | 8 ++-- + arch/arm/boot/dts/st/stm32mp157c-ev1.dts | 8 ++-- + arch/arm/boot/dts/st/stm32mp157c-lxa-mc1.dts | 2 +- + arch/arm/boot/dts/st/stm32mp15xc-lxa-tac.dtsi | 2 +- + arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi | 34 +++++++++++-- + arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi | 2 + + arch/arm64/boot/dts/st/stm32mp257f-ev1.dts | 56 +++++++++++----------- + 16 files changed, 187 insertions(+), 89 deletions(-) +Merging sunxi/sunxi/for-next (913f7eda9f3e3 ARM: dts: allwinner: sun7i-a20: Replace spaces indentation with tabs) +$ git merge -m Merge branch 'sunxi/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sunxi/linux.git sunxi/sunxi/for-next +Merge made by the 'ort' strategy. + arch/arm/boot/dts/allwinner/sun7i-a20.dtsi | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) +Merging tee/next (a3067938fd192 Merge branches 'qcomtee_for_v7.3', 'optee_fix_for_v7.2' and 'jw-korg' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee.git tee/next +Merge made by the 'ort' strategy. +Merging tegra/for-next (a53c665d36f61 Merge branch for-7.4/arm64/dt into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git tegra/for-next +Merge made by the 'ort' strategy. + .../memory-controllers/nvidia,tegra124-emc.yaml | 10 ++- + .../memory-controllers/nvidia,tegra124-mc.yaml | 1 + + arch/arm64/boot/dts/nvidia/tegra194.dtsi | 4 + + drivers/soc/tegra/pmc.c | 98 ++++++++++++++++++++++ + 4 files changed, 110 insertions(+), 3 deletions(-) +Merging tenstorrent-dt/tenstorrent-dt-for-next (33583baeb1ba7 dt-bindings: iommu: riscv: Add bindings for Tenstorrent RISC-V IOMMU) +$ git merge -m Merge branch 'tenstorrent-dt-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-dt/tenstorrent-dt-for-next +Already up to date. +Merging fustini-config/riscv-config-for-next (020e209272bee riscv: defconfig: thead: enable PCA953X GPIO driver) +$ git merge -m Merge branch 'riscv-config-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git fustini-config/riscv-config-for-next +Already up to date. +Merging thead-dt/thead-dt-for-next (15d32aaf6300b riscv: dts: thead: Add remaining Lichee Pi 4A IO expansions) +$ git merge -m Merge branch 'thead-dt-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-dt/thead-dt-for-next +Already up to date. +Merging ti/ti-next (7abd3f29fdf16 MAINTAINERS: Add git tree for TI K3 ARCHITECTURE) +$ git merge -m Merge branch 'ti-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ti/linux.git ti/ti-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 1 + + 1 file changed, 1 insertion(+) +Merging xilinx/for-next (20df11b6d0666 ARM: dts: xilinx: Replace spaces indentation with tabs) +$ git merge -m Merge branch 'for-next' of https://github.com/Xilinx/linux-xlnx.git xilinx/for-next +Merge made by the 'ort' strategy. + arch/arm/boot/dts/xilinx/zynq-7000.dtsi | 8 ++++---- + arch/arm/boot/dts/xilinx/zynq-parallella.dts | 4 ++-- + 2 files changed, 6 insertions(+), 6 deletions(-) +Merging socfpga/for-next (ff98c9832fd43 arm64: dts: socfpga: use consistent QSPI boot partition label) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux.git socfpga/for-next +Already up to date. +Merging clk/clk-next (551ba775497e9 clk: davinci: add COMPILE_TEST support) +$ git merge -m Merge branch 'clk-next' of https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git clk/clk-next +Auto-merging drivers/clk/clk-scpi.c +Merge made by the 'ort' strategy. + drivers/clk/Kconfig | 1 + + drivers/clk/Makefile | 2 +- + drivers/clk/actions/owl-composite.h | 1 - + drivers/clk/actions/owl-fixed-factor.h | 2 -- + drivers/clk/aspeed/clk-aspeed.c | 2 +- + drivers/clk/aspeed/clk-ast2600.c | 2 +- + drivers/clk/aspeed/clk-ast2700.c | 2 +- + drivers/clk/at91/clk-audio-pll.c | 4 +-- + drivers/clk/at91/clk-h32mx.c | 2 +- + drivers/clk/at91/clk-main.c | 2 +- + drivers/clk/at91/clk-pll.c | 2 +- + drivers/clk/at91/clk-plldiv.c | 2 +- + drivers/clk/at91/clk-slow.c | 2 +- + drivers/clk/at91/clk-smd.c | 2 +- + drivers/clk/at91/clk-usb.c | 6 ++-- + drivers/clk/at91/sckc.c | 2 +- + drivers/clk/bcm/clk-iproc-armpll.c | 2 +- + drivers/clk/bcm/clk-iproc-asiu.c | 2 +- + drivers/clk/bcm/clk-iproc-pll.c | 2 +- + drivers/clk/berlin/berlin2-avpll.c | 4 +-- + drivers/clk/berlin/berlin2-pll.c | 2 +- + drivers/clk/clk-axi-clkgen.c | 2 +- + drivers/clk/clk-cdce925.c | 2 +- + drivers/clk/clk-cs2000-cp.c | 2 +- + drivers/clk/clk-fractional-divider.c | 2 +- + drivers/clk/clk-gemini.c | 2 +- + drivers/clk/clk-highbank.c | 2 +- + drivers/clk/clk-lmk04832.c | 6 ++-- + drivers/clk/clk-milbeaut.c | 4 +-- + drivers/clk/clk-nomadik.c | 4 +-- + drivers/clk/clk-npcm7xx.c | 2 +- + drivers/clk/clk-pwm.c | 2 +- + drivers/clk/clk-scpi.c | 2 +- + drivers/clk/clk-si514.c | 2 +- + drivers/clk/clk-si521xx.c | 8 ++--- + drivers/clk/clk-si5341.c | 2 +- + drivers/clk/clk-si544.c | 2 +- + drivers/clk/clk-si570.c | 2 +- + drivers/clk/clk-stm32f4.c | 4 +-- + drivers/clk/clk-stm32h7.c | 2 +- + drivers/clk/clk-versaclock5.c | 8 ++--- + drivers/clk/clk-vt8500.c | 4 +-- + drivers/clk/clk-xgene.c | 6 ++-- + drivers/clk/clk.c | 46 +++++++++++++----------- + drivers/clk/clk.h | 5 +-- + drivers/clk/clkdev.c | 4 +-- + drivers/clk/davinci/Kconfig | 10 ++++++ + drivers/clk/davinci/Makefile | 8 ++--- + drivers/clk/davinci/da8xx-cfgchip.c | 4 +-- + drivers/clk/davinci/pll.c | 2 +- + drivers/clk/davinci/psc.c | 2 +- + drivers/clk/hisilicon/clk-hi3559a.c | 2 +- + drivers/clk/hisilicon/clk-hi3620.c | 2 +- + drivers/clk/hisilicon/clk-hi6220-stub.c | 2 +- + drivers/clk/hisilicon/clk-hisi-phase.c | 2 +- + drivers/clk/hisilicon/clk-hix5hd2.c | 2 +- + drivers/clk/hisilicon/clkdivider-hi6220.c | 2 +- + drivers/clk/hisilicon/clkgate-separated.c | 2 +- + drivers/clk/imx/clk-busy.c | 4 +-- + drivers/clk/imx/clk-cpu.c | 2 +- + drivers/clk/imx/clk-divider-gate.c | 2 +- + drivers/clk/imx/clk-fixup-div.c | 2 +- + drivers/clk/imx/clk-fixup-mux.c | 2 +- + drivers/clk/imx/clk-frac-pll.c | 2 +- + drivers/clk/imx/clk-fracn-gppll.c | 2 +- + drivers/clk/imx/clk-gate-93.c | 2 +- + drivers/clk/imx/clk-gate-exclusive.c | 2 +- + drivers/clk/imx/clk-gate2.c | 2 +- + drivers/clk/imx/clk-lpcg-scu.c | 2 +- + drivers/clk/imx/clk-pfd.c | 2 +- + drivers/clk/imx/clk-pfdv2.c | 2 +- + drivers/clk/imx/clk-pll14xx.c | 2 +- + drivers/clk/imx/clk-pllv1.c | 2 +- + drivers/clk/imx/clk-pllv2.c | 2 +- + drivers/clk/imx/clk-pllv3.c | 2 +- + drivers/clk/imx/clk-pllv4.c | 2 +- + drivers/clk/imx/clk-scu.c | 4 +-- + drivers/clk/imx/clk-sscg-pll.c | 2 +- + drivers/clk/ingenic/cgu.c | 2 +- + drivers/clk/keystone/gate.c | 2 +- + drivers/clk/keystone/pll.c | 2 +- + drivers/clk/keystone/syscon-clk.c | 2 +- + drivers/clk/mediatek/clk-cpumux.c | 2 +- + drivers/clk/mmp/clk-apbc.c | 2 +- + drivers/clk/mmp/clk-apmu.c | 2 +- + drivers/clk/mmp/clk-frac.c | 2 +- + drivers/clk/mmp/clk-gate.c | 2 +- + drivers/clk/mmp/clk-mix.c | 2 +- + drivers/clk/mmp/clk-pll.c | 2 +- + drivers/clk/mstar/clk-msc313-mpll.c | 2 +- + drivers/clk/mvebu/ap-cpu-clk.c | 2 +- + drivers/clk/mvebu/clk-corediv.c | 2 +- + drivers/clk/mvebu/clk-cpu.c | 2 +- + drivers/clk/mxs/clk-div.c | 2 +- + drivers/clk/mxs/clk-frac.c | 2 +- + drivers/clk/mxs/clk-pll.c | 2 +- + drivers/clk/mxs/clk-ref.c | 2 +- + drivers/clk/nxp/clk-lpc18xx-creg.c | 2 +- + drivers/clk/pistachio/clk-pll.c | 2 +- + drivers/clk/rockchip/clk-cpu.c | 2 +- + drivers/clk/rockchip/clk-ddr.c | 2 +- + drivers/clk/rockchip/clk-gate-grf.c | 2 +- + drivers/clk/rockchip/clk-inverter.c | 2 +- + drivers/clk/rockchip/clk-mmc-phase.c | 2 +- + drivers/clk/rockchip/clk-muxgrf.c | 2 +- + drivers/clk/rockchip/clk-pll.c | 2 +- + drivers/clk/rockchip/clk.c | 2 +- + drivers/clk/samsung/clk-cpu.c | 2 +- + drivers/clk/samsung/clk-exynos-clkout.c | 8 ++--- + drivers/clk/samsung/clk-pll.c | 2 +- + drivers/clk/socfpga/clk-gate-a10.c | 2 +- + drivers/clk/socfpga/clk-gate-s10.c | 6 ++-- + drivers/clk/socfpga/clk-gate.c | 2 +- + drivers/clk/socfpga/clk-periph-a10.c | 2 +- + drivers/clk/socfpga/clk-periph-s10.c | 8 ++--- + drivers/clk/socfpga/clk-periph.c | 2 +- + drivers/clk/socfpga/clk-pll-a10.c | 2 +- + drivers/clk/socfpga/clk-pll-s10.c | 8 ++--- + drivers/clk/socfpga/clk-pll.c | 2 +- + drivers/clk/spear/clk-aux-synth.c | 2 +- + drivers/clk/spear/clk-frac-synth.c | 2 +- + drivers/clk/spear/clk-gpt-synth.c | 2 +- + drivers/clk/spear/clk-vco-pll.c | 2 +- + drivers/clk/st/clk-flexgen.c | 2 +- + drivers/clk/st/clkgen-fsyn.c | 4 +-- + drivers/clk/st/clkgen-pll.c | 2 +- + drivers/clk/stm32/clk-stm32mp1.c | 4 +-- + drivers/clk/sunxi/clk-sun4i-tcon-ch1.c | 2 +- + drivers/clk/tegra/clk-audio-sync.c | 2 +- + drivers/clk/tegra/clk-divider.c | 2 +- + drivers/clk/tegra/clk-periph-fixed.c | 2 +- + drivers/clk/tegra/clk-periph-gate.c | 2 +- + drivers/clk/tegra/clk-periph.c | 2 +- + drivers/clk/tegra/clk-pll-out.c | 2 +- + drivers/clk/tegra/clk-pll.c | 2 +- + drivers/clk/tegra/clk-sdmmc-mux.c | 2 +- + drivers/clk/tegra/clk-super.c | 4 +-- + drivers/clk/tegra/clk-tegra-super-cclk.c | 2 +- + drivers/clk/tegra/clk-tegra124-emc.c | 2 +- + drivers/clk/tegra/clk-tegra20-emc.c | 2 +- + drivers/clk/tegra/clk-tegra210-emc.c | 2 +- + drivers/clk/uniphier/clk-uniphier-cpugear.c | 2 +- + drivers/clk/uniphier/clk-uniphier-fixed-factor.c | 2 +- + drivers/clk/uniphier/clk-uniphier-fixed-rate.c | 2 +- + drivers/clk/uniphier/clk-uniphier-gate.c | 2 +- + drivers/clk/uniphier/clk-uniphier-mux.c | 2 +- + drivers/clk/ux500/clk-prcc.c | 2 +- + drivers/clk/ux500/clk-prcmu.c | 4 +-- + drivers/clk/ux500/clk-sysctrl.c | 2 +- + drivers/clk/versatile/clk-icst.c | 2 +- + drivers/clk/versatile/clk-sp810.c | 2 +- + drivers/clk/versatile/clk-vexpress-osc.c | 2 +- + drivers/clk/x86/clk-pmc-atom.c | 2 +- + drivers/clk/xilinx/clk-xlnx-clock-wizard.c | 14 ++++---- + drivers/clk/xilinx/xlnx_vcu.c | 2 +- + drivers/clk/zynqmp/clk-gate-zynqmp.c | 2 +- + drivers/clk/zynqmp/clk-mux-zynqmp.c | 2 +- + drivers/clk/zynqmp/divider.c | 2 +- + drivers/clk/zynqmp/pll.c | 2 +- + include/linux/clk-provider.h | 22 ++++++++++-- + 160 files changed, 257 insertions(+), 228 deletions(-) + create mode 100644 drivers/clk/davinci/Kconfig +Merging clk-imx/for-next (39ec460b56b26 clk: imx95-blk-ctl: Fix REFCLK rise-fall mismatch on i.MX95) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/abelvesa/linux.git clk-imx/for-next +Already up to date. +Merging clk-renesas/renesas-clk (26e80ec751c87 clk: renesas: Make sure clk_init_data is fully initialized) +$ git merge -m Merge branch 'renesas-clk' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git clk-renesas/renesas-clk +Merge made by the 'ort' strategy. + drivers/clk/renesas/r9a06g032-clocks.c | 2 +- + drivers/clk/renesas/r9a08g046-cpg.c | 83 ++++- + drivers/clk/renesas/rzg2l-cpg.c | 554 ++++++++++++++++++++++++++++++++- + drivers/clk/renesas/rzg2l-cpg.h | 51 ++- + drivers/clk/renesas/rzv2h-cpg.c | 8 +- + 5 files changed, 662 insertions(+), 36 deletions(-) +Merging thead-clk/thead-clk-for-next (b2ee00d0bf4cc clk: thead: allow COMPILE_TEST builds) +$ git merge -m Merge branch 'thead-clk-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-clk/thead-clk-for-next +Already up to date. +Merging tenstorrent-clk/tenstorrent-clk-for-next (23c8ebc952849 clk: tenstorrent: Add Atlantis clock controller driver) +$ git merge -m Merge branch 'tenstorrent-clk-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-clk/tenstorrent-clk-for-next +Already up to date. +Merging csky/linux-next (abb81e5ce7d99 csky: Fix a4/a5 restoration in syscall trace path) +$ git merge -m Merge branch 'linux-next' of https://github.com/c-sky/csky-linux.git csky/linux-next +Already up to date. +Merging loongarch/loongarch-next (e2a8488684418 Merge branch 'loongarch-kvm' into loongarch-next) +$ git merge -m Merge branch 'loongarch-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chenhuacai/linux-loongson.git loongarch/loongarch-next +Merge made by the 'ort' strategy. +Merging m68k/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git m68k/for-next +Already up to date. +Merging m68knommu/for-next (de0dab22cbf69 m68k: coldfire/5441x: register mcf-rcm-reset platform device) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gerg/m68knommu.git m68knommu/for-next +Already up to date. +Merging microblaze/next (8518fd17ccef4 microblaze: prevent ptrace writes to MSR and pt_mode) +$ git merge -m Merge branch 'next' of git://git.monstr.eu/linux-2.6-microblaze.git microblaze/next +Merge made by the 'ort' strategy. + arch/microblaze/include/asm/entry.h | 13 + + arch/microblaze/include/asm/processor.h | 2 +- + arch/microblaze/kernel/cpu/cpuinfo-pvr-full.c | 2 +- + arch/microblaze/kernel/cpu/cpuinfo-static.c | 2 +- + arch/microblaze/kernel/cpu/cpuinfo.c | 1 + + arch/microblaze/kernel/entry.S | 348 ++++++++++++++------------ + arch/microblaze/kernel/hw_exception_handler.S | 5 + + arch/microblaze/kernel/process.c | 5 +- + arch/microblaze/kernel/ptrace.c | 3 + + arch/microblaze/kernel/signal.c | 26 ++ + arch/microblaze/kernel/syscalls/syscall.tbl | 2 +- + 11 files changed, 239 insertions(+), 170 deletions(-) +Merging mips/mips-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'mips-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git mips/mips-next +Already up to date. +Merging openrisc/for-next (6620f5e8c11c4 openrisc: drop unneeded semicolon) +$ git merge -m Merge branch 'for-next' of https://github.com/openrisc/linux.git openrisc/for-next +Already up to date. +Merging parisc-hd/for-next (8d3ae59288f1e Linux 7.2) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/deller/parisc-linux.git parisc-hd/for-next +Already up to date. +Merging powerpc/next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git powerpc/next +Already up to date. +Merging risc-v/for-next (77ae27fd98f3b Merge tag 'printk-for-7.3' of git://git.kernel.org/pub/scm/linux/kernel/git/printk/linux) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git risc-v/for-next +Already up to date. +Merging riscv-dt/riscv-dt-for-next (82ab962ef6c2a Merge branch 'k230-basic' into riscv-dt-for-next) +$ git merge -m Merge branch 'riscv-dt-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-dt/riscv-dt-for-next +Already up to date. +Merging riscv-soc/riscv-soc-for-next (8cdeaa50eae8d Linux 7.2-rc2) +$ git merge -m Merge branch 'riscv-soc-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-soc/riscv-soc-for-next +Already up to date. +Merging s390/for-next (11acf42968807 Merge branch 'fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git s390/for-next +Merge made by the 'ort' strategy. +Merging sh/for-next (b0aa5e4b087b6 sh: Fix fallout from ZERO_PAGE consolidation) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/glaubitz/sh-linux.git sh/for-next +Already up to date. +Merging sparc/for-next (5b2a3b1a98fb4 sparc: Remove remaining defconfig references to the pktcdvd driver) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git sparc/for-next +Already up to date. +Merging uml/next (1590cf0329716 Linux 7.2-rc4) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git uml/next +Already up to date. +Merging xtensa/xtensa-for-next (eb049bdbf2b98 xtensa: remove unused setup_profiling_timer function) +$ git merge -m Merge branch 'xtensa-for-next' of https://github.com/jcmvbkbc/linux-xtensa.git xtensa/xtensa-for-next +Already up to date. +Merging fs-next (054b582169847 Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git) +$ git merge -m Merge branch 'fs-next' of linux-next fs-next +Auto-merging MAINTAINERS +Auto-merging fs/fat/misc.c +Auto-merging fs/ocfs2/journal.c +Merge made by the 'ort' strategy. + Documentation/admin-guide/binfmt-misc.rst | 3 + + Documentation/filesystems/exfat.rst | 117 ++ + Documentation/filesystems/index.rst | 1 + + Documentation/filesystems/proc.rst | 4 + + Documentation/filesystems/smb/ksmbd.rst | 14 +- + MAINTAINERS | 3 +- + arch/powerpc/Kconfig | 1 - + arch/powerpc/include/asm/elf.h | 6 - + arch/powerpc/include/asm/spu.h | 3 - + arch/powerpc/platforms/cell/Kconfig | 1 - + arch/powerpc/platforms/cell/spu_syscalls.c | 20 - + arch/powerpc/platforms/cell/spufs/Makefile | 1 - + arch/powerpc/platforms/cell/spufs/coredump.c | 183 -- + arch/powerpc/platforms/cell/spufs/file.c | 114 -- + arch/powerpc/platforms/cell/spufs/spufs.h | 12 - + arch/powerpc/platforms/cell/spufs/syscalls.c | 4 - + drivers/base/devtmpfs.c | 108 +- + fs/9p/vfs_inode.c | 3 + + fs/9p/vfs_inode_dotl.c | 3 + + fs/adfs/dir.c | 2 +- + fs/aio.c | 11 +- + fs/binfmt_elf.c | 16 +- + fs/binfmt_elf_fdpic.c | 12 +- + fs/binfmt_misc.c | 12 +- + fs/bpf_fs_kfuncs.c | 4 +- + fs/btrfs/Kconfig | 1 + + fs/btrfs/bio.c | 141 +- + fs/btrfs/bio.h | 5 +- + fs/btrfs/block-group.c | 20 +- + fs/btrfs/btrfs_inode.h | 12 +- + fs/btrfs/dev-replace.c | 10 +- + fs/btrfs/disk-io.c | 27 +- + fs/btrfs/extent-tree.c | 10 + + fs/btrfs/extent_io.c | 64 +- + fs/btrfs/extent_map.c | 8 + + fs/btrfs/file-item.c | 19 +- + fs/btrfs/fs.h | 2 +- + fs/btrfs/inode.c | 114 +- + fs/btrfs/ioctl.c | 2 +- + fs/btrfs/qgroup.c | 126 +- + fs/btrfs/qgroup.h | 16 +- + fs/btrfs/raid56.c | 46 +- + fs/btrfs/sysfs.c | 129 +- + fs/btrfs/tree-checker.c | 66 +- + fs/btrfs/verity.c | 31 +- + fs/btrfs/volumes.c | 39 +- + fs/btrfs/volumes.h | 1 + + fs/btrfs/zoned.c | 5 + + fs/buffer.c | 37 +- + fs/ceph/file.c | 3 + + fs/ceph/mds_client.c | 4 + + fs/ceph/mds_client.h | 1 + + fs/ceph/super.c | 5 + + fs/configfs/dir.c | 9 + + fs/configfs/symlink.c | 24 +- + fs/coredump.c | 438 +++-- + fs/dlm/config.c | 30 +- + fs/dlm/dlm_internal.h | 5 + + fs/dlm/lock.c | 18 +- + fs/dlm/lock.h | 4 +- + fs/dlm/lowcomms.c | 1 + + fs/dlm/midcomms.c | 1 + + fs/dlm/plock.c | 14 +- + fs/dlm/user.c | 23 + + fs/exfat/dir.c | 49 +- + fs/exfat/iomap.c | 28 +- + fs/exfat/misc.c | 2 +- + fs/exfat/namei.c | 3 +- + fs/ext2/Makefile | 2 + + fs/ext2/balloc.c | 4 + + fs/ext2/ext2.h | 19 +- + fs/ext2/inode.c | 7 + + fs/ext2/super.c | 33 +- + fs/ext2/xattr.c | 2 +- + fs/ext4/ext4_jbd2.c | 2 +- + fs/ext4/mmp.c | 2 +- + fs/fat/misc.c | 2 +- + fs/fs-writeback.c | 2 +- + fs/fuse/dax.c | 1 + + fs/fuse/dir.c | 3 + + fs/fuse/file.c | 7 +- + fs/fuse/fuse_i.h | 9 + + fs/fuse/inode.c | 28 + + fs/fuse/ioctl.c | 3 - + fs/fuse/req.c | 7 +- + fs/gfs2/acl.c | 11 +- + fs/gfs2/aops.c | 17 +- + fs/gfs2/bmap.c | 91 +- + fs/gfs2/dentry.c | 6 +- + fs/gfs2/dir.c | 61 +- + fs/gfs2/export.c | 7 +- + fs/gfs2/file.c | 72 +- + fs/gfs2/glock.c | 60 +- + fs/gfs2/glops.c | 13 +- + fs/gfs2/incore.h | 8 +- + fs/gfs2/inode.c | 125 +- + fs/gfs2/log.c | 7 +- + fs/gfs2/lops.c | 22 +- + fs/gfs2/meta_io.c | 7 +- + fs/gfs2/ops_fstype.c | 26 +- + fs/gfs2/quota.c | 54 +- + fs/gfs2/recovery.c | 17 +- + fs/gfs2/rgrp.c | 129 +- + fs/gfs2/super.c | 98 +- + fs/gfs2/trace_gfs2.h | 6 +- + fs/gfs2/util.c | 8 +- + fs/gfs2/xattr.c | 69 +- + fs/inode.c | 2 +- + fs/internal.h | 1 + + fs/iomap/buffered-io.c | 6 +- + fs/jbd2/commit.c | 22 +- + fs/jbd2/journal.c | 31 +- + fs/jbd2/transaction.c | 2 +- + fs/kernfs/dir.c | 76 +- + fs/kernfs/kernfs-internal.h | 9 +- + fs/lockd/svc.c | 1 - + fs/lockd/trace.h | 1 - + fs/lockd/xdr.h | 2 +- + fs/namei.c | 229 ++- + fs/nfs/dir.c | 6 + + fs/nfs/nfs4file.c | 1 + + fs/nfs/super.c | 25 - + fs/nfs_common/nfs_ssc.c | 126 +- + fs/nfsd/blocklayout.c | 1 + + fs/nfsd/blocklayoutxdr.c | 11 + + fs/nfsd/export.c | 8 +- + fs/nfsd/export.h | 3 +- + fs/nfsd/filecache.c | 1 + + fs/nfsd/flexfilelayout.c | 24 +- + fs/nfsd/flexfilelayoutxdr.c | 9 +- + fs/nfsd/flexfilelayoutxdr.h | 8 +- + fs/nfsd/localio.c | 10 +- + fs/nfsd/lockd.c | 4 +- + fs/nfsd/nfs2acl.c | 1 + + fs/nfsd/nfs3acl.c | 1 + + fs/nfsd/nfs3proc.c | 37 +- + fs/nfsd/nfs3xdr.c | 3 + + fs/nfsd/nfs4acl.c | 1 + + fs/nfsd/nfs4callback.c | 8 +- + fs/nfsd/nfs4ctl.h | 83 + + fs/nfsd/nfs4idmap.c | 1 + + fs/nfsd/nfs4layouts.c | 1 + + fs/nfsd/nfs4proc.c | 362 ++-- + fs/nfsd/nfs4recover.c | 1 + + fs/nfsd/nfs4state.c | 244 ++- + fs/nfsd/nfs4xdr.c | 104 +- + fs/nfsd/nfscache.c | 1 + + fs/nfsd/nfsctl.c | 2 + + fs/nfsd/nfsd.h | 236 +-- + fs/nfsd/nfserr.h | 158 ++ + fs/nfsd/nfsfh.c | 72 +- + fs/nfsd/nfsfh.h | 28 +- + fs/nfsd/nfsproc.c | 20 +- + fs/nfsd/nfssvc.c | 8 + + fs/nfsd/nfsxdr.c | 1 + + fs/nfsd/state.h | 43 +- + fs/nfsd/vfs.c | 91 +- + fs/nfsd/vfs.h | 10 +- + fs/nfsd/xdr3.h | 2 +- + fs/nfsd/xdr4.h | 152 +- + fs/nfsd/xdr4cb.h | 20 +- + fs/ntfs/attrib.c | 18 +- + fs/ntfs/bdev-io.c | 2 +- + fs/ntfs/bitmap.c | 8 +- + fs/ntfs/compress.c | 2 +- + fs/ntfs/ea.c | 49 +- + fs/ntfs/file.c | 50 +- + fs/ntfs/inode.c | 10 +- + fs/ntfs/lcnalloc.c | 9 +- + fs/ntfs/mft.c | 16 +- + fs/ntfs/ntfs.h | 10 +- + fs/ntfs/reparse.c | 7 +- + fs/ntfs/super.c | 15 +- + fs/ntfs/wof.c | 127 +- + fs/ocfs2/buffer_head_io.c | 12 +- + fs/ocfs2/journal.c | 25 +- + fs/omfs/inode.c | 4 +- + fs/open.c | 59 +- + fs/smb/client/cifsacl.c | 15 + + fs/smb/client/cifssmb.c | 25 +- + fs/smb/client/dir.c | 3 + + fs/smb/client/trace.h | 1 + + fs/smb/client/transport.c | 13 +- + fs/smb/server/Kconfig | 2 + + fs/smb/server/Makefile | 1 + + fs/smb/server/ksmbd_work.c | 3 - + fs/smb/server/ksmbd_work.h | 7 +- + fs/smb/server/mgmt/tree_connect.c | 8 + + fs/smb/server/oplock.c | 73 +- + fs/smb/server/smb2pdu.c | 412 ++-- + fs/smb/server/smbacl.c | 2 + + fs/smb/server/tests/Kconfig | 15 + + fs/smb/server/tests/Makefile | 4 + + fs/smb/server/tests/smbacl_kunit.c | 301 +++ + fs/smb/server/vfs.c | 14 +- + fs/smb/server/vfs_cache.c | 48 - + fs/smb/server/vfs_cache.h | 6 - + fs/udf/inode.c | 52 +- + fs/udf/misc.c | 15 +- + fs/udf/super.c | 5 +- + fs/vboxsf/dir.c | 3 + + fs/xfs/libxfs/xfs_da_btree.c | 1 + + fs/xfs/libxfs/xfs_defer.c | 18 +- + fs/xfs/libxfs/xfs_exchmaps.c | 10 + + fs/xfs/libxfs/xfs_parent.c | 12 +- + fs/xfs/libxfs/xfs_trans_space.c | 19 +- + fs/xfs/scrub/dir_repair.c | 27 +- + fs/xfs/scrub/metapath.c | 87 +- + fs/xfs/scrub/symlink_repair.c | 2 +- + fs/xfs/xfs_buf.c | 4 +- + fs/xfs/xfs_extent_busy.c | 4 +- + fs/xfs/xfs_healthmon.c | 6 + + fs/xfs/xfs_icache.c | 3 +- + fs/xfs/xfs_ioctl.c | 381 ++-- + fs/xfs/xfs_ioctl.h | 4 +- + fs/xfs/xfs_ioctl32.c | 188 +- + fs/xfs/xfs_log.c | 33 +- + fs/xfs/xfs_log_cil.c | 3 +- + fs/xfs/xfs_log_priv.h | 4 +- + fs/xfs/xfs_mru_cache.c | 2 +- + fs/xfs/xfs_trans_ail.c | 4 +- + fs/xfs/xfs_verify_media.c | 24 +- + fs/xfs/xfs_zone_alloc.c | 2 + + fs/xfs/xfs_zone_space_resv.c | 8 +- + include/linux/binfmts.h | 3 +- + include/linux/buffer_head.h | 36 +- + include/linux/coredump.h | 37 +- + include/linux/fcntl.h | 6 + + include/linux/nfs.h | 55 +- + include/linux/nfs3.h | 8 + + include/linux/nfs4.h | 6 + + include/linux/nfs_fh.h | 63 + + include/linux/nfs_ssc.h | 69 +- + include/linux/nfsd_ssc.h | 38 + + include/linux/nfslocalio.h | 11 +- + include/linux/sched.h | 2 +- + include/linux/sched/signal.h | 27 +- + include/trace/misc/nfs.h | 1 + + include/uapi/linux/btrfs_tree.h | 21 +- + include/uapi/linux/coredump.h | 149 +- + include/uapi/linux/fuse.h | 12 +- + kernel/pid_namespace.c | 3 +- + kernel/user_namespace.c | 3 - + net/ceph/messenger.c | 1 - + tools/include/uapi/linux/coredump.h | 149 +- + tools/testing/selftests/Makefile | 1 + + tools/testing/selftests/coredump/Makefile | 7 +- + .../selftests/coredump/coredump_notify_signal.h | 29 + + .../coredump/coredump_notify_signal_helper.c | 46 + + .../coredump/coredump_notify_signal_test.c | 245 +++ + .../coredump/coredump_socket_protocol_test.c | 1983 +++++++++++++++----- + tools/testing/selftests/coredump/coredump_test.h | 32 +- + .../selftests/coredump/coredump_test_helpers.c | 1742 ++++++++++++++++- + .../selftests/coredump/coredump_test_helpers.h | 79 + + tools/testing/selftests/exec/Makefile | 4 + + tools/testing/selftests/exec/binfmt_misc_delim.c | 127 ++ + tools/testing/selftests/filesystems/.gitignore | 2 +- + tools/testing/selftests/filesystems/Makefile | 2 +- + .../selftests/filesystems/file_stressor/.gitignore | 2 + + .../selftests/filesystems/file_stressor/Makefile | 6 + + .../{ => file_stressor}/file_stressor.c | 0 + .../selftests/filesystems/file_stressor/settings | 3 + + .../selftests/filesystems/fscontext_ns/.gitignore | 2 + + .../testing/selftests/filesystems/fuse/.gitignore | 1 + + tools/testing/selftests/filesystems/fuse/Makefile | 2 +- + tools/testing/selftests/filesystems/kernfs_test.c | 16 +- + .../selftests/filesystems/open_o_creat_o_dir.c | 201 ++ + tools/testing/selftests/filesystems/wrappers.h | 11 + + 268 files changed, 9211 insertions(+), 3937 deletions(-) + create mode 100644 Documentation/filesystems/exfat.rst + delete mode 100644 arch/powerpc/platforms/cell/spufs/coredump.c + create mode 100644 fs/nfsd/nfs4ctl.h + create mode 100644 fs/nfsd/nfserr.h + create mode 100644 fs/smb/server/tests/Kconfig + create mode 100644 fs/smb/server/tests/Makefile + create mode 100644 fs/smb/server/tests/smbacl_kunit.c + create mode 100644 include/linux/nfs_fh.h + create mode 100644 include/linux/nfsd_ssc.h + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal.h + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_helper.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_test_helpers.h + create mode 100644 tools/testing/selftests/exec/binfmt_misc_delim.c + create mode 100644 tools/testing/selftests/filesystems/file_stressor/.gitignore + create mode 100644 tools/testing/selftests/filesystems/file_stressor/Makefile + rename tools/testing/selftests/filesystems/{ => file_stressor}/file_stressor.c (100%) + create mode 100644 tools/testing/selftests/filesystems/file_stressor/settings + create mode 100644 tools/testing/selftests/filesystems/fscontext_ns/.gitignore + create mode 100644 tools/testing/selftests/filesystems/open_o_creat_o_dir.c +Merging printk/for-next (4f85c1c6efdb6 Merge branch 'for-7.3-irq-work-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/printk/linux.git printk/for-next +Merge made by the 'ort' strategy. + include/linux/console.h | 2 +- + kernel/printk/nbcon.c | 6 ++++-- + kernel/printk/printk.c | 2 +- + 3 files changed, 6 insertions(+), 4 deletions(-) +Merging pci/next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git pci/next +Already up to date. +Merging pstore/for-next/pstore (24b8f8dcb9a13 pstore/ftrace: Factor KASLR offset in the core kernel instruction addresses) +$ git merge -m Merge branch 'for-next/pstore' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git pstore/for-next/pstore +Already up to date. +Merging hid/for-next (8dd52b64d0822 Merge branch 'for-7.3/upstream-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/hid/hid.git hid/for-next +Merge made by the 'ort' strategy. + drivers/hid/Kconfig | 2 +- + drivers/hid/bpf/hid_bpf_struct_ops.c | 23 ++++++++-- + drivers/hid/hid-hyperv.c | 4 +- + drivers/hid/hid-ids.h | 1 + + drivers/hid/hid-multitouch.c | 19 ++++---- + drivers/hid/hid-rmi.c | 46 +++++++++++++++++-- + drivers/hid/i2c-hid/i2c-hid-core.c | 2 + + drivers/hid/wacom_wac.c | 13 ++++++ + tools/testing/selftests/hid/hid_bpf.c | 51 ++++++++++++++++++---- + tools/testing/selftests/hid/progs/hid.c | 26 +++++++++++ + .../testing/selftests/hid/progs/hid_bpf_helpers.h | 3 ++ + 11 files changed, 161 insertions(+), 29 deletions(-) +Merging i2c/i2c/for-next (8cd9520d35a6c Linux 7.1) +$ git merge -m Merge branch 'i2c/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wsa/linux.git i2c/i2c/for-next +Already up to date. +$ git am -3 ../patches/0001-i2c-Fix-up-the-rest-of-the-merge.patch +Applying: i2c: Fix up the rest of the merge +Using index info to reconstruct a base tree... +M drivers/i2c/i2c-core-base.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/i2c/i2c-core-base.c +No changes -- Patch already applied. +Merging i2c-andi/i2c/i2c-next (1f3e66348d252 Merge branch 'i2c/i2c' into i2c/i2c-next) +$ git merge -m Merge branch 'i2c/i2c-next' of https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git i2c-andi/i2c/i2c-next +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/i2c/xlnx,xps-iic-2.00.a.yaml | 2 +- + drivers/i2c/busses/i2c-bcm2835.c | 2 +- + drivers/i2c/busses/i2c-qcom-geni.c | 3 +-- + drivers/i2c/i2c-core-acpi.c | 1 + + 4 files changed, 4 insertions(+), 4 deletions(-) +Merging i2c-rust/rust-i2c-next (61ddec70c9bcc i2c: rust: mark I2cAdapter methods as inline) +$ git merge -m Merge branch 'rust-i2c-next' of https://github.com/ikrtn/rust-for-linux i2c-rust/rust-i2c-next +Already up to date. +Merging i3c/i3c/next (cab40cfc9e116 i3c: dw: reduce do_daa time if there's no client) +$ git merge -m Merge branch 'i3c/next' of https://git.kernel.org/pub/scm/linux/kernel/git/i3c/linux.git i3c/i3c/next +Already up to date. +Merging dmi/dmi-for-next (1afafbaf749d8 firmware/dmi: Include product_family info to modalias) +$ git merge -m Merge branch 'dmi-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jdelvare/staging.git dmi/dmi-for-next +Already up to date. +Merging hwmon-staging/hwmon-next (ba08432bda66a hwmon: (pmbus/tps53679) Add support for TPS53622 and TPS53659) +$ git merge -m Merge branch 'hwmon-next' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git hwmon-staging/hwmon-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-class-hwmon | 36 ++ + .../devicetree/bindings/hwmon/national,lm90.yaml | 28 +- + .../bindings/hwmon/pmbus/ti,tps25990.yaml | 8 +- + .../devicetree/bindings/hwmon/ti,tmp102.yaml | 16 +- + .../devicetree/bindings/trivial-devices.yaml | 8 +- + Documentation/hwmon/asus_ec_sensors.rst | 1 + + Documentation/hwmon/honor-fmi.rst | 32 ++ + Documentation/hwmon/index.rst | 2 + + Documentation/hwmon/it87.rst | 8 + + Documentation/hwmon/minisforum-um780xtx.rst | 68 +++ + Documentation/hwmon/sht4x.rst | 17 +- + Documentation/hwmon/sysfs-interface.rst | 12 + + Documentation/hwmon/tps25990.rst | 15 +- + Documentation/hwmon/tps53679.rst | 24 +- + Documentation/hwmon/yogafan.rst | 3 +- + MAINTAINERS | 14 + + drivers/hwmon/Kconfig | 26 ++ + drivers/hwmon/Makefile | 2 + + drivers/hwmon/asus-ec-sensors.c | 2 + + drivers/hwmon/dell-smm-hwmon.c | 16 + + drivers/hwmon/honor-fmi.c | 178 ++++++++ + drivers/hwmon/hwmon.c | 6 + + drivers/hwmon/it87.c | 308 ++++++++++--- + drivers/hwmon/lm90.c | 5 + + drivers/hwmon/minisforum-um780xtx.c | 501 +++++++++++++++++++++ + drivers/hwmon/pmbus/Kconfig | 5 +- + drivers/hwmon/pmbus/tps25990.c | 249 +++++++--- + drivers/hwmon/pmbus/tps53679.c | 9 +- + drivers/hwmon/sht4x.c | 62 ++- + drivers/hwmon/spd5118.c | 67 +-- + drivers/hwmon/yogafan.c | 19 + + include/linux/hwmon.h | 12 + + 32 files changed, 1567 insertions(+), 192 deletions(-) + create mode 100644 Documentation/hwmon/honor-fmi.rst + create mode 100644 Documentation/hwmon/minisforum-um780xtx.rst + create mode 100644 drivers/hwmon/honor-fmi.c + create mode 100644 drivers/hwmon/minisforum-um780xtx.c +Merging jc_docs/docs-next (ab2704c2a8840 docs: sysctl: timer_migration is for hrtimer only) +$ git merge -m Merge branch 'docs-next' of git://git.lwn.net/linux.git jc_docs/docs-next +Auto-merging Documentation/admin-guide/sysctl/kernel.rst +Auto-merging Documentation/filesystems/proc.rst +Merge made by the 'ort' strategy. + Documentation/admin-guide/parport.rst | 2 +- + Documentation/admin-guide/sysctl/kernel.rst | 6 ++-- + Documentation/core-api/cpu_hotplug.rst | 4 +-- + Documentation/core-api/debug-objects.rst | 2 +- + Documentation/core-api/dma-attributes.rst | 2 +- + Documentation/core-api/irq/irq-affinity.rst | 2 +- + Documentation/core-api/real-time/differences.rst | 10 +++++- + Documentation/core-api/swiotlb.rst | 2 +- + Documentation/core-api/this_cpu_ops.rst | 6 ++-- + Documentation/core-api/xarray.rst | 4 ++- + Documentation/filesystems/proc.rst | 10 +++--- + Documentation/process/1.Intro.rst | 2 +- + Documentation/process/2.Process.rst | 18 +++++----- + Documentation/process/3.Early-stage.rst | 2 +- + Documentation/process/5.Posting.rst | 10 +++--- + Documentation/process/7.AdvancedTopics.rst | 36 ++++++++++---------- + Documentation/process/backporting.rst | 18 +++++----- + .../process/embargoed-hardware-issues.rst | 2 +- + Documentation/process/maintainer-pgp-guide.rst | 38 +++++++++++----------- + 19 files changed, 94 insertions(+), 82 deletions(-) +Merging v4l-dvb/next (ce92674d8d8b2 media: ipu-bridge: do not use the CVS device lookup for IVSC) +$ git merge -m Merge branch 'next' of git://linuxtv.org/media-ci/media-pending.git v4l-dvb/next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../bindings/clock/nxp,imx95-blk-ctl.yaml | 71 ++ + .../bindings/media/fsl,imx95-csi-formatter.yaml | 88 +++ + .../devicetree/bindings/media/i2c/hynix,hi846.yaml | 3 +- + .../bindings/media/i2c/ovti,ov08d10.yaml | 3 +- + .../devicetree/bindings/media/i2c/ovti,ov4689.yaml | 3 +- + .../devicetree/bindings/media/i2c/ovti,ov5675.yaml | 3 +- + .../devicetree/bindings/media/i2c/ovti,ov5693.yaml | 3 +- + .../bindings/media/i2c/ovti,ov64a40.yaml | 3 +- + .../devicetree/bindings/media/i2c/sony,imx111.yaml | 3 +- + .../devicetree/bindings/media/i2c/sony,imx355.yaml | 3 +- + .../devicetree/bindings/media/i2c/sony,imx415.yaml | 3 +- + .../devicetree/bindings/media/i2c/st,vd55g1.yaml | 3 +- + .../devicetree/bindings/media/i2c/st,vd56g3.yaml | 3 +- + .../bindings/media/i2c/thine,thp7312.yaml | 3 +- + .../bindings/media/nxp,imx8mq-mipi-csi2.yaml | 4 +- + .../bindings/media/video-interface-devices.yaml | 17 +- + MAINTAINERS | 24 + + drivers/media/i2c/cvs/v4l2.c | 13 +- + drivers/media/pci/cx88/cx88-input.c | 22 +- + drivers/media/pci/intel/ipu-bridge.c | 38 +- + drivers/media/pci/saa7134/saa7134-input.c | 7 +- + drivers/media/platform/nxp/Kconfig | 16 + + drivers/media/platform/nxp/Makefile | 1 + + drivers/media/platform/nxp/imx-jpeg/mxc-jpeg.c | 23 +- + drivers/media/platform/nxp/imx95-csi-formatter.c | 758 +++++++++++++++++++++ + .../media/platform/renesas/rzg2l-cru/rzg2l-core.c | 3 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-cru.h | 2 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-video.c | 13 +- + drivers/media/rc/bpf-lirc.c | 18 +- + drivers/media/rc/imon.c | 9 +- + drivers/media/rc/ir-hix5hd2.c | 5 +- + drivers/media/rc/meson-ir-tx.c | 17 +- + drivers/media/rc/rc-ir-raw.c | 47 +- + drivers/media/rc/rc-main.c | 19 +- + drivers/media/rc/redrat3.c | 32 +- + drivers/media/rc/streamzap.c | 1 + + drivers/media/rc/sunxi-cir.c | 2 +- + drivers/media/v4l2-core/v4l2-common.c | 20 +- + drivers/platform/x86/intel/int3472/discrete.c | 81 ++- + drivers/platform/x86/intel/int3472/tps68470.c | 2 +- + drivers/staging/media/ipu3/ipu3-css.c | 1 - + .../dt-bindings/media/video-interface-devices.h | 13 + + include/linux/platform_data/x86/int3472.h | 2 + + include/media/v4l2-common.h | 75 +- + 44 files changed, 1299 insertions(+), 181 deletions(-) + create mode 100644 Documentation/devicetree/bindings/media/fsl,imx95-csi-formatter.yaml + create mode 100644 drivers/media/platform/nxp/imx95-csi-formatter.c + create mode 100644 include/dt-bindings/media/video-interface-devices.h +Merging v4l-dvb-next/master (adc218676eef2 Linux 6.12) +$ git merge -m Merge branch 'master' of git://linuxtv.org/mchehab/media-next.git v4l-dvb-next/master +Already up to date. +Merging pm/linux-next (208027d2c8957 Merge branch 'acpi-bus' into linux-next) +$ git merge -m Merge branch 'linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rafael/linux-pm.git pm/linux-next +Merge made by the 'ort' strategy. + drivers/acpi/scan.c | 4 ---- + include/acpi/acpi_bus.h | 11 +++-------- + 2 files changed, 3 insertions(+), 12 deletions(-) +Merging cpufreq-arm/cpufreq/arm/linux-next (40bad7f0aaf00 cpufreq: sti: avoid NULL dereference in dev_err()) +$ git merge -m Merge branch 'cpufreq/arm/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git cpufreq-arm/cpufreq/arm/linux-next +Merge made by the 'ort' strategy. + drivers/cpufreq/airoha-cpufreq.c | 2 +- + drivers/cpufreq/sparc-us2e-cpufreq.c | 11 +++++------ + drivers/cpufreq/sti-cpufreq.c | 4 ++-- + drivers/cpufreq/tegra194-cpufreq.c | 2 +- + rust/kernel/cpufreq.rs | 18 ++++++++++++++---- + 5 files changed, 23 insertions(+), 14 deletions(-) +Merging cpupower/cpupower (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'cpupower' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux.git cpupower/cpupower +Already up to date. +Merging devfreq/devfreq-next (9a222650d9e70 PM / devfreq: Fix governor_store() failing when device has no current governor) +$ git merge -m Merge branch 'devfreq-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/linux.git devfreq/devfreq-next +Auto-merging drivers/devfreq/event/rockchip-dfi.c +Merge made by the 'ort' strategy. + drivers/devfreq/devfreq.c | 50 +++++++----------------------------- + drivers/devfreq/event/rockchip-dfi.c | 4 ++- + 2 files changed, 12 insertions(+), 42 deletions(-) +Merging pmdomain/next (685596d9996f2 pmdomain: Merge branch fixes into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git pmdomain/next +Merge made by the 'ort' strategy. + .../devicetree/bindings/power/qcom,rpmpd.yaml | 2 ++ + drivers/pmdomain/qcom/rpmhpd.c | 41 ++++++++++++++++++++++ + drivers/pmdomain/renesas/rcar-sysc.h | 12 +++++-- + include/dt-bindings/power/qcom,rpmhpd.h | 1 + + 4 files changed, 53 insertions(+), 3 deletions(-) +Merging opp/opp/linux-next (a5096d4927d1e opp: Use %pe to print symbolic error name) +$ git merge -m Merge branch 'opp/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git opp/opp/linux-next +Merge made by the 'ort' strategy. + drivers/opp/core.c | 24 ++++++++++++------------ + drivers/opp/of.c | 6 +++--- + 2 files changed, 15 insertions(+), 15 deletions(-) +Merging thermal/thermal/linux-next (b115930d716de thermal/drivers/armada: Fix missing bitfields include) +$ git merge -m Merge branch 'thermal/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/thermal/linux.git thermal/thermal/linux-next +Already up to date. +Merging rdma/for-next (6ea2157ce8c74 RDMA/hns: Support setting GSI QP SL via debugfs) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git rdma/for-next +Merge made by the 'ort' strategy. + drivers/infiniband/hw/hfi1/pcie.c | 14 ++--- + drivers/infiniband/hw/hns/hns_roce_debugfs.c | 55 ++++++++++++++++++++ + drivers/infiniband/hw/hns/hns_roce_debugfs.h | 1 + + drivers/infiniband/hw/hns/hns_roce_device.h | 2 + + drivers/infiniband/hw/hns/hns_roce_hw_v2.c | 16 ++++-- + drivers/infiniband/hw/ionic/ionic_controlpath.c | 20 ++++++-- + drivers/infiniband/hw/ionic/ionic_fw.h | 2 + + drivers/infiniband/hw/ionic/ionic_lif_cfg.c | 1 + + drivers/infiniband/hw/ionic/ionic_lif_cfg.h | 1 + + drivers/infiniband/hw/mana/cq.c | 26 ++++++---- + drivers/infiniband/hw/mana/device.c | 3 +- + drivers/infiniband/hw/mana/main.c | 32 +++++++++++- + drivers/infiniband/hw/mana/mana_ib.h | 26 ++++++++-- + drivers/infiniband/hw/mana/qp.c | 59 +++++++++++++++++---- + drivers/infiniband/hw/mlx5/data_direct.c | 10 ++-- + drivers/infiniband/sw/rxe/rxe_loc.h | 1 + + drivers/infiniband/sw/rxe/rxe_net.c | 68 +++++++++++++++++++++++-- + drivers/infiniband/sw/rxe/rxe_req.c | 2 +- + drivers/infiniband/sw/rxe/rxe_resp.c | 4 +- + drivers/infiniband/ulp/rtrs/rtrs-clt.c | 23 +++++---- + include/net/mana/gdma.h | 1 + + include/uapi/rdma/ionic-abi.h | 9 +++- + include/uapi/rdma/mana-abi.h | 11 ++++ + 23 files changed, 315 insertions(+), 72 deletions(-) +Merging net-next/main (b35d3d2fae305 octeon_ep: remove redundant memset in octep_setup_pfvf_mbox()) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next.git net-next/main +Auto-merging MAINTAINERS +Auto-merging drivers/net/bonding/bond_alb.c +Auto-merging drivers/net/ethernet/stmicro/stmmac/stmmac_main.c +Auto-merging net/ipv4/udp.c +Auto-merging net/packet/af_packet.c +Merge made by the 'ort' strategy. + .../bindings/net/dsa/motorcomm,yt921x.yaml | 23 + + .../devicetree/bindings/net/dsa/realtek.yaml | 33 + + .../devicetree/bindings/net/fsl,fman-dtsec.yaml | 4 - + .../bindings/net/realtek,rtl9301-mdio.yaml | 14 +- + Documentation/netlink/specs/rt-link.yaml | 6 +- + .../device_drivers/ethernet/stmicro/stmmac.rst | 1 - + Documentation/networking/proc_net_tcp.rst | 20 +- + MAINTAINERS | 2 +- + drivers/net/bonding/bond_alb.c | 63 +- + drivers/net/dsa/Kconfig | 10 +- + drivers/net/dsa/Makefile | 2 +- + drivers/net/dsa/motorcomm/Kconfig | 17 + + drivers/net/dsa/motorcomm/Makefile | 5 + + drivers/net/dsa/{yt921x.c => motorcomm/chip.c} | 228 +----- + drivers/net/dsa/{yt921x.h => motorcomm/chip.h} | 15 + + drivers/net/dsa/motorcomm/leds.c | 641 ++++++++++++++++ + drivers/net/dsa/motorcomm/leds.h | 118 +++ + drivers/net/dsa/motorcomm/smi.c | 180 +++++ + drivers/net/dsa/motorcomm/smi.h | 61 ++ + drivers/net/dsa/mt7530-mdio.c | 12 +- + drivers/net/dsa/mt7530.c | 832 ++++++++++----------- + drivers/net/dsa/mt7530.h | 221 +++--- + drivers/net/dsa/realtek/realtek.h | 3 + + drivers/net/dsa/realtek/rtl8365mb_main.c | 6 + + drivers/net/dsa/realtek/rtl83xx.c | 23 +- + .../ethernet/marvell/octeon_ep/octep_pfvf_mbox.c | 1 - + .../net/ethernet/marvell/octeontx2/nic/otx2_pf.c | 99 ++- + .../net/ethernet/marvell/octeontx2/nic/otx2_reg.h | 2 + + .../net/ethernet/marvell/octeontx2/nic/otx2_vf.c | 9 +- + drivers/net/ethernet/meta/fbnic/fbnic_fw.c | 9 +- + drivers/net/ethernet/meta/fbnic/fbnic_tlv.c | 13 +- + drivers/net/ethernet/stmicro/stmmac/common.h | 5 + + drivers/net/ethernet/stmicro/stmmac/dwmac4.h | 6 +- + drivers/net/ethernet/stmicro/stmmac/dwmac4_core.c | 79 +- + drivers/net/ethernet/stmicro/stmmac/dwmac4_dma.c | 2 + + .../net/ethernet/stmicro/stmmac/dwxgmac2_core.c | 18 - + drivers/net/ethernet/stmicro/stmmac/hwif.h | 3 - + drivers/net/ethernet/stmicro/stmmac/stmmac_main.c | 62 +- + .../net/ethernet/stmicro/stmmac/stmmac_selftests.c | 112 --- + drivers/net/hyperv/netvsc.c | 10 +- + drivers/net/mdio/Kconfig | 4 +- + drivers/net/mdio/mdio-realtek-rtl9300.c | 426 ++++++++++- + drivers/net/netdevsim/netdev.c | 10 +- + drivers/net/pcs/pcs-lynx.c | 6 +- + drivers/net/phy/air_en8811h.c | 15 +- + drivers/net/phy/phy_device.c | 270 ++++--- + drivers/net/phy/phy_led_triggers.c | 11 +- + drivers/net/phy/phylink.c | 5 + + drivers/usb/atm/cxacru.c | 16 +- + drivers/usb/atm/speedtch.c | 4 +- + include/linux/netdevice.h | 4 +- + include/linux/phy.h | 18 + + include/net/bond_alb.h | 14 +- + net/atm/proc.c | 7 +- + net/core/netdev-genl.c | 4 +- + net/core/skbuff.c | 32 +- + net/ipv4/ping.c | 5 +- + net/ipv4/raw.c | 4 +- + net/ipv4/tcp_ipv4.c | 13 +- + net/ipv4/udp.c | 5 +- + net/ipv6/datagram.c | 5 +- + net/ipv6/tcp_ipv6.c | 12 +- + net/key/af_key.c | 5 +- + net/netlink/af_netlink.c | 5 +- + net/packet/af_packet.c | 7 +- + net/phonet/socket.c | 5 +- + net/sctp/proc.c | 6 +- + net/unix/af_unix.c | 3 +- + tools/testing/selftests/drivers/net/Makefile | 9 +- + tools/testing/selftests/drivers/net/gro_hw.py | 13 + + .../selftests/drivers/net/{gro.py => gro_lib.py} | 68 +- + tools/testing/selftests/drivers/net/gro_lro.py | 14 + + tools/testing/selftests/drivers/net/gro_sw.py | 13 + + tools/testing/selftests/drivers/net/hw/Makefile | 2 +- + .../drivers/net/hw/{gro_hw.py => gro_stats.py} | 0 + tools/testing/selftests/drivers/net/pppoe_gro.py | 45 ++ + tools/testing/selftests/drivers/net/settings | 2 +- + .../selftests/net/lib/ksft_setup_loopback.sh | 2 +- + 78 files changed, 2803 insertions(+), 1256 deletions(-) + create mode 100644 drivers/net/dsa/motorcomm/Kconfig + create mode 100644 drivers/net/dsa/motorcomm/Makefile + rename drivers/net/dsa/{yt921x.c => motorcomm/chip.c} (96%) + rename drivers/net/dsa/{yt921x.h => motorcomm/chip.h} (99%) + create mode 100644 drivers/net/dsa/motorcomm/leds.c + create mode 100644 drivers/net/dsa/motorcomm/leds.h + create mode 100644 drivers/net/dsa/motorcomm/smi.c + create mode 100644 drivers/net/dsa/motorcomm/smi.h + create mode 100755 tools/testing/selftests/drivers/net/gro_hw.py + rename tools/testing/selftests/drivers/net/{gro.py => gro_lib.py} (92%) + mode change 100755 => 100644 + create mode 100755 tools/testing/selftests/drivers/net/gro_lro.py + create mode 100755 tools/testing/selftests/drivers/net/gro_sw.py + rename tools/testing/selftests/drivers/net/hw/{gro_hw.py => gro_stats.py} (100%) + create mode 100755 tools/testing/selftests/drivers/net/pppoe_gro.py +Merging bpf-next/for-next (d761934c9483e selftests/bpf: Convert lirc_mode2 to prog_tests and extend coverage) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git bpf-next/for-next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging arch/x86/net/bpf_jit_comp.c +Auto-merging include/linux/bpf.h +Auto-merging include/linux/filter.h +Auto-merging kernel/bpf/backtrack.c +CONFLICT (content): Merge conflict in kernel/bpf/backtrack.c +Auto-merging kernel/bpf/core.c +Auto-merging kernel/bpf/verifier.c +Auto-merging tools/testing/selftests/bpf/Makefile +Recorded preimage for 'kernel/bpf/backtrack.c' +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +Recorded resolution for 'kernel/bpf/backtrack.c'. +[master c37d54aa2116b] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/kernel-parameters.txt | 16 + + Documentation/bpf/kfuncs.rst | 74 +++ + Documentation/bpf/signing.rst | 283 ++++++++-- + arch/arm64/net/bpf_jit_comp.c | 5 + + arch/riscv/net/bpf_jit_comp64.c | 21 +- + arch/x86/net/bpf_jit_comp.c | 27 +- + include/linux/bpf.h | 21 +- + include/linux/bpf_verifier.h | 16 + + include/linux/btf.h | 1 + + include/linux/filter.h | 1 + + include/linux/key.h | 2 + + include/uapi/linux/keyctl.h | 1 + + kernel/bpf/Makefile | 3 + + kernel/bpf/backtrack.c | 17 +- + kernel/bpf/btf.c | 109 ++-- + kernel/bpf/cnum_defs.h | 37 +- + kernel/bpf/core.c | 5 + + kernel/bpf/keys.c | 73 +++ + kernel/bpf/liveness.c | 20 +- + kernel/bpf/stream.c | 55 +- + kernel/bpf/verifier.c | 432 ++++++++++++--- + security/keys/process_keys.c | 25 + + tools/bpf/bpftool/btf.c | 6 +- + tools/bpf/bpftool/main.c | 30 +- + tools/bpf/bpftool/main.h | 2 +- + tools/bpf/bpftool/sign.c | 24 +- + tools/lib/bpf/btf.c | 14 +- + tools/lib/bpf/libbpf.c | 2 +- + tools/lib/bpf/libbpf_internal.h | 4 +- + tools/lib/bpf/usdt.c | 4 + + tools/testing/selftests/bpf/.gitignore | 1 - + tools/testing/selftests/bpf/Makefile | 31 +- + tools/testing/selftests/bpf/bench.c | 6 + + .../testing/selftests/bpf/benchs/bench_libarena.c | 210 ++++++++ + .../selftests/bpf/benchs/run_bench_libarena.sh | 31 ++ + .../selftests/bpf/bpftool_btf_dump_sorted.expected | 48 ++ + .../bpf/bpftool_btf_dump_unsorted.expected | 48 ++ + tools/testing/selftests/bpf/bpftool_helpers.c | 7 +- + tools/testing/selftests/bpf/config | 2 + + tools/testing/selftests/bpf/libarena/Makefile | 31 +- + .../bpf/libarena/benchs/bench_malloc.bpf.c | 51 ++ + .../bpf/libarena/include/bpf_arena_spin_lock.h | 2 +- + .../selftests/bpf/libarena/include/bpf_atomic.h | 2 +- + .../selftests/bpf/libarena/include/bpf_may_goto.h | 1 + + .../bpf/libarena/include/libarena/bitmap.h | 30 +- + .../bpf/libarena/include/libarena/common.h | 71 +++ + .../bpf/libarena/selftests/test_bitmap.bpf.c | 3 + + .../testing/selftests/bpf/libarena/src/asan.bpf.c | 21 +- + .../selftests/bpf/libarena/src/bitmap.bpf.c | 18 - + .../testing/selftests/bpf/libarena/src/buddy.bpf.c | 44 +- + .../selftests/bpf/libarena/src/common.bpf.c | 27 +- + tools/testing/selftests/bpf/network_helpers.c | 46 +- + .../selftests/bpf/prog_tests/aggregate_ret.c | 53 ++ + .../selftests/bpf/prog_tests/bpftool_btf_dump.c | 178 ++++++ + tools/testing/selftests/bpf/prog_tests/btf_dump.c | 127 +++-- + .../testing/selftests/bpf/prog_tests/core_reloc.c | 10 +- + .../selftests/bpf/prog_tests/fexit_bpf2bpf.c | 16 + + .../selftests/bpf/prog_tests/global_data_init.c | 96 +++- + .../testing/selftests/bpf/prog_tests/lirc_mode2.c | 334 ++++++++++++ + .../bpf/prog_tests/prog_tests_framework.c | 23 + + .../selftests/bpf/prog_tests/signed_loader.c | 600 +++++++++++++++++---- + tools/testing/selftests/bpf/prog_tests/stream.c | 67 +++ + tools/testing/selftests/bpf/prog_tests/usdt.c | 54 ++ + tools/testing/selftests/bpf/prog_tests/verifier.c | 2 + + .../selftests/bpf/prog_tests/verify_pkcs7_sig.c | 4 +- + .../selftests/bpf/progs/aggregate_ret_func.c | 383 +++++++++++++ + .../selftests/bpf/progs/aggregate_ret_kfunc.c | 203 +++++++ + .../bpf/progs/aggregate_ret_kfunc_arena.c | 121 +++++ + .../selftests/bpf/progs/aggregate_ret_target.c | 29 + + tools/testing/selftests/bpf/progs/arena_kfunc.c | 3 +- + tools/testing/selftests/bpf/progs/bpf_misc.h | 10 +- + .../selftests/bpf/progs/compute_live_registers.c | 36 ++ + .../testing/selftests/bpf/progs/exceptions_fail.c | 2 +- + .../selftests/bpf/progs/freplace_ret_pair.c | 12 + + tools/testing/selftests/bpf/progs/lirc_mode2.c | 32 ++ + tools/testing/selftests/bpf/progs/stack_arg_fail.c | 3 +- + .../selftests/bpf/progs/stack_arg_precision.c | 1 + + tools/testing/selftests/bpf/progs/stream.c | 28 + + .../selftests/bpf/progs/test_global_percpu_data.c | 10 +- + .../selftests/bpf/progs/test_lirc_mode2_kern.c | 26 - + .../selftests/bpf/progs/verifier_aggregate_ret.c | 179 ++++++ + tools/testing/selftests/bpf/progs/verifier_arena.c | 48 ++ + tools/testing/selftests/bpf/progs/verifier_bswap.c | 3 +- + tools/testing/selftests/bpf/progs/verifier_gotol.c | 1 + + tools/testing/selftests/bpf/progs/verifier_ldsx.c | 36 +- + .../selftests/bpf/progs/verifier_load_acquire.c | 1 + + tools/testing/selftests/bpf/progs/verifier_movsx.c | 3 +- + .../selftests/bpf/progs/verifier_percpu_addr.c | 1 + + .../selftests/bpf/progs/verifier_private_stack.c | 1 + + tools/testing/selftests/bpf/progs/verifier_sdiv.c | 3 +- + .../selftests/bpf/progs/verifier_stack_arg.c | 1 + + .../selftests/bpf/progs/verifier_stack_arg_order.c | 1 + + .../selftests/bpf/progs/verifier_store_release.c | 1 + + tools/testing/selftests/bpf/test_btf.h | 2 +- + .../testing/selftests/bpf/test_kmods/bpf_testmod.c | 127 +++++ + .../selftests/bpf/test_kmods/bpf_testmod_kfunc.h | 101 ++++ + tools/testing/selftests/bpf/test_lirc_mode2.sh | 41 -- + tools/testing/selftests/bpf/test_lirc_mode2_user.c | 177 ------ + tools/testing/selftests/bpf/test_loader.c | 9 + + tools/testing/selftests/bpf/testing_helpers.c | 43 ++ + tools/testing/selftests/bpf/testing_helpers.h | 3 + + tools/testing/selftests/bpf/usdt_2.c | 16 + + tools/testing/selftests/bpf/verify_sig_setup.sh | 63 ++- + tools/testing/selftests/bpf/vmtest.sh | 16 +- + 104 files changed, 4627 insertions(+), 774 deletions(-) + create mode 100644 kernel/bpf/keys.c + create mode 100644 tools/testing/selftests/bpf/benchs/bench_libarena.c + create mode 100755 tools/testing/selftests/bpf/benchs/run_bench_libarena.sh + create mode 100644 tools/testing/selftests/bpf/bpftool_btf_dump_sorted.expected + create mode 100644 tools/testing/selftests/bpf/bpftool_btf_dump_unsorted.expected + create mode 100644 tools/testing/selftests/bpf/libarena/benchs/bench_malloc.bpf.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/aggregate_ret.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpftool_btf_dump.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/lirc_mode2.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_func.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_kfunc.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_kfunc_arena.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_target.c + create mode 100644 tools/testing/selftests/bpf/progs/freplace_ret_pair.c + create mode 100644 tools/testing/selftests/bpf/progs/lirc_mode2.c + delete mode 100644 tools/testing/selftests/bpf/progs/test_lirc_mode2_kern.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_aggregate_ret.c + delete mode 100755 tools/testing/selftests/bpf/test_lirc_mode2.sh + delete mode 100644 tools/testing/selftests/bpf/test_lirc_mode2_user.c +$ git am -3 ../patches/0001-perf-Fixup-for-btf_vlan-API-change.patch +Applying: perf: Fixup for btf_vlan() API change +Using index info to reconstruct a base tree... +M tools/perf/builtin-trace.c +M tools/perf/util/btf.c +Falling back to patching base and 3-way merge... +Auto-merging tools/perf/builtin-trace.c +No changes -- Patch already applied. +Merging ipsec-next/master (298bb2b890332 Merge git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec-next.git ipsec-next/master +Already up to date. +Merging mlx5-next/mlx5-next (36b1d3299d0b6 net/mlx5: Add qp_latency_sensitive_disable cap bit) +$ git merge -m Merge branch 'mlx5-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mellanox/linux.git mlx5-next/mlx5-next +Already up to date. +Merging netfilter-next/main (91ec203513498 Merge tag 'net-next-7.3' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf-next.git netfilter-next/main +Already up to date. +Merging ipvs-next/main (91ec203513498 Merge tag 'net-next-7.3' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs-next.git ipvs-next/main +Already up to date. +Merging bluetooth/master (6696072ffe072 Bluetooth: L2CAP: refuse __l2cap_chan_add if chan already has conn) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth-next.git bluetooth/master +Auto-merging drivers/bluetooth/btintel.c +Auto-merging drivers/bluetooth/btintel_pcie.c +Auto-merging drivers/bluetooth/btmtk.c +Auto-merging drivers/bluetooth/btmtksdio.c +Auto-merging drivers/bluetooth/btnxpuart.c +Auto-merging drivers/bluetooth/btusb.c +Auto-merging drivers/bluetooth/hci_serdev.c +Auto-merging include/net/bluetooth/hci_core.h +Auto-merging include/net/bluetooth/l2cap.h +Auto-merging net/bluetooth/hci_core.c +Auto-merging net/bluetooth/hci_sync.c +CONFLICT (content): Merge conflict in net/bluetooth/hci_sync.c +Auto-merging net/bluetooth/l2cap_core.c +CONFLICT (content): Merge conflict in net/bluetooth/l2cap_core.c +Auto-merging net/bluetooth/l2cap_sock.c +Auto-merging net/bluetooth/mgmt.c +Resolved 'net/bluetooth/hci_sync.c' using previous resolution. +Recorded preimage for 'net/bluetooth/l2cap_core.c' +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +Recorded resolution for 'net/bluetooth/l2cap_core.c'. +[master 195f2733b052a] Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth-next.git +$ git diff -M --stat --summary HEAD^.. + drivers/bluetooth/btintel.c | 7 + + drivers/bluetooth/btintel.h | 1 + + drivers/bluetooth/btintel_pcie.c | 428 +++++++++++++++++++++++++++++-- + drivers/bluetooth/btintel_pcie.h | 77 +++++- + drivers/bluetooth/btmrvl_main.c | 4 +- + drivers/bluetooth/btmtk.c | 14 +- + drivers/bluetooth/btmtksdio.c | 35 ++- + drivers/bluetooth/btusb.c | 75 +++++- + drivers/bluetooth/hci_serdev.c | 46 +++- + drivers/bluetooth/virtio_bt.c | 21 +- + include/net/bluetooth/hci_core.h | 44 ++++ + include/net/bluetooth/l2cap.h | 102 +++++--- + net/bluetooth/6lowpan.c | 44 ++-- + net/bluetooth/bnep/core.c | 4 +- + net/bluetooth/hci_core.c | 2 +- + net/bluetooth/hci_sync.c | 64 ++--- + net/bluetooth/l2cap_core.c | 287 +++++++++++++++++++-- + net/bluetooth/l2cap_sock.c | 96 ++++--- + net/bluetooth/mgmt.c | 14 +- + scripts/context-analysis-suppression.txt | 1 + + 20 files changed, 1136 insertions(+), 230 deletions(-) +Merging wireless-next/for-next (1b78070aaef63 Merge tag 'net-7.3-rc1' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless-next.git wireless-next/for-next +Already up to date. +Merging ath-next/for-next (1b78070aaef63 Merge tag 'net-7.3-rc1' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git ath-next/for-next +Already up to date. +Merging iwlwifi-next/next (4a2610a5a9fcf wifi: iwlwifi: bump core version for BZ/SC/DR) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git iwlwifi-next/next +Already up to date. +Merging wpan-next/master (a6bfdfcc6711d ieee802154: allow legacy LLSEC ADD/DEL ops to pass strict validation) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git wpan-next/master +Already up to date. +Merging wpan-staging/staging (a6bfdfcc6711d ieee802154: allow legacy LLSEC ADD/DEL ops to pass strict validation) +$ git merge -m Merge branch 'staging' of https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git wpan-staging/staging +Already up to date. +Merging mtd/mtd/next (94d32f1ace8ee mtd: maps: remove dead select of MTD_CFI_BE_BYTE_SWAP) +$ git merge -m Merge branch 'mtd/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git mtd/mtd/next +Already up to date. +Merging nand/nand/next (15a3cbce32994 mtd: rawnand: sunxi: fix H6/H616 controller timings) +$ git merge -m Merge branch 'nand/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git nand/nand/next +Already up to date. +Merging spi-nor/spi-nor/next (df415c5e1de0f mtd: spi-nor: spansion: add die erase support in s28hx-t) +$ git merge -m Merge branch 'spi-nor/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git spi-nor/spi-nor/next +Already up to date. +Merging crypto/master (7537036a2e6fe crypto: lskcipher - propagate errors from unaligned crypt) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/herbert/cryptodev-2.6.git crypto/master +Already up to date. +Merging libcrypto/libcrypto-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'libcrypto-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git libcrypto/libcrypto-next +Already up to date. +Merging drm/drm-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'drm-next' of https://gitlab.freedesktop.org/drm/kernel.git drm/drm-next +Already up to date. +$ git am -3 ../patches/0001-drm-Fix-up-lut3d-mismerge.patch +Applying: drm: Fix up lut3d mismerge +Using index info to reconstruct a base tree... +M drivers/gpu/drm/drm_atomic.c +M include/drm/drm_colorop.h +Falling back to patching base and 3-way merge... +Auto-merging include/drm/drm_colorop.h +Auto-merging drivers/gpu/drm/drm_atomic.c +No changes -- Patch already applied. +$ git am -3 ../patches/0001-drm-amdgpu_dpm-Fix-up-commiting-unresolved-merge.patch +Applying: drm: amdgpu_dpm: Fix up commiting unresolved merge +Using index info to reconstruct a base tree... +M drivers/gpu/drm/amd/pm/amdgpu_dpm.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/gpu/drm/amd/pm/amdgpu_dpm.c +No changes -- Patch already applied. +Merging drm-exynos/for-linux-next (3a8660878839f Linux 6.18-rc1) +$ git merge -m Merge branch 'for-linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/daeinki/drm-exynos.git drm-exynos/for-linux-next +Already up to date. +Merging drm-misc/for-linux-next (99c95ce1b0708 drm/tidss: dispc: Switch to drm_fb_dma_get_gem_addr() for framebuffer addresses) +$ git merge -m Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/misc/kernel.git drm-misc/for-linux-next +Auto-merging MAINTAINERS +Auto-merging drivers/accel/amdxdna/amdxdna_gem.c +Auto-merging drivers/accel/ethosu/ethosu_job.c +Auto-merging drivers/dma-buf/dma-buf.c +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +Auto-merging drivers/gpu/drm/drm_atomic_uapi.c +Auto-merging drivers/gpu/drm/gud/gud_drv.c +Auto-merging drivers/gpu/drm/nouveau/nvkm/engine/device/base.c +Auto-merging drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/gsp.c +Auto-merging drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/rm.h +Auto-merging drivers/gpu/drm/tegra/dc.c +Auto-merging drivers/gpu/drm/tiny/cirrus-qemu.c +Auto-merging drivers/gpu/drm/virtio/virtgpu_display.c +Merge made by the 'ort' strategy. + .../bindings/display/panel/ilitek,ili7836a.yaml | 58 + + .../bindings/display/panel/novatek,nt36532.yaml | 83 ++ + .../panel/samsung,s6e8aa5x01-ams561ra01.yaml | 2 +- + Documentation/gpu/todo.rst | 15 - + MAINTAINERS | 22 + + drivers/accel/amdxdna/aie2_ctx.c | 9 +- + drivers/accel/amdxdna/amdxdna_gem.c | 197 ++- + drivers/accel/amdxdna/amdxdna_gem.h | 5 +- + drivers/accel/ethosu/ethosu_job.c | 4 +- + drivers/accel/ivpu/ivpu_drv.h | 2 +- + drivers/accel/qaic/qaic_data.c | 8 +- + drivers/accel/qaic/qaic_debugfs.c | 1 + + drivers/accel/qaic/qaic_drv.c | 1 + + drivers/accel/qaic/qaic_ras.c | 1 + + drivers/accel/qaic/qaic_ssr.c | 1 + + drivers/accel/qaic/qaic_timesync.c | 1 + + drivers/accel/qaic/sahara.c | 1 + + drivers/dma-buf/dma-buf.c | 2 +- + drivers/dma-buf/dma-resv.c | 2 +- + drivers/gpu/buddy.c | 1345 ++++++++++++++------ + drivers/gpu/drm/adp/adp_drv.c | 2 +- + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c | 28 +- + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.h | 7 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_color.c | 85 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_colorop.c | 27 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_colorop.h | 1 + + .../amd/display/amdgpu_dm/amdgpu_dm_connector.c | 255 +--- + .../amd/display/amdgpu_dm/tests/amdgpu_dm_test.c | 16 +- + drivers/gpu/drm/amd/display/dc/Makefile | 1 - + drivers/gpu/drm/amd/display/dc/dc_edid_parser.c | 80 -- + drivers/gpu/drm/amd/display/dc/dc_edid_parser.h | 44 - + drivers/gpu/drm/amd/display/dc/dce/dce_dmcu.c | 121 -- + drivers/gpu/drm/amd/display/dc/inc/hw/dmcu.h | 10 - + drivers/gpu/drm/amd/display/dmub/inc/dmub_cmd.h | 71 -- + .../drm/amd/display/modules/color/color_gamma.c | 3 +- + .../gpu/drm/arm/display/include/malidp_product.h | 10 +- + drivers/gpu/drm/arm/display/komeda/komeda_crtc.c | 27 +- + drivers/gpu/drm/arm/display/komeda/komeda_plane.c | 18 +- + .../drm/arm/display/komeda/komeda_wb_connector.c | 33 + + drivers/gpu/drm/arm/hdlcd_crtc.c | 4 +- + drivers/gpu/drm/arm/malidp_crtc.c | 18 +- + drivers/gpu/drm/arm/malidp_planes.c | 18 +- + drivers/gpu/drm/armada/armada_crtc.c | 2 +- + drivers/gpu/drm/ast/ast_dp.c | 24 +- + drivers/gpu/drm/ast/ast_mode.c | 18 +- + drivers/gpu/drm/atmel-hlcdc/atmel_hlcdc_crtc.c | 19 +- + drivers/gpu/drm/bridge/Kconfig | 7 +- + drivers/gpu/drm/bridge/analogix/analogix_dp_core.c | 4 +- + .../gpu/drm/bridge/cadence/cdns-mhdp8546-core.c | 1 - + drivers/gpu/drm/bridge/chipone-icn6211.c | 8 +- + drivers/gpu/drm/bridge/ite-it6505.c | 4 +- + drivers/gpu/drm/bridge/lontium-lt9611.c | 4 +- + drivers/gpu/drm/bridge/samsung-dsim.c | 4 +- + drivers/gpu/drm/bridge/sil-sii8620.c | 3 +- + drivers/gpu/drm/bridge/synopsys/dw-dp.c | 4 +- + drivers/gpu/drm/bridge/synopsys/dw-hdmi.c | 2 +- + drivers/gpu/drm/bridge/tc358767.c | 4 +- + drivers/gpu/drm/bridge/ti-sn65dsi83.c | 45 +- + drivers/gpu/drm/clients/drm_log.c | 54 +- + drivers/gpu/drm/display/drm_bridge_connector.c | 3 + + drivers/gpu/drm/display/drm_hdmi_state_helper.c | 156 +++ + drivers/gpu/drm/display/drm_scdc_helper.c | 289 ++++- + drivers/gpu/drm/drm_atomic.c | 4 + + drivers/gpu/drm/drm_atomic_uapi.c | 7 + + drivers/gpu/drm/drm_buddy.c | 3 +- + drivers/gpu/drm/drm_colorop.c | 107 ++ + drivers/gpu/drm/drm_debugfs.c | 157 --- + drivers/gpu/drm/drm_drv.c | 29 +- + drivers/gpu/drm/drm_edid.c | 189 ++- + drivers/gpu/drm/drm_gem_atomic_helper.c | 80 +- + drivers/gpu/drm/drm_of.c | 38 +- + drivers/gpu/drm/drm_simple_kms_helper.c | 37 +- + drivers/gpu/drm/exynos/exynos_drm_crtc.c | 2 +- + drivers/gpu/drm/exynos/exynos_drm_dpi.c | 3 +- + drivers/gpu/drm/exynos/exynos_drm_vidi.c | 3 +- + drivers/gpu/drm/exynos/exynos_hdmi.c | 3 +- + drivers/gpu/drm/fsl-dcu/fsl_dcu_drm_crtc.c | 2 +- + drivers/gpu/drm/fsl-dcu/fsl_dcu_drm_rgb.c | 10 +- + drivers/gpu/drm/gud/gud_drv.c | 2 +- + drivers/gpu/drm/hisilicon/hibmc/hibmc_drm_de.c | 2 +- + drivers/gpu/drm/hisilicon/kirin/dw_drm_dsi.c | 9 +- + drivers/gpu/drm/hisilicon/kirin/kirin_drm_ade.c | 2 +- + drivers/gpu/drm/hyperv/hyperv_drm.h | 1 - + drivers/gpu/drm/hyperv/hyperv_drm_modeset.c | 2 +- + drivers/gpu/drm/hyperv/hyperv_drm_proto.c | 44 +- + drivers/gpu/drm/imx/dc/dc-crtc.c | 2 +- + drivers/gpu/drm/imx/dc/dc-kms.c | 8 +- + drivers/gpu/drm/imx/dcss/dcss-crtc.c | 2 +- + drivers/gpu/drm/imx/dcss/dcss-plane.c | 2 +- + drivers/gpu/drm/imx/ipuv3/ipuv3-crtc.c | 18 +- + drivers/gpu/drm/ingenic/ingenic-drm-drv.c | 4 +- + drivers/gpu/drm/ingenic/ingenic-ipu.c | 2 +- + drivers/gpu/drm/kmb/kmb_crtc.c | 2 +- + drivers/gpu/drm/kmb/kmb_dsi.c | 9 +- + drivers/gpu/drm/mediatek/mtk_dsi.c | 10 +- + drivers/gpu/drm/meson/meson_crtc.c | 2 +- + drivers/gpu/drm/meson/meson_encoder_cvbs.c | 11 +- + drivers/gpu/drm/meson/meson_encoder_dsi.c | 11 +- + drivers/gpu/drm/meson/meson_encoder_hdmi.c | 11 +- + drivers/gpu/drm/meson/meson_overlay.c | 2 +- + drivers/gpu/drm/meson/meson_plane.c | 2 +- + drivers/gpu/drm/mgag200/mgag200_drv.h | 4 +- + drivers/gpu/drm/mgag200/mgag200_mode.c | 15 +- + drivers/gpu/drm/msm/disp/dpu1/dpu_crtc.c | 18 +- + drivers/gpu/drm/msm/disp/dpu1/dpu_plane.c | 18 +- + drivers/gpu/drm/msm/disp/mdp4/mdp4_crtc.c | 2 +- + drivers/gpu/drm/msm/disp/mdp4/mdp4_plane.c | 2 +- + drivers/gpu/drm/msm/disp/mdp5/mdp5_crtc.c | 20 +- + drivers/gpu/drm/msm/disp/mdp5/mdp5_plane.c | 16 +- + drivers/gpu/drm/mxsfb/lcdif_kms.c | 17 +- + drivers/gpu/drm/mxsfb/mxsfb_kms.c | 4 +- + drivers/gpu/drm/nouveau/dispnv04/dfp.c | 5 +- + drivers/gpu/drm/nouveau/dispnv50/disp.c | 4 +- + drivers/gpu/drm/nouveau/dispnv50/head.c | 14 +- + drivers/gpu/drm/nouveau/dispnv50/headca7d.c | 21 +- + .../gpu/drm/nouveau/include/nvhw/class/clca7d.h | 4 + + drivers/gpu/drm/nouveau/include/nvkm/subdev/pci.h | 1 + + drivers/gpu/drm/nouveau/nouveau_abi16.c | 4 + + drivers/gpu/drm/nouveau/nouveau_acpi.c | 32 +- + drivers/gpu/drm/nouveau/nouveau_acpi.h | 10 +- + drivers/gpu/drm/nouveau/nouveau_connector.c | 151 ++- + drivers/gpu/drm/nouveau/nouveau_connector.h | 12 +- + drivers/gpu/drm/nouveau/nouveau_dp.c | 4 +- + drivers/gpu/drm/nouveau/nouveau_drm.c | 114 +- + drivers/gpu/drm/nouveau/nouveau_fence.c | 2 +- + drivers/gpu/drm/nouveau/nouveau_svm.c | 13 + + drivers/gpu/drm/nouveau/nvkm/engine/device/base.c | 2 +- + drivers/gpu/drm/nouveau/nvkm/subdev/clk/gk20a.h | 4 +- + .../gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/fbsr.c | 2 +- + .../gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/gsp.c | 8 +- + .../gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/fbsr.c | 8 +- + .../gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/gsp.c | 3 +- + .../drm/nouveau/nvkm/subdev/gsp/rm/r570/nvrm/gsp.h | 8 + + drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/rm.h | 2 +- + drivers/gpu/drm/nouveau/nvkm/subdev/pci/Kbuild | 1 + + drivers/gpu/drm/nouveau/nvkm/subdev/pci/mcp79.c | 35 + + drivers/gpu/drm/omapdrm/dss/dsi.c | 63 +- + drivers/gpu/drm/omapdrm/dss/hdmi.h | 6 + + drivers/gpu/drm/omapdrm/dss/hdmi4.c | 29 +- + drivers/gpu/drm/omapdrm/dss/hdmi5.c | 37 +- + drivers/gpu/drm/omapdrm/dss/hdmi_common.c | 14 + + drivers/gpu/drm/omapdrm/omap_crtc.c | 17 +- + drivers/gpu/drm/panel/Kconfig | 25 + + drivers/gpu/drm/panel/Makefile | 2 + + drivers/gpu/drm/panel/panel-edp.c | 3 + + drivers/gpu/drm/panel/panel-himax-hx83102.c | 24 +- + drivers/gpu/drm/panel/panel-himax-hx83112a.c | 26 +- + drivers/gpu/drm/panel/panel-himax-hx83112b.c | 23 +- + drivers/gpu/drm/panel/panel-himax-hx8394.c | 26 +- + drivers/gpu/drm/panel/panel-ilitek-ili7836a.c | 297 +++++ + drivers/gpu/drm/panel/panel-ilitek-ili9805.c | 21 +- + drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.c | 12 +- + drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.h | 1 - + drivers/gpu/drm/panel/panel-ilitek-ili9806e-dsi.c | 16 +- + drivers/gpu/drm/panel/panel-ilitek-ili9806e-spi.c | 6 - + drivers/gpu/drm/panel/panel-ilitek-ili9881c.c | 15 +- + drivers/gpu/drm/panel/panel-ilitek-ili9882t.c | 24 +- + drivers/gpu/drm/panel/panel-jdi-fhd-r63452.c | 19 +- + drivers/gpu/drm/panel/panel-jdi-lt070me05000.c | 32 +- + drivers/gpu/drm/panel/panel-leadtek-ltk050h3146w.c | 20 +- + drivers/gpu/drm/panel/panel-leadtek-ltk500hd1829.c | 20 +- + drivers/gpu/drm/panel/panel-novatek-nt36532.c | 431 +++++++ + drivers/gpu/drm/panel/panel-samsung-s6d16d0.c | 74 +- + drivers/gpu/drm/panel/panel-samsung-s6d7aa0.c | 20 +- + drivers/gpu/drm/panel/panel-samsung-s6e3fa7.c | 20 +- + drivers/gpu/drm/panel/panel-samsung-s6e3fc2x01.c | 20 +- + drivers/gpu/drm/panel/panel-samsung-s6e3ha2.c | 30 +- + drivers/gpu/drm/panel/panel-samsung-s6e3ha8.c | 20 +- + drivers/gpu/drm/panel/panel-samsung-s6e63j0x03.c | 31 +- + drivers/gpu/drm/panel/panel-samsung-s6e63m0-dsi.c | 13 +- + drivers/gpu/drm/panel/panel-samsung-s6e63m0-spi.c | 6 - + drivers/gpu/drm/panel/panel-samsung-s6e63m0.c | 12 +- + drivers/gpu/drm/panel/panel-samsung-s6e63m0.h | 1 - + .../drm/panel/panel-samsung-s6e88a0-ams427ap24.c | 20 +- + .../drm/panel/panel-samsung-s6e88a0-ams452ef01.c | 20 +- + drivers/gpu/drm/panel/panel-samsung-s6e8aa0.c | 19 +- + .../gpu/drm/panel/panel-samsung-s6e8fc0-m1906f9.c | 23 +- + drivers/gpu/drm/panel/panel-samsung-sofef00.c | 20 +- + drivers/gpu/drm/panel/panel-sharp-ls043t1le01.c | 31 +- + drivers/gpu/drm/panel/panel-sharp-ls060t1sx01.c | 20 +- + drivers/gpu/drm/panel/panel-sony-td4353-jdi.c | 20 +- + .../gpu/drm/panel/panel-sony-tulip-truly-nt35521.c | 20 +- + drivers/gpu/drm/panel/panel-visionox-r66451.c | 23 +- + drivers/gpu/drm/panel/panel-visionox-rm69299.c | 21 +- + drivers/gpu/drm/panfrost/panfrost_devfreq.c | 2 +- + drivers/gpu/drm/panfrost/panfrost_device.c | 1 - + drivers/gpu/drm/panfrost/panfrost_drv.c | 2 +- + drivers/gpu/drm/panthor/panthor_device.h | 286 ++--- + drivers/gpu/drm/panthor/panthor_fw.c | 22 +- + drivers/gpu/drm/panthor/panthor_gpu.c | 117 +- + drivers/gpu/drm/panthor/panthor_mmu.c | 39 +- + drivers/gpu/drm/panthor/panthor_pwr.c | 24 +- + drivers/gpu/drm/panthor/panthor_sched.c | 518 ++++---- + drivers/gpu/drm/panthor/panthor_trace.h | 38 + + drivers/gpu/drm/qxl/qxl_display.c | 2 +- + drivers/gpu/drm/renesas/rcar-du/rcar_du_crtc.c | 17 +- + drivers/gpu/drm/renesas/rz-du/rzg2l_du_crtc.c | 15 +- + drivers/gpu/drm/renesas/shmobile/shmob_drm_crtc.c | 12 +- + drivers/gpu/drm/rockchip/rockchip_drm_vop.c | 18 +- + drivers/gpu/drm/rockchip/rockchip_drm_vop2.c | 18 +- + drivers/gpu/drm/scheduler/sched_entity.c | 10 +- + drivers/gpu/drm/scheduler/tests/tests_basic.c | 2 +- + drivers/gpu/drm/sitronix/st7571.c | 2 +- + drivers/gpu/drm/sitronix/st7920.c | 12 +- + drivers/gpu/drm/solomon/ssd130x.c | 22 +- + drivers/gpu/drm/sprd/sprd_dpu.c | 2 +- + drivers/gpu/drm/sti/sti_crtc.c | 2 +- + drivers/gpu/drm/stm/ltdc.c | 4 +- + drivers/gpu/drm/sun4i/sun4i_crtc.c | 2 +- + drivers/gpu/drm/sun4i/sun4i_hdmi_enc.c | 1 + + drivers/gpu/drm/sun4i/sun8i_ui_layer.c | 2 +- + drivers/gpu/drm/sun4i/sun8i_vi_layer.c | 2 +- + drivers/gpu/drm/sysfb/drm_sysfb_helper.h | 8 +- + drivers/gpu/drm/sysfb/drm_sysfb_modeset.c | 64 +- + drivers/gpu/drm/tegra/dc.c | 18 +- + drivers/gpu/drm/tegra/dsi.c | 16 +- + drivers/gpu/drm/tegra/plane.c | 28 +- + drivers/gpu/drm/tegra/rgb.c | 15 +- + drivers/gpu/drm/tests/drm_kunit_helpers.c | 2 +- + drivers/gpu/drm/tidss/tidss_dispc.c | 38 +- + drivers/gpu/drm/tidss/tidss_dispc.h | 2 +- + drivers/gpu/drm/tidss/tidss_encoder.c | 10 +- + drivers/gpu/drm/tiny/appletbdrm.c | 14 +- + drivers/gpu/drm/tiny/bochs.c | 2 +- + drivers/gpu/drm/tiny/cirrus-qemu.c | 2 +- + drivers/gpu/drm/tiny/pixpaper.c | 2 +- + drivers/gpu/drm/tiny/sharp-memory.c | 2 +- + drivers/gpu/drm/ttm/ttm_bo.c | 3 +- + drivers/gpu/drm/udl/udl_modeset.c | 2 +- + drivers/gpu/drm/vboxvideo/vbox_mode.c | 2 +- + drivers/gpu/drm/vc4/tests/vc4_mock_crtc.c | 2 +- + drivers/gpu/drm/vc4/vc4_crtc.c | 14 +- + drivers/gpu/drm/vc4/vc4_drv.h | 2 +- + drivers/gpu/drm/vc4/vc4_hdmi.c | 1 + + drivers/gpu/drm/vc4/vc4_plane.c | 15 +- + drivers/gpu/drm/vc4/vc4_txp.c | 2 +- + drivers/gpu/drm/verisilicon/vs_crtc.c | 2 +- + drivers/gpu/drm/verisilicon/vs_cursor_plane.c | 8 +- + drivers/gpu/drm/verisilicon/vs_plane.c | 20 - + drivers/gpu/drm/verisilicon/vs_plane.h | 2 - + drivers/gpu/drm/verisilicon/vs_primary_plane.c | 7 +- + drivers/gpu/drm/virtio/virtgpu_display.c | 14 +- + drivers/gpu/drm/vkms/tests/gen_yuv_conversion.py | 87 ++ + drivers/gpu/drm/vkms/tests/vkms_format_test.c | 40 +- + drivers/gpu/drm/vkms/vkms_colorop.c | 66 +- + drivers/gpu/drm/vkms/vkms_composer.c | 6 + + drivers/gpu/drm/vkms/vkms_crtc.c | 18 +- + drivers/gpu/drm/vkms/vkms_formats.c | 64 +- + drivers/gpu/drm/vkms/vkms_formats.h | 2 +- + drivers/gpu/drm/vkms/vkms_plane.c | 55 +- + drivers/gpu/drm/vmwgfx/vmwgfx_kms.c | 22 +- + drivers/gpu/drm/vmwgfx/vmwgfx_kms.h | 2 +- + drivers/gpu/drm/vmwgfx/vmwgfx_ldu.c | 2 +- + drivers/gpu/drm/vmwgfx/vmwgfx_scrn.c | 2 +- + drivers/gpu/drm/vmwgfx/vmwgfx_stdu.c | 2 +- + drivers/gpu/drm/xlnx/zynqmp_kms.c | 16 +- + drivers/gpu/tests/gpu_buddy_test.c | 160 ++- + include/drm/display/drm_dp.h | 1 + + include/drm/display/drm_hdmi_state_helper.h | 3 + + include/drm/display/drm_scdc.h | 21 +- + include/drm/display/drm_scdc_helper.h | 103 +- + include/drm/drm_colorop.h | 127 ++ + include/drm/drm_connector.h | 69 +- + include/drm/drm_gem_atomic_helper.h | 11 +- + include/drm/drm_mipi_dbi.h | 2 +- + include/drm/drm_print.h | 3 + + include/drm/drm_simple_kms_helper.h | 6 +- + include/linux/gpu_buddy.h | 101 +- + include/sound/omap-hdmi-audio.h | 1 + + include/uapi/drm/drm_mode.h | 12 + + kernel/cgroup/dmem.c | 4 +- + sound/soc/ti/omap-hdmi.c | 52 +- + 272 files changed, 5652 insertions(+), 3322 deletions(-) + create mode 100644 Documentation/devicetree/bindings/display/panel/ilitek,ili7836a.yaml + create mode 100644 Documentation/devicetree/bindings/display/panel/novatek,nt36532.yaml + delete mode 100644 drivers/gpu/drm/amd/display/dc/dc_edid_parser.c + delete mode 100644 drivers/gpu/drm/amd/display/dc/dc_edid_parser.h + create mode 100644 drivers/gpu/drm/nouveau/nvkm/subdev/pci/mcp79.c + create mode 100644 drivers/gpu/drm/panel/panel-ilitek-ili7836a.c + create mode 100644 drivers/gpu/drm/panel/panel-novatek-nt36532.c + create mode 100755 drivers/gpu/drm/vkms/tests/gen_yuv_conversion.py +$ git am -3 ../patches/0001-drm-bridge-microchip-lvds-Rename-drm_atomic_state-to.patch +Applying: drm/bridge: microchip-lvds: Rename drm_atomic_state to drm_atomic_commit +Using index info to reconstruct a base tree... +M drivers/gpu/drm/bridge/microchip-lvds.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/gpu/drm/bridge/microchip-lvds.c +No changes -- Patch already applied. +$ git reset --hard HEAD^ +HEAD is now at 195f2733b052a Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth-next.git +Merging next-20260831 version of drm-misc +$ git merge -m next-20260831/drm-misc 271e90eb5f9ff34951647e5ed33c1775eebcca50 +Already up to date. +$ git am -3 ../patches/0001-drm-bridge-microchip-lvds-Rename-drm_atomic_state-to.patch +Applying: drm/bridge: microchip-lvds: Rename drm_atomic_state to drm_atomic_commit +Using index info to reconstruct a base tree... +M drivers/gpu/drm/bridge/microchip-lvds.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/gpu/drm/bridge/microchip-lvds.c +No changes -- Patch already applied. +Merging amdgpu/drm-next (8fad652e7235d drm/amdgpu: correct mcm_addr_lut value for gc v12_1) +$ git merge -m Merge branch 'drm-next' of https://gitlab.freedesktop.org/agd5f/linux.git amdgpu/drm-next +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_plane.c +Merge made by the 'ort' strategy. + drivers/gpu/drm/amd/amdgpu/Makefile | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu.h | 5 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_acp.c | 6 - + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v10.c | 3 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v10.h | 3 +- + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c | 36 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v9.c | 3 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v9.h | 3 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_device.c | 13 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_discovery.c | 316 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_discovery.h | 13 + + drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c | 19 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gfx.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.c | 15 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ih.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_imu.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_irq.c | 18 + + drivers/gpu/drm/amd/amdgpu/amdgpu_irq.h | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_isp.c | 6 - + drivers/gpu/drm/amd/amdgpu/amdgpu_jpeg.c | 38 + + drivers/gpu/drm/amd/amdgpu/amdgpu_jpeg.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_kms.c | 11 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_lsdma.c | 20 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_lsdma.h | 4 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mca.c | 6 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 93 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h | 11 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mmhub.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_psp.c | 307 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_psp.h | 59 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ras.c | 105 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ras.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ras_eeprom.c | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_reset.c | 7 + + drivers/gpu/drm/amd/amdgpu/amdgpu_sdma.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c | 41 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_umc.c | 46 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_umc.h | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_userq_fence.c | 52 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_virt.c | 39 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_virt.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vkms.c | 6 - + drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_sdma.c | 5 + + drivers/gpu/drm/amd/amdgpu/amdgpu_xcp.c | 5 + + drivers/gpu/drm/amd/amdgpu/amdgv_sriovmsg.h | 49 +- + drivers/gpu/drm/amd/amdgpu/cik.c | 7 - + drivers/gpu/drm/amd/amdgpu/cik_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/cik_sdma.c | 1 - + drivers/gpu/drm/amd/amdgpu/cz_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/dce_v10_0.c | 6 - + drivers/gpu/drm/amd/amdgpu/dce_v6_0.c | 6 - + drivers/gpu/drm/amd/amdgpu/dce_v8_0.c | 6 - + drivers/gpu/drm/amd/amdgpu/gfx_v10_0.c | 12 - + drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c | 12 - + drivers/gpu/drm/amd/amdgpu/gfx_v12_0.c | 12 - + drivers/gpu/drm/amd/amdgpu/gfx_v12_1.c | 200 +- + drivers/gpu/drm/amd/amdgpu/gfx_v12_1_pkt.h | 39 - + drivers/gpu/drm/amd/amdgpu/gfx_v6_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/gfx_v7_0.c | 11 - + drivers/gpu/drm/amd/amdgpu/gfx_v8_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/gfx_v9_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_3.c | 1 - + drivers/gpu/drm/amd/amdgpu/gfxhub_v12_1.c | 39 +- + drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c | 7 - + drivers/gpu/drm/amd/amdgpu/gmc_v11_0.c | 7 - + drivers/gpu/drm/amd/amdgpu/gmc_v12_0.c | 26 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_1.c | 81 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_1.h | 1 + + drivers/gpu/drm/amd/amdgpu/gmc_v6_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/gmc_v7_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/gmc_v8_0.c | 13 - + drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c | 7 - + drivers/gpu/drm/amd/amdgpu/iceland_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/ih_v6_0.c | 7 - + drivers/gpu/drm/amd/amdgpu/ih_v6_1.c | 7 - + drivers/gpu/drm/amd/amdgpu/ih_v7_0.c | 20 +- + drivers/gpu/drm/amd/amdgpu/imu_v12_1.c | 89 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v2_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v2_5.c | 2 - + drivers/gpu/drm/amd/amdgpu/jpeg_v3_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v4_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v4_0_3.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v4_0_5.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_1.c | 3 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_1.h | 8 - + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_2.c | 265 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_3_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/lsdma_v7_1.c | 27 +- + drivers/gpu/drm/amd/amdgpu/mes_userqueue.c | 10 +- + drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 11 +- + drivers/gpu/drm/amd/amdgpu/mes_v12_1.c | 75 +- + drivers/gpu/drm/amd/amdgpu/mmhub_v4_2_0.c | 372 +- + drivers/gpu/drm/amd/amdgpu/navi10_ih.c | 7 - + drivers/gpu/drm/amd/amdgpu/nbif_v6_3_1.c | 153 +- + drivers/gpu/drm/amd/amdgpu/nbio_v6_3_2.c | 18 +- + drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.c | 351 + + drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.h | 32 + + drivers/gpu/drm/amd/amdgpu/nv.c | 6 - + drivers/gpu/drm/amd/amdgpu/psp_v15_0_8.c | 100 + + drivers/gpu/drm/amd/amdgpu/sdma_v2_4.c | 13 - + drivers/gpu/drm/amd/amdgpu/sdma_v3_0.c | 13 - + drivers/gpu/drm/amd/amdgpu/sdma_v4_0.c | 16 - + drivers/gpu/drm/amd/amdgpu/sdma_v4_4_2.c | 16 - + drivers/gpu/drm/amd/amdgpu/sdma_v5_0.c | 16 - + drivers/gpu/drm/amd/amdgpu/sdma_v5_2.c | 16 - + drivers/gpu/drm/amd/amdgpu/sdma_v6_0.c | 16 - + drivers/gpu/drm/amd/amdgpu/sdma_v7_0.c | 16 - + drivers/gpu/drm/amd/amdgpu/sdma_v7_1.c | 45 +- + drivers/gpu/drm/amd/amdgpu/si.c | 6 - + drivers/gpu/drm/amd/amdgpu/si_dma.c | 1 - + drivers/gpu/drm/amd/amdgpu/si_ih.c | 1 - + drivers/gpu/drm/amd/amdgpu/soc15.c | 6 - + drivers/gpu/drm/amd/amdgpu/soc21.c | 6 - + drivers/gpu/drm/amd/amdgpu/soc24.c | 6 - + drivers/gpu/drm/amd/amdgpu/soc_v1_0.c | 476 +- + drivers/gpu/drm/amd/amdgpu/soc_v1_0.h | 3 + + drivers/gpu/drm/amd/amdgpu/ta_ras_if.h | 1 + + drivers/gpu/drm/amd/amdgpu/tonga_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/uvd_v3_1.c | 8 - + drivers/gpu/drm/amd/amdgpu/uvd_v4_2.c | 8 - + drivers/gpu/drm/amd/amdgpu/uvd_v5_0.c | 8 - + drivers/gpu/drm/amd/amdgpu/uvd_v6_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vce_v1_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vce_v2_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vce_v3_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vcn_v1_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vcn_v2_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vcn_v2_5.c | 2 - + drivers/gpu/drm/amd/amdgpu/vcn_v3_0.c | 16 - + drivers/gpu/drm/amd/amdgpu/vcn_v4_0.c | 23 - + drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c | 21 - + drivers/gpu/drm/amd/amdgpu/vcn_v4_0_5.c | 23 - + drivers/gpu/drm/amd/amdgpu/vcn_v5_0_0.c | 23 - + drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c | 19 - + drivers/gpu/drm/amd/amdgpu/vcn_v5_0_2.c | 351 +- + drivers/gpu/drm/amd/amdgpu/vega10_ih.c | 7 - + drivers/gpu/drm/amd/amdgpu/vega20_ih.c | 7 - + drivers/gpu/drm/amd/amdgpu/vi.c | 6 - + drivers/gpu/drm/amd/amdkfd/cwsr_trap_handler.h | 1290 +- + .../gpu/drm/amd/amdkfd/cwsr_trap_handler_gfx12.asm | 135 +- + drivers/gpu/drm/amd/amdkfd/kfd_device.c | 7 +- + .../gpu/drm/amd/amdkfd/kfd_device_queue_manager.c | 79 +- + .../gpu/drm/amd/amdkfd/kfd_device_queue_manager.h | 1 + + drivers/gpu/drm/amd/amdkfd/kfd_migrate.c | 5 +- + drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12_1.c | 23 +- + drivers/gpu/drm/amd/amdkfd/kfd_packet_manager_v9.c | 3 +- + drivers/gpu/drm/amd/amdkfd/kfd_process.c | 44 +- + .../gpu/drm/amd/amdkfd/kfd_process_queue_manager.c | 3 +- + drivers/gpu/drm/amd/amdkfd/kfd_svm.c | 3 +- + drivers/gpu/drm/amd/display/Kconfig | 9 + + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c | 8 - + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.h | 11 +- + .../amd/display/amdgpu_dm/amdgpu_dm_connector.c | 30 +- + .../amd/display/amdgpu_dm/amdgpu_dm_connector.h | 7 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.c | 10 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.h | 6 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_cursor.c | 44 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_cursor.h | 6 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_debugfs.c | 68 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_dmub.c | 47 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_dmub.h | 15 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c | 3 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c | 24 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.c | 3 + + .../amd/display/amdgpu_dm/amdgpu_dm_mst_types.c | 12 +- + .../amd/display/amdgpu_dm/amdgpu_dm_mst_types.h | 4 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_plane.c | 48 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_plane.h | 12 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_services.c | 35 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_wb.c | 72 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_wb.h | 16 + + .../gpu/drm/amd/display/amdgpu_dm/tests/Makefile | 4 +- + .../amdgpu_dm/tests/amdgpu_dm_connector_test.c | 3067 + + .../display/amdgpu_dm/tests/amdgpu_dm_crtc_test.c | 645 + + .../amdgpu_dm/tests/amdgpu_dm_cursor_test.c | 732 + + .../display/amdgpu_dm/tests/amdgpu_dm_dmub_test.c | 578 +- + .../amdgpu_dm/tests/amdgpu_dm_freesync_test.c | 407 + + .../amdgpu_dm/tests/amdgpu_dm_helpers_test.c | 879 +- + .../display/amdgpu_dm/tests/amdgpu_dm_irq_test.c | 1255 +- + .../amdgpu_dm/tests/amdgpu_dm_mst_types_test.c | 741 + + .../display/amdgpu_dm/tests/amdgpu_dm_plane_test.c | 703 +- + .../amdgpu_dm/tests/amdgpu_dm_services_test.c | 228 +- + .../amd/display/amdgpu_dm/tests/amdgpu_dm_test.c | 10 - + .../display/amdgpu_dm/tests/amdgpu_dm_wb_test.c | 330 + + .../drm/amd/display/dc/bios/command_table_helper.c | 7 +- + .../amd/display/dc/bios/command_table_helper2.c | 2 + + drivers/gpu/drm/amd/display/dc/clk_mgr/Makefile | 8 + + drivers/gpu/drm/amd/display/dc/clk_mgr/clk_mgr.c | 4 + + .../amd/display/dc/clk_mgr/dce112/dce112_clk_mgr.c | 12 +- + .../amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.c | 181 + + .../amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.h | 29 + + .../drm/amd/display/dc/clk_mgr/dcn10/rv1_clk_mgr.c | 9 +- + .../drm/amd/display/dc/clk_mgr/dcn10/rv2_clk_mgr.c | 6 +- + .../amd/display/dc/clk_mgr/dcn20/dcn20_clk_mgr.c | 12 +- + .../amd/display/dc/clk_mgr/dcn201/dcn201_clk_mgr.c | 6 +- + .../drm/amd/display/dc/clk_mgr/dcn21/rn_clk_mgr.c | 5 +- + .../amd/display/dc/clk_mgr/dcn30/dcn30_clk_mgr.c | 8 +- + .../drm/amd/display/dc/clk_mgr/dcn301/vg_clk_mgr.c | 8 +- + .../amd/display/dc/clk_mgr/dcn31/dcn31_clk_mgr.c | 10 +- + .../amd/display/dc/clk_mgr/dcn314/dcn314_clk_mgr.c | 10 +- + .../amd/display/dc/clk_mgr/dcn315/dcn315_clk_mgr.c | 14 +- + .../amd/display/dc/clk_mgr/dcn316/dcn316_clk_mgr.c | 10 +- + .../amd/display/dc/clk_mgr/dcn32/dcn32_clk_mgr.c | 18 +- + .../amd/display/dc/clk_mgr/dcn35/dcn35_clk_mgr.c | 12 +- + .../amd/display/dc/clk_mgr/dcn401/dcn401_clk_mgr.c | 18 +- + .../amd/display/dc/clk_mgr/dcn42/dcn42_clk_mgr.c | 26 +- + .../amd/display/dc/clk_mgr/dcn42/dcn42_clk_mgr.h | 5 + + .../amd/display/dc/clk_mgr/dcn42b/dcn42b_clk_mgr.c | 41 +- + .../gpu/drm/amd/display/dc/clk_mgr/dcn60/dalsmc.h | 344 +- + .../amd/display/dc/clk_mgr/dcn60/dcn60_clk_mgr.c | 124 +- + .../amd/display/dc/clk_mgr/dcn60/dcn60_clk_mgr.h | 17 +- + .../dc/clk_mgr/dcn60/dcn60_clk_mgr_smu_msg.c | 214 +- + .../dc/clk_mgr/dcn60/dcn60_clk_mgr_smu_msg.h | 167 +- + .../display/dc/clk_mgr/dcn60/dcn60_smu_driver_if.h | 76 + + drivers/gpu/drm/amd/display/dc/core/dc.c | 24 +- + .../gpu/drm/amd/display/dc/core/dc_hw_sequencer.c | 327 +- + .../gpu/drm/amd/display/dc/core/dc_link_exports.c | 22 + + drivers/gpu/drm/amd/display/dc/core/dc_resource.c | 86 + + drivers/gpu/drm/amd/display/dc/dc.h | 13 +- + drivers/gpu/drm/amd/display/dc/dc_dmub_srv.c | 72 +- + drivers/gpu/drm/amd/display/dc/dc_dmub_srv.h | 39 + + drivers/gpu/drm/amd/display/dc/dc_hw_types.h | 18 +- + drivers/gpu/drm/amd/display/dc/dc_types.h | 1 + + drivers/gpu/drm/amd/display/dc/dce/dce_aux.c | 12 +- + .../amd/display/dc/dio/dcn10/dcn10_link_encoder.c | 10 +- + .../display/dc/dio/dcn42/dcn42_dio_link_encoder.c | 4 +- + drivers/gpu/drm/amd/display/dc/dml2_0/Makefile | 3 +- + .../dc/dml2_0/dml21/dml21_translation_helper.c | 32 +- + .../drm/amd/display/dc/dml2_0/dml21/dml21_utils.c | 15 +- + .../dml2_0/dml21/inc/bounding_boxes/dcn42_soc_bb.h | 39 +- + .../dml21/inc/bounding_boxes/dcn42b_soc_bb.h | 35 +- + .../dml2_0/dml21/inc/bounding_boxes/dcn4_soc_bb.h | 27 +- + .../dml2_0/dml21/inc/dml_top_display_cfg_types.h | 1 + + .../dml2_0/dml21/inc/dml_top_soc_parameter_types.h | 44 +- + .../display/dc/dml2_0/dml21/inc/dml_top_types.h | 1 + + .../dc/dml2_0/dml21/src/dml2_core/dml2_core_dcn4.c | 4 + + .../dml21/src/dml2_core/dml2_core_dcn4_calcs.c | 210 +- + .../dml21/src/dml2_core/dml2_core_dcn4_calcs.h | 1 + + .../src/dml2_core/dml2_core_dcn5_calcs_dchub.c | 2 +- + .../dml2_core/dml2_core_dcn5_funcs_initialize.c | 2 +- + .../dml2_core/dml2_core_dcn5_funcs_mode_support.c | 23 +- + .../src/dml2_core/dml2_core_dcn6_calcs_dchub.c | 47 +- + .../dml2_core/dml2_core_dcn6_funcs_initialize.c | 2 + + .../dml2_core/dml2_core_dcn6_funcs_mode_support.c | 34 +- + .../dml21/src/dml2_core/dml2_core_shared_types.h | 7 + + .../dc/dml2_0/dml21/src/dml2_dpmm/dml2_dpmm_dcn4.c | 106 +- + .../dc/dml2_0/dml21/src/dml2_top/dml2_top_utm.c | 2 +- + .../dml21/src/inc/dml2_internal_shared_types.h | 1 + + drivers/gpu/drm/amd/display/dc/gpio/hw_ddc.c | 7 +- + drivers/gpu/drm/amd/display/dc/gpio/hw_factory.c | 4 + + drivers/gpu/drm/amd/display/dc/gpio/hw_translate.c | 4 + + .../drm/amd/display/dc/hubbub/dcn10/dcn10_hubbub.h | 11 +- + .../drm/amd/display/dc/hubbub/dcn35/dcn35_hubbub.c | 7 + + .../drm/amd/display/dc/hubbub/dcn60/dcn60_hubbub.c | 206 +- + .../drm/amd/display/dc/hubbub/dcn60/dcn60_hubbub.h | 19 +- + .../gpu/drm/amd/display/dc/hubp/dcn60/dcn60_hubp.c | 3 + + drivers/gpu/drm/amd/display/dc/hwss/Makefile | 6 + + .../gpu/drm/amd/display/dc/hwss/dce/dce_hwseq.c | 46 +- + .../gpu/drm/amd/display/dc/hwss/dce/dce_hwseq.h | 12 +- + .../drm/amd/display/dc/hwss/dce110/dce110_hwseq.c | 25 +- + .../drm/amd/display/dc/hwss/dce60/dce60_hwseq.c | 8 +- + .../drm/amd/display/dc/hwss/dce80/dce80_hwseq.c | 3 +- + .../drm/amd/display/dc/hwss/dcn10/dcn10_hwseq.c | 126 +- + .../drm/amd/display/dc/hwss/dcn10/dcn10_hwseq.h | 13 +- + .../gpu/drm/amd/display/dc/hwss/dcn10/dcn10_init.c | 4 +- + .../drm/amd/display/dc/hwss/dcn20/dcn20_hwseq.c | 233 +- + .../drm/amd/display/dc/hwss/dcn20/dcn20_hwseq.h | 21 +- + .../gpu/drm/amd/display/dc/hwss/dcn20/dcn20_init.c | 3 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_hwseq.c | 34 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_hwseq.h | 5 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_init.c | 4 +- + .../gpu/drm/amd/display/dc/hwss/dcn21/dcn21_init.c | 3 +- + .../drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c | 51 +- + .../drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.h | 8 +- + .../gpu/drm/amd/display/dc/hwss/dcn30/dcn30_init.c | 3 +- + .../drm/amd/display/dc/hwss/dcn301/dcn301_init.c | 3 +- + .../gpu/drm/amd/display/dc/hwss/dcn31/dcn31_init.c | 3 +- + .../drm/amd/display/dc/hwss/dcn314/dcn314_hwseq.c | 8 +- + .../drm/amd/display/dc/hwss/dcn314/dcn314_init.c | 3 +- + .../drm/amd/display/dc/hwss/dcn32/dcn32_hwseq.c | 71 +- + .../drm/amd/display/dc/hwss/dcn32/dcn32_hwseq.h | 10 +- + .../gpu/drm/amd/display/dc/hwss/dcn32/dcn32_init.c | 3 +- + .../gpu/drm/amd/display/dc/hwss/dcn35/dcn35_init.c | 3 +- + .../drm/amd/display/dc/hwss/dcn351/dcn351_init.c | 3 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_hwseq.c | 201 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_hwseq.h | 12 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_init.c | 3 +- + .../drm/amd/display/dc/hwss/dcn42/dcn42_hwseq.c | 49 +- + .../drm/amd/display/dc/hwss/dcn42/dcn42_hwseq.h | 12 +- + .../gpu/drm/amd/display/dc/hwss/dcn42/dcn42_init.c | 3 +- + .../drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.c | 24 +- + .../gpu/drm/amd/display/dc/hwss/dcn60/dcn60_init.c | 3 +- + drivers/gpu/drm/amd/display/dc/hwss/hw_sequencer.h | 78 +- + .../drm/amd/display/dc/hwss/hw_sequencer_private.h | 21 +- + drivers/gpu/drm/amd/display/dc/inc/hw/dchubbub.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/hw_shared.h | 10 - + drivers/gpu/drm/amd/display/dc/inc/link_service.h | 5 +- + drivers/gpu/drm/amd/display/dc/irq/Makefile | 2 + + .../amd/display/dc/irq/dce110/irq_service_dce110.c | 21 - + .../amd/display/dc/irq/dce110/irq_service_dce110.h | 9 - + drivers/gpu/drm/amd/display/dc/irq/irq_service.c | 28 + + drivers/gpu/drm/amd/display/dc/irq/irq_service.h | 9 + + .../amd/display/dc/link/hwss/link_hwss_hpo_dp.c | 23 +- + .../gpu/drm/amd/display/dc/link/link_detection.c | 2 +- + drivers/gpu/drm/amd/display/dc/link/link_factory.c | 4 + + .../display/dc/link/protocols/link_dp_capability.c | 3 +- + .../dc/link/protocols/link_edp_panel_control.c | 30 + + .../dc/link/protocols/link_edp_panel_control.h | 4 + + .../amd/display/dc/link/protocols/link_hdmi_frl.c | 32 +- + drivers/gpu/drm/amd/display/dc/resource/Makefile | 4 + + .../display/dc/resource/dce112/dce112_resource.c | 62 - + .../amd/display/dc/resource/dcn42/dcn42_resource.c | 4 +- + .../display/dc/resource/dcn42b/dcn42b_resource.c | 4 +- + .../amd/display/dc/resource/dcn60/dcn60_resource.c | 13 +- + .../amd/display/dc/resource/dcn60/dcn60_resource.h | 9 +- + .../dcn401/dcn401_soc_and_ip_translator.c | 6 +- + .../dcn60/dcn60_soc_and_ip_translator.c | 6 +- + drivers/gpu/drm/amd/display/dmub/inc/dmub_cmd.h | 39 + + drivers/gpu/drm/amd/display/include/dal_asic_id.h | 5 + + .../gpu/drm/amd/display/modules/inc/mod_power.h | 1 + + drivers/gpu/drm/amd/include/amd_shared.h | 2 - + .../drm/amd/include/asic_reg/gc/gc_12_1_0_offset.h | 2 + + .../amd/include/asic_reg/gc/gc_12_1_0_sh_mask.h | 4 +- + .../amd/include/asic_reg/nbio/nbio_7_11_5_offset.h | 11074 ++++ + .../include/asic_reg/nbio/nbio_7_11_5_sh_mask.h | 63248 +++++++++++++++++++ + drivers/gpu/drm/amd/include/discovery.h | 179 +- + drivers/gpu/drm/amd/include/kgd_kfd_interface.h | 3 +- + drivers/gpu/drm/amd/pm/legacy-dpm/kv_dpm.c | 6 - + drivers/gpu/drm/amd/pm/legacy-dpm/si_dpm.c | 7 - + drivers/gpu/drm/amd/pm/powerplay/amd_powerplay.c | 6 - + drivers/gpu/drm/amd/pm/swsmu/amdgpu_smu.c | 19 +- + drivers/gpu/drm/amd/pm/swsmu/inc/amdgpu_smu.h | 13 +- + .../pm/swsmu/inc/pmfw_if/smu15_driver_if_v15_0_0.h | 46 - + drivers/gpu/drm/amd/pm/swsmu/inc/smu_v15_0.h | 6 - + .../gpu/drm/amd/pm/swsmu/smu13/smu_v13_0_6_ppt.c | 7 +- + drivers/gpu/drm/amd/pm/swsmu/smu15/smu_v15_0.c | 166 - + .../gpu/drm/amd/pm/swsmu/smu15/smu_v15_0_0_ppt.c | 270 +- + .../gpu/drm/amd/pm/swsmu/smu15/smu_v15_0_8_ppt.c | 11 +- + drivers/gpu/drm/amd/ras/core/Makefile | 9 +- + drivers/gpu/drm/amd/ras/core/aca.c | 341 +- + drivers/gpu/drm/amd/ras/core/aca.h | 44 +- + drivers/gpu/drm/amd/ras/core/aca_v1_0.c | 41 +- + drivers/gpu/drm/amd/ras/core/cmd.c | 145 +- + drivers/gpu/drm/amd/ras/core/cmd.h | 32 +- + drivers/gpu/drm/amd/ras/core/core.c | 309 +- + drivers/gpu/drm/amd/ras/core/eeprom.c | 504 +- + drivers/gpu/drm/amd/ras/core/eeprom.h | 27 +- + drivers/gpu/drm/amd/ras/core/eeprom_fw.c | 640 +- + drivers/gpu/drm/amd/ras/core/eeprom_fw.h | 67 +- + drivers/gpu/drm/amd/ras/core/log_ring.c | 56 +- + drivers/gpu/drm/amd/ras/core/log_ring.h | 46 +- + drivers/gpu/drm/amd/ras/core/ras.h | 129 +- + drivers/gpu/drm/amd/ras/core/ras_aca_v5_0.c | 448 + + drivers/gpu/drm/amd/ras/core/ras_aca_v5_0.h | 62 + + drivers/gpu/drm/amd/ras/core/ras_bert.c | 546 + + drivers/gpu/drm/amd/ras/core/ras_bert.h | 33 + + drivers/gpu/drm/amd/ras/core/ras_cper.c | 844 +- + drivers/gpu/drm/amd/ras/core/ras_cper.h | 177 +- + drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.c | 418 + + drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.h | 124 + + drivers/gpu/drm/amd/ras/core/ras_gfx.c | 10 +- + drivers/gpu/drm/amd/ras/core/ras_mce.c | 151 + + drivers/gpu/drm/amd/ras/core/ras_mce.h | 51 + + drivers/gpu/drm/amd/ras/core/ras_mp1.c | 156 +- + drivers/gpu/drm/amd/ras/core/ras_mp1.h | 52 +- + drivers/gpu/drm/amd/ras/core/ras_mp1_v13_0.c | 12 +- + drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.c | 249 + + drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.h | 30 + + drivers/gpu/drm/amd/ras/core/ras_nbio.c | 9 +- + drivers/gpu/drm/amd/ras/core/ras_process.c | 36 +- + drivers/gpu/drm/amd/ras/core/ras_psp.c | 599 +- + drivers/gpu/drm/amd/ras/core/ras_psp.h | 111 +- + drivers/gpu/drm/amd/ras/core/ras_psp_v13_0.c | 76 + + drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.c | 133 + + drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.h | 31 + + drivers/gpu/drm/amd/ras/core/ras_umc.c | 635 +- + drivers/gpu/drm/amd/ras/core/ras_umc.h | 52 +- + drivers/gpu/drm/amd/ras/core/ras_umc_v12_0.c | 143 +- + drivers/gpu/drm/amd/ras/core/ras_umc_v12_0.h | 3 - + drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.c | 220 + + drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.h | 69 + + drivers/gpu/drm/amd/ras/core/ta_if.h | 35 + + drivers/gpu/drm/amd/ras/ras_mgr/Makefile | 5 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.c | 191 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.h | 30 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_cmd.c | 38 + + .../drm/amd/ras/ras_mgr/amdgpu_ras_eeprom_i2c.c | 141 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.c | 267 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.h | 31 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.c | 348 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.h | 8 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1.c | 71 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1.h | 30 + + .../gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1_v13_0.c | 17 +- + .../gpu/drm/amd/ras/ras_mgr/amdgpu_ras_process.c | 9 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_sys.c | 100 +- + .../gpu/drm/amd/ras/ras_mgr/amdgpu_virt_ras_cmd.c | 42 +- + drivers/gpu/drm/amd/ras/ras_mgr/ras_sys.h | 2 + + include/uapi/drm/amdgpu_drm.h | 5 + + 402 files changed, 98674 insertions(+), 6412 deletions(-) + create mode 100644 drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.h + create mode 100644 drivers/gpu/drm/amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.c + create mode 100644 drivers/gpu/drm/amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.h + create mode 100644 drivers/gpu/drm/amd/display/dc/clk_mgr/dcn60/dcn60_smu_driver_if.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/nbio/nbio_7_11_5_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/nbio/nbio_7_11_5_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_aca_v5_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_aca_v5_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_bert.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_bert.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mce.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mce.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.c + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.h + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.c + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.h + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1.c + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1.h +$ git am -3 ../patches/0001-amdgpu-Fix-up-drm_atomic_commit-rename.patch +Applying: amdgpu: Fix up drm_atomic_commit rename +Using index info to reconstruct a base tree... +M drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +No changes -- Patch already applied. +Merging drm-intel/for-linux-next (0d63d6fc993b1 drm/i915/display: Reset use_flipq when duplicating crtc state) +$ git merge -m Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/i915/kernel.git drm-intel/for-linux-next +Auto-merging drivers/gpu/drm/i915/display/intel_cdclk.c +CONFLICT (content): Merge conflict in drivers/gpu/drm/i915/display/intel_cdclk.c +Auto-merging drivers/gpu/drm/i915/display/intel_cursor.c +Auto-merging drivers/gpu/drm/i915/display/intel_ddi.c +Auto-merging drivers/gpu/drm/i915/display/intel_dp.c +Auto-merging drivers/gpu/drm/i915/display/intel_dp_mst.c +Auto-merging drivers/gpu/drm/xe/Makefile +Resolved 'drivers/gpu/drm/i915/display/intel_cdclk.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 7b348431de78a] Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/i915/kernel.git +$ git diff -M --stat --summary HEAD^.. + drivers/gpu/drm/i915/display/intel_atomic.c | 1 + + drivers/gpu/drm/i915/display/intel_audio.c | 153 ++++++++ + drivers/gpu/drm/i915/display/intel_audio_regs.h | 16 +- + drivers/gpu/drm/i915/display/intel_cdclk.c | 326 ++++++++++------- + drivers/gpu/drm/i915/display/intel_cmtg.c | 2 +- + .../gpu/drm/i915/display/intel_crtc_state_dump.c | 3 + + drivers/gpu/drm/i915/display/intel_cursor.c | 296 +++++++++++---- + drivers/gpu/drm/i915/display/intel_cursor_regs.h | 8 +- + drivers/gpu/drm/i915/display/intel_ddi.c | 36 +- + drivers/gpu/drm/i915/display/intel_display.c | 76 +++- + drivers/gpu/drm/i915/display/intel_display.h | 3 +- + .../drm/i915/display/intel_display_clock_gating.c | 67 +++- + .../drm/i915/display/intel_display_clock_gating.h | 16 +- + .../gpu/drm/i915/display/intel_display_debugfs.c | 21 +- + .../drm/i915/display/intel_display_power_well.c | 6 +- + drivers/gpu/drm/i915/display/intel_display_regs.h | 18 +- + drivers/gpu/drm/i915/display/intel_display_types.h | 19 +- + drivers/gpu/drm/i915/display/intel_display_wa.c | 4 +- + drivers/gpu/drm/i915/display/intel_display_wa.h | 9 - + drivers/gpu/drm/i915/display/intel_dp.c | 96 +++-- + drivers/gpu/drm/i915/display/intel_dp_mst.c | 4 + + drivers/gpu/drm/i915/display/intel_dpll.c | 22 +- + drivers/gpu/drm/i915/display/intel_dpll_mgr.c | 221 +++++++++++- + drivers/gpu/drm/i915/display/intel_dpll_mgr.h | 22 ++ + drivers/gpu/drm/i915/display/intel_frontbuffer.c | 3 +- + drivers/gpu/drm/i915/display/intel_hdmi.c | 53 ++- + drivers/gpu/drm/i915/display/intel_hdmi.h | 11 +- + drivers/gpu/drm/i915/display/intel_hti.c | 3 - + .../gpu/drm/i915/display/intel_modeset_verify.c | 1 - + drivers/gpu/drm/i915/display/intel_parent.c | 10 +- + drivers/gpu/drm/i915/display/intel_parent.h | 3 +- + drivers/gpu/drm/i915/display/intel_psr.c | 18 +- + drivers/gpu/drm/i915/display/intel_snps_phy.c | 60 +--- + drivers/gpu/drm/i915/display/intel_snps_phy.h | 2 + + drivers/gpu/drm/i915/display/intel_tdf.h | 25 -- + drivers/gpu/drm/i915/display/intel_vrr.c | 395 ++++++++++++++++----- + drivers/gpu/drm/i915/display/intel_vrr.h | 2 + + drivers/gpu/drm/i915/gvt/handlers.c | 2 +- + drivers/gpu/drm/i915/i915_dpt.c | 1 - + drivers/gpu/drm/i915/i915_reg.h | 2 +- + drivers/gpu/drm/i915/intel_clock_gating.c | 30 +- + drivers/gpu/drm/i915/intel_gvt_mmio_table.c | 2 +- + drivers/gpu/drm/i915/intel_pcode.c | 14 +- + drivers/gpu/drm/i915/intel_pcode.h | 2 +- + drivers/gpu/drm/xe/Makefile | 3 +- + drivers/gpu/drm/xe/display/xe_display.c | 22 +- + drivers/gpu/drm/xe/display/xe_display_rpm.c | 2 - + drivers/gpu/drm/xe/display/xe_display_wa.c | 13 +- + drivers/gpu/drm/xe/display/xe_display_wa.h | 9 + + drivers/gpu/drm/xe/display/xe_tdf.c | 15 - + include/drm/intel/display_parent_interface.h | 15 +- + 51 files changed, 1544 insertions(+), 619 deletions(-) + delete mode 100644 drivers/gpu/drm/i915/display/intel_tdf.h + create mode 100644 drivers/gpu/drm/xe/display/xe_display_wa.h + delete mode 100644 drivers/gpu/drm/xe/display/xe_tdf.c +Merging drm-msm/msm-next (140b134753026 drm/msm: detach the ARM DMA mapping before attaching our own domain) +$ git merge -m Merge branch 'msm-next' of https://gitlab.freedesktop.org/drm/msm.git drm-msm/msm-next +Already up to date. +Merging drm-msm-lumag/msm-next-lumag (140b134753026 drm/msm: detach the ARM DMA mapping before attaching our own domain) +$ git merge -m Merge branch 'msm-next-lumag' of https://gitlab.freedesktop.org/lumag/msm.git drm-msm-lumag/msm-next-lumag +Already up to date. +Merging drm-xe/drm-xe-next (c874bc70c897c drm/xe/vram: add early VRAM health check) +$ git merge -m Merge branch 'drm-xe-next' of https://gitlab.freedesktop.org/drm/xe/kernel.git drm-xe/drm-xe-next +Auto-merging MAINTAINERS +Auto-merging drivers/gpu/drm/xe/Makefile +Merge made by the 'ort' strategy. + Documentation/gpu/drm-ras.rst | 39 ++ + Documentation/gpu/drm-uapi.rst | 93 +++- + Documentation/gpu/xe/index.rst | 1 + + Documentation/gpu/xe/xe_sigid.rst | 14 + + Documentation/netlink/specs/drm_ras.yaml | 80 +++ + MAINTAINERS | 13 + + drivers/gpu/drm/drm_drv.c | 2 + + drivers/gpu/drm/drm_ras.c | 278 +++++++++- + drivers/gpu/drm/drm_ras_nl.c | 33 ++ + drivers/gpu/drm/drm_ras_nl.h | 8 + + drivers/gpu/drm/xe/Makefile | 1 + + drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h | 1 + + drivers/gpu/drm/xe/abi/xe_log_abi.h | 199 +++++++ + drivers/gpu/drm/xe/abi/xe_sigid_abi.h | 172 +++++++ + drivers/gpu/drm/xe/display/xe_display_pcode.c | 4 +- + drivers/gpu/drm/xe/display/xe_dsb_buffer.c | 2 +- + drivers/gpu/drm/xe/display/xe_fb_pin.c | 2 +- + drivers/gpu/drm/xe/display/xe_panic.c | 3 +- + drivers/gpu/drm/xe/regs/xe_gt_regs.h | 6 + + drivers/gpu/drm/xe/regs/xe_lrc_layout.h | 2 + + drivers/gpu/drm/xe/tests/Makefile | 1 + + drivers/gpu/drm/xe/tests/xe_any_kunit.c | 213 ++++++++ + drivers/gpu/drm/xe/tests/xe_kunit_helpers.c | 4 + + drivers/gpu/drm/xe/tests/xe_log_kunit.c | 553 ++++++++++++++++++++ + drivers/gpu/drm/xe/xe_any.h | 137 +++++ + drivers/gpu/drm/xe/xe_debugfs.c | 15 + + drivers/gpu/drm/xe/xe_debugfs.h | 2 + + drivers/gpu/drm/xe/xe_defaults.h | 1 + + drivers/gpu/drm/xe/xe_device.c | 44 +- + drivers/gpu/drm/xe/xe_device.h | 2 +- + drivers/gpu/drm/xe/xe_device_types.h | 23 +- + drivers/gpu/drm/xe/xe_drm_ras.c | 65 +++ + drivers/gpu/drm/xe/xe_drm_ras.h | 3 + + drivers/gpu/drm/xe/xe_exec_queue.c | 6 +- + drivers/gpu/drm/xe/xe_exec_queue_types.h | 21 +- + drivers/gpu/drm/xe/xe_ggtt.c | 74 ++- + drivers/gpu/drm/xe/xe_gsc.c | 3 +- + drivers/gpu/drm/xe/xe_gt.c | 13 +- + drivers/gpu/drm/xe/xe_gt.h | 13 + + drivers/gpu/drm/xe/xe_gt_debugfs.c | 146 ++++++ + drivers/gpu/drm/xe/xe_gt_idle.c | 45 +- + drivers/gpu/drm/xe/xe_gt_printk.h | 3 + + drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c | 191 ++++--- + drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h | 6 + + drivers/gpu/drm/xe/xe_gt_sriov_printk.h | 3 + + drivers/gpu/drm/xe/xe_gt_stats.c | 7 + + drivers/gpu/drm/xe/xe_gt_stats_types.h | 22 + + drivers/gpu/drm/xe/xe_gt_types.h | 29 ++ + drivers/gpu/drm/xe/xe_guc.c | 18 +- + drivers/gpu/drm/xe/xe_guc_ct.c | 95 +++- + drivers/gpu/drm/xe/xe_guc_ct.h | 38 +- + drivers/gpu/drm/xe/xe_guc_exec_queue_types.h | 32 +- + drivers/gpu/drm/xe/xe_guc_pagefault.c | 39 +- + drivers/gpu/drm/xe/xe_guc_pc.c | 15 +- + drivers/gpu/drm/xe/xe_guc_submit.c | 267 +++++++--- + drivers/gpu/drm/xe/xe_guc_tlb_inval.c | 29 ++ + drivers/gpu/drm/xe/xe_guc_types.h | 6 + + drivers/gpu/drm/xe/xe_hwmon.c | 88 ++-- + drivers/gpu/drm/xe/xe_i2c.c | 22 +- + drivers/gpu/drm/xe/xe_i2c.h | 1 + + drivers/gpu/drm/xe/xe_log.c | 235 +++++++++ + drivers/gpu/drm/xe/xe_log.h | 194 +++++++ + drivers/gpu/drm/xe/xe_lrc.c | 33 ++ + drivers/gpu/drm/xe/xe_mert.c | 2 +- + drivers/gpu/drm/xe/xe_migrate.c | 123 ++++- + drivers/gpu/drm/xe/xe_migrate.h | 6 + + drivers/gpu/drm/xe/xe_module.c | 4 + + drivers/gpu/drm/xe/xe_module.h | 1 + + drivers/gpu/drm/xe/xe_oa.c | 13 +- + drivers/gpu/drm/xe/xe_pagefault.c | 716 +++++++++++++++++++++----- + drivers/gpu/drm/xe/xe_pagefault.h | 74 +++ + drivers/gpu/drm/xe/xe_pagefault_types.h | 114 +++- + drivers/gpu/drm/xe/xe_pci.c | 32 +- + drivers/gpu/drm/xe/xe_pci_error.c | 13 +- + drivers/gpu/drm/xe/xe_pcode.c | 27 +- + drivers/gpu/drm/xe/xe_pcode.h | 2 +- + drivers/gpu/drm/xe/xe_pm.c | 5 + + drivers/gpu/drm/xe/xe_printk.h | 3 + + drivers/gpu/drm/xe/xe_pt.c | 6 + + drivers/gpu/drm/xe/xe_pxp_submit.c | 5 +- + drivers/gpu/drm/xe/xe_ras.c | 236 ++++++++- + drivers/gpu/drm/xe/xe_ras.h | 2 + + drivers/gpu/drm/xe/xe_ras_types.h | 50 ++ + drivers/gpu/drm/xe/xe_sriov_printk.h | 3 + + drivers/gpu/drm/xe/xe_sriov_vf_ccs.c | 1 - + drivers/gpu/drm/xe/xe_survivability_mode.c | 85 +-- + drivers/gpu/drm/xe/xe_svm.c | 146 ++++-- + drivers/gpu/drm/xe/xe_svm.h | 59 ++- + drivers/gpu/drm/xe/xe_sysctrl_event.c | 28 +- + drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h | 4 + + drivers/gpu/drm/xe/xe_tile_printk.h | 3 + + drivers/gpu/drm/xe/xe_tile_sriov_printk.h | 3 + + drivers/gpu/drm/xe/xe_tile_types.h | 4 + + drivers/gpu/drm/xe/xe_tlb_inval.c | 42 +- + drivers/gpu/drm/xe/xe_tlb_inval.h | 1 + + drivers/gpu/drm/xe/xe_tlb_inval_types.h | 10 + + drivers/gpu/drm/xe/xe_userptr.c | 19 +- + drivers/gpu/drm/xe/xe_vm.c | 301 ++++++++--- + drivers/gpu/drm/xe/xe_vm_types.h | 42 +- + drivers/gpu/drm/xe/xe_vram.c | 258 +++++++++- + drivers/gpu/drm/xe/xe_vram.h | 10 + + drivers/gpu/drm/xe/xe_wa.c | 4 +- + include/drm/drm_device.h | 1 + + include/drm/drm_ras.h | 34 ++ + include/uapi/drm/drm_ras.h | 18 + + 105 files changed, 5516 insertions(+), 704 deletions(-) + create mode 100644 Documentation/gpu/xe/xe_sigid.rst + create mode 100644 drivers/gpu/drm/xe/abi/xe_log_abi.h + create mode 100644 drivers/gpu/drm/xe/abi/xe_sigid_abi.h + create mode 100644 drivers/gpu/drm/xe/tests/xe_any_kunit.c + create mode 100644 drivers/gpu/drm/xe/tests/xe_log_kunit.c + create mode 100644 drivers/gpu/drm/xe/xe_any.h + create mode 100644 drivers/gpu/drm/xe/xe_log.c + create mode 100644 drivers/gpu/drm/xe/xe_log.h +$ git am -3 ../patches/0001-drm-xe-Fix-up-merge-issue.patch +Applying: drm: xe: Fix up merge issue +Using index info to reconstruct a base tree... +M drivers/gpu/drm/xe/xe_ttm_vram_mgr.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/gpu/drm/xe/xe_ttm_vram_mgr.c +No changes -- Patch already applied. +Merging drm-rust/for-linux-next (6cb331644c441 gpu: nova-core: fix barrier usage in GSP->CPU messaging path) +$ git merge -m Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/rust/kernel.git drm-rust/for-linux-next +Merge made by the 'ort' strategy. + drivers/gpu/nova-core/falcon.rs | 2 +- + drivers/gpu/nova-core/firmware.rs | 2 +- + .../gpu/nova-core/firmware/{fsp.rs => gsp_fmc.rs} | 55 ++++++++------- + drivers/gpu/nova-core/fsp.rs | 29 ++++---- + drivers/gpu/nova-core/gpu.rs | 7 +- + drivers/gpu/nova-core/gpu/regs.rs | 82 ++++++++++++++++++++++ + drivers/gpu/nova-core/gsp/boot.rs | 10 +-- + drivers/gpu/nova-core/gsp/cmdq.rs | 41 ++++++----- + drivers/gpu/nova-core/regs.rs | 76 -------------------- + 9 files changed, 166 insertions(+), 138 deletions(-) + rename drivers/gpu/nova-core/firmware/{fsp.rs => gsp_fmc.rs} (71%) + create mode 100644 drivers/gpu/nova-core/gpu/regs.rs +Merging drm-nova/nova-next (93296e9d9528f gpu: nova-core: vbios: store reference to Device where relevant) +$ git merge -m Merge branch 'nova-next' of https://gitlab.freedesktop.org/drm/nova.git drm-nova/nova-next +Already up to date. +Merging etnaviv/etnaviv/next (6bde14ba5f7ef drm/etnaviv: add optional reset support) +$ git merge -m Merge branch 'etnaviv/next' of https://git.pengutronix.de/git/lst/linux etnaviv/etnaviv/next +Already up to date. +Merging fbdev/for-next (94e6a058b1682 fbcon: Fix KASAN slab-out-of-bounds Read in fbcon_prepare_logo) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/deller/linux-fbdev.git fbdev/for-next +Merge made by the 'ort' strategy. + drivers/tty/vt/vt.c | 9 +++++---- + drivers/video/fbdev/core/fbcon.c | 7 +++++++ + drivers/video/fbdev/omap2/omapfb/displays/panel-sony-acx565akm.c | 5 +++-- + 3 files changed, 15 insertions(+), 6 deletions(-) +$ git am -3 ../patches/0001-fix-up-for-drm-hyperv-Remove-reference-to-hyperv_fb-.patch +Applying: fix up for "drm/hyperv: Remove reference to hyperv_fb driver" +$ git reset HEAD^ +Unstaged changes after reset: +M drivers/gpu/drm/hyperv/Kconfig +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master 33eb7468272e0] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/deller/linux-fbdev.git + Date: Thu Sep 3 16:48:08 2026 +0100 +Merging regmap/for-next (92e6b920e7f4a Merge remote-tracking branch 'regmap/for-7.4' into regmap-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git regmap/for-next +Merge made by the 'ort' strategy. + drivers/base/regmap/internal.h | 1 + + drivers/base/regmap/regmap-kunit.c | 98 ++++++++++++++++++++++++++++++++++-- + drivers/base/regmap/regmap-ram.c | 26 +++++++--- + drivers/base/regmap/regmap-raw-ram.c | 21 +++++--- + 4 files changed, 129 insertions(+), 17 deletions(-) +Merging sound/for-next (bdeed06421576 ALSA: usb-audio: set Roland Capture rate during stream preparation) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git sound/for-next +Merge made by the 'ort' strategy. + sound/usb/quirks-table.h | 206 +++++++++++++++++++++++++++++++++++++++++++++-- + sound/usb/quirks.c | 74 +++++++++++++++++ + 2 files changed, 273 insertions(+), 7 deletions(-) +Merging ieee1394/for-next (68097f9cdd526 tools/firewire: nosy-dump: fix input file handle leak) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ieee1394/linux1394.git ieee1394/for-next +Merge made by the 'ort' strategy. + drivers/firewire/.kunitconfig | 1 + + drivers/firewire/Kconfig | 16 ++ + drivers/firewire/config-rom-generator-test.c | 407 +++++++++++++++++++++++++++ + drivers/firewire/config-rom-parser-test.c | 355 +++++++++++++++++++++++ + drivers/firewire/core-card.c | 4 + + drivers/firewire/core-device.c | 4 + + drivers/firewire/core-transaction.c | 4 + + drivers/firewire/ohci.c | 8 +- + include/linux/firewire.h | 4 +- + tools/firewire/nosy-dump.c | 10 +- + 10 files changed, 805 insertions(+), 8 deletions(-) + create mode 100644 drivers/firewire/config-rom-generator-test.c + create mode 100644 drivers/firewire/config-rom-parser-test.c +Merging sound-asoc/for-next (d687afbc2b9be Merge remote-tracking branch 'asoc/for-7.4' into asoc-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git sound-asoc/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../bindings/sound/davinci-evm-audio.txt | 49 - + .../bindings/sound/foursemi,fs2105s.yaml | 2 +- + .../devicetree/bindings/sound/nuvoton,nau8360.yaml | 115 + + .../bindings/sound/ti,da830-evm-audio.yaml | 83 + + MAINTAINERS | 3 + + include/sound/soc_sdw_utils.h | 20 + + include/sound/tas2781-dsp.h | 4 + + sound/soc/codecs/Kconfig | 24 + + sound/soc/codecs/Makefile | 4 + + sound/soc/codecs/ak4642.c | 5 +- + sound/soc/codecs/lpass-rx-macro.c | 2 +- + sound/soc/codecs/lpass-tx-macro.c | 2 +- + sound/soc/codecs/lpass-va-macro.c | 2 +- + sound/soc/codecs/lpass-wsa-macro.c | 2 +- + sound/soc/codecs/nau8360-dsp.c | 634 ++++++ + sound/soc/codecs/nau8360-dsp.h | 122 ++ + sound/soc/codecs/nau8360.c | 2300 ++++++++++++++++++++ + sound/soc/codecs/nau8360.h | 911 ++++++++ + sound/soc/codecs/sn624x-sdca-sdw.c | 1888 ++++++++++++++++ + sound/soc/codecs/sn624x-sdca.h | 152 ++ + sound/soc/codecs/sta32x.c | 6 +- + sound/soc/codecs/sta350.c | 6 +- + sound/soc/codecs/tas2764.c | 2 +- + sound/soc/codecs/tas2770.c | 2 +- + sound/soc/codecs/tas2781-comlib-i2c.c | 7 +- + sound/soc/codecs/tas2781-comlib.c | 22 +- + sound/soc/codecs/tas2781-fmwlib.c | 4 +- + sound/soc/codecs/tlv320aic32x4-clk.c | 2 +- + sound/soc/codecs/twl4030.c | 5 +- + sound/soc/codecs/wcd-mbhc-v2.c | 8 +- + sound/soc/codecs/wcd934x.c | 2 +- + sound/soc/fsl/fsl_mqs.c | 14 + + sound/soc/intel/boards/Kconfig | 1 + + sound/soc/intel/common/soc-acpi-intel-arl-match.c | 71 + + sound/soc/intel/common/soc-acpi-intel-lnl-match.c | 71 + + sound/soc/intel/common/soc-acpi-intel-mtl-match.c | 71 + + sound/soc/intel/common/soc-acpi-intel-ptl-match.c | 82 + + sound/soc/mediatek/mt2701/mt2701-afe-clock-ctrl.c | 61 +- + sound/soc/mediatek/mt2701/mt2701-afe-pcm.c | 4 +- + sound/soc/mediatek/mt2701/mt2701-cs42448.c | 4 +- + sound/soc/mediatek/mt2701/mt2701-wm8960.c | 4 +- + sound/soc/mediatek/mt7986/mt7986-afe-pcm.c | 8 +- + sound/soc/mediatek/mt7986/mt7986-dai-etdm.c | 2 +- + sound/soc/mediatek/mt7986/mt7986-wm8960.c | 4 +- + sound/soc/mediatek/mt8173/mt8173-afe-pcm.c | 24 +- + sound/soc/samsung/aries_wm8994.c | 14 +- + sound/soc/samsung/pcm.c | 19 +- + sound/soc/samsung/spdif.c | 8 +- + sound/soc/samsung/tm2_wm5110.c | 12 +- + sound/soc/sdw_utils/Makefile | 4 +- + sound/soc/sdw_utils/soc_sdw_senary_amp.c | 80 + + sound/soc/sdw_utils/soc_sdw_senary_dmic.c | 49 + + sound/soc/sdw_utils/soc_sdw_senary_sdca.c | 63 + + .../sdw_utils/soc_sdw_senary_sdca_jack_common.c | 197 ++ + sound/soc/sdw_utils/soc_sdw_utils.c | 166 ++ + sound/soc/soc-component.c | 8 +- + sound/soc/soc-ops.c | 4 +- + sound/soc/tegra/tegra210_adx.c | 12 +- + sound/soc/tegra/tegra210_adx.h | 2 +- + sound/soc/uniphier/aio-dma.c | 5 +- + 60 files changed, 7250 insertions(+), 204 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/sound/davinci-evm-audio.txt + create mode 100644 Documentation/devicetree/bindings/sound/nuvoton,nau8360.yaml + create mode 100644 Documentation/devicetree/bindings/sound/ti,da830-evm-audio.yaml + create mode 100644 sound/soc/codecs/nau8360-dsp.c + create mode 100644 sound/soc/codecs/nau8360-dsp.h + create mode 100644 sound/soc/codecs/nau8360.c + create mode 100644 sound/soc/codecs/nau8360.h + create mode 100644 sound/soc/codecs/sn624x-sdca-sdw.c + create mode 100644 sound/soc/codecs/sn624x-sdca.h + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_amp.c + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_dmic.c + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_sdca.c + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_sdca_jack_common.c +Merging modules/modules-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'modules-next' of https://git.kernel.org/pub/scm/linux/kernel/git/modules/linux.git modules/modules-next +Already up to date. +Merging input/next (fcc1d6eab4ce4 Input: st-keyscan - improve probe error handling) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git input/next +Merge made by the 'ort' strategy. + drivers/input/keyboard/snvs_pwrkey.c | 9 ++++----- + drivers/input/keyboard/st-keyscan.c | 13 +++++-------- + 2 files changed, 9 insertions(+), 13 deletions(-) +Merging block/for-next (c300d74d44c27 Merge branch 'block-7.3' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux.git block/for-next +Merge made by the 'ort' strategy. + block/bio.c | 9 +++++++++ + block/genhd.c | 7 +++++++ + drivers/block/loop.c | 8 +++++--- + drivers/block/ublk_drv.c | 6 ++++++ + drivers/block/zloop.c | 8 +++++--- + 5 files changed, 32 insertions(+), 6 deletions(-) +$ git am -3 ../patches/0001-Revert-block-remove-bio_last_bvec_all.patch +Applying: Revert "block: remove bio_last_bvec_all" +$ git reset HEAD^ +Unstaged changes after reset: +M Documentation/block/biovecs.rst +M include/linux/bio.h +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master b5938f45abc15] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux.git + Date: Thu Sep 3 16:48:20 2026 +0100 +Merging device-mapper/for-next (39c5aa3bd8ec3 dm-era: fix shadowed superblock leak on take-snap failure) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/device-mapper/linux-dm.git device-mapper/for-next +Already up to date. +Merging libata/for-next (bd46a0b22933f ata: pata_parport: Fix use-after-free in new_device_store) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/libata/linux libata/for-next +Merge made by the 'ort' strategy. + drivers/ata/Kconfig | 10 ++ + drivers/ata/Makefile | 1 + + drivers/ata/ahci.c | 1 + + drivers/ata/ahci_brcm.c | 1 + + drivers/ata/ahci_ceva.c | 1 + + drivers/ata/ahci_qoriq.c | 1 + + drivers/ata/ata_generic.c | 16 +++ + drivers/ata/libahci.c | 1 + + drivers/ata/pata_cswarp.c | 183 ++++++++++++++++++++++++++++++++ + drivers/ata/pata_parport/pata_parport.c | 7 +- + 10 files changed, 219 insertions(+), 3 deletions(-) + create mode 100644 drivers/ata/pata_cswarp.c +Merging pcmcia/pcmcia-next (b3c26ea81ccc5 pcmcia: remove obsolete host controller drivers) +$ git merge -m Merge branch 'pcmcia-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brodo/linux.git pcmcia/pcmcia-next +Already up to date. +Merging mmc/next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git mmc/next +Already up to date. +Merging mfd/for-mfd-next (9d0e4b1ae5b04 mfd: cs42l43: Fix regmap defaults ordering) +$ git merge -m Merge branch 'for-mfd-next' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git mfd/for-mfd-next +Already up to date. +Merging backlight/for-backlight-next (cf1a12e080451 backlight: Use sysfs_emit() instead of sprintf()) +$ git merge -m Merge branch 'for-backlight-next' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git backlight/for-backlight-next +Already up to date. +Merging battery/for-next (2da28b059e0dd power: supply: bq27xxx: bq27z561: fix invalid AverageEnergy address) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git battery/for-next +Already up to date. +Merging regulator/for-next (f656e44fc0c84 Merge remote-tracking branch 'regulator/for-7.4' into regulator-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git regulator/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/regulator/ti,tps65023.yaml | 90 ++++++++++++++++++++++ + .../devicetree/bindings/regulator/tps65023.txt | 60 --------------- + drivers/regulator/fixed.c | 4 + + drivers/regulator/pca9450-regulator.c | 81 ++++++++++++++++--- + include/linux/regulator/pca9450.h | 4 +- + 5 files changed, 167 insertions(+), 72 deletions(-) + create mode 100644 Documentation/devicetree/bindings/regulator/ti,tps65023.yaml + delete mode 100644 Documentation/devicetree/bindings/regulator/tps65023.txt +Merging security/next (3a84fc1a21577 lsm: don't call security_backing_file_free() multiple times) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git security/next +Auto-merging include/linux/ns/ns_common_types.h +CONFLICT (content): Merge conflict in include/linux/ns/ns_common_types.h +Auto-merging include/linux/sched.h +Resolved 'include/linux/ns/ns_common_types.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 97ffa32a43dfb] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git +$ git diff -M --stat --summary HEAD^.. + fs/namespace.c | 3 +- + include/linux/cred.h | 17 ++++++--- + include/linux/lsm_audit.h | 5 +++ + include/linux/lsm_hook_defs.h | 3 ++ + include/linux/lsm_hooks.h | 1 + + include/linux/ns/ns_common_types.h | 3 ++ + include/linux/sched.h | 8 +++- + include/linux/security.h | 20 ++++++++++ + include/uapi/linux/nsfs.h | 1 + + init/init_task.c | 2 +- + kernel/auditsc.c | 5 ++- + kernel/cred.c | 2 +- + kernel/nscommon.c | 17 ++++++++- + kernel/nsproxy.c | 6 +++ + security/lsm_audit.c | 6 ++- + security/lsm_init.c | 6 +-- + security/security.c | 75 +++++++++++++++++++++++++++++++++++++- + 17 files changed, 159 insertions(+), 21 deletions(-) +Merging apparmor/apparmor-next (3daad923a8685 apparmor: policy_int make sure list heads are initialized before fail path) +$ git merge -m Merge branch 'apparmor-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jj/linux-apparmor apparmor/apparmor-next +Already up to date. +Merging integrity/next-integrity (8861f6d5c0678 ima: Check for ERR_PTR from dentry_path() in validate_hash_algo()) +$ git merge -m Merge branch 'next-integrity' of https://git.kernel.org/pub/scm/linux/kernel/git/zohar/linux-integrity integrity/next-integrity +Merge made by the 'ort' strategy. + Documentation/ABI/testing/ima_policy | 3 +++ + fs/configfs/mount.c | 4 +--- + include/uapi/linux/magic.h | 1 + + security/integrity/ima/ima_appraise.c | 2 ++ + security/integrity/ima/ima_policy.c | 7 ++++++- + 5 files changed, 13 insertions(+), 4 deletions(-) +Merging selinux/next (23be82ec7c9fb Automated merge of 'dev' into 'next') +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/selinux.git selinux/next +Merge made by the 'ort' strategy. + security/selinux/hooks.c | 36 ++++++++++-------------------------- + security/selinux/selinuxfs.c | 34 ++++++++++++++++------------------ + security/selinux/ss/mls.c | 38 -------------------------------------- + security/selinux/ss/mls.h | 3 --- + security/selinux/ss/mls_types.h | 3 --- + security/selinux/ss/policydb.c | 11 ++++------- + 6 files changed, 30 insertions(+), 95 deletions(-) +Merging smack/next (fedc88e38ce97 smack: fix cred UAF in smack_file_send_sigiotask()) +$ git merge -m Merge branch 'next' of https://github.com/cschaufler/smack-next smack/next +Already up to date. +Merging tomoyo/master (2d2338c93da79 Merge tag 'i2c-fixes-7.2-rc6' of git://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux) +$ git merge -m Merge branch 'master' of git://git.code.sf.net/p/tomoyo/tomoyo.git tomoyo/master +Already up to date. +Merging tpmdd-tpm/for-next-tpm (22a50c3745c5c tpm: Call cmd_ready/go_idle for each command transmission) +$ git merge -m Merge branch 'for-next-tpm' of https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git tpmdd-tpm/for-next-tpm +Merge made by the 'ort' strategy. + drivers/char/tpm/tpm-chip.c | 24 ------------------------ + drivers/char/tpm/tpm-interface.c | 23 ++++++++++++++++++++++- + include/keys/request_key_auth-type.h | 2 +- + security/keys/gc.c | 4 +--- + security/keys/request_key_auth.c | 12 +++++++++--- + 5 files changed, 33 insertions(+), 32 deletions(-) +Merging tpmdd-keys/for-next-keys (bf0d7882cd43d keys: translate request_key_auth pid for the reading procfs instance) +$ git merge -m Merge branch 'for-next-keys' of https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git tpmdd-keys/for-next-keys +Already up to date. +Merging watchdog/watchdog-next (b85ed7f7259b4 watchdog: Differentiate scenarios when watchdog is closed) +$ git merge -m Merge branch 'watchdog-next' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git watchdog/watchdog-next +Merge made by the 'ort' strategy. + .../bindings/watchdog/renesas,r9a09g057-wdt.yaml | 29 +++++++++- + .../devicetree/bindings/watchdog/samsung-wdt.yaml | 23 +++++++- + Documentation/watchdog/watchdog-parameters.rst | 2 + + drivers/watchdog/mtk_wdt.c | 62 +++++++++++++++++++--- + drivers/watchdog/s3c2410_wdt.c | 16 ++++++ + drivers/watchdog/sbsa_gwdt.c | 19 ++++++- + drivers/watchdog/sp5100_tco.c | 61 +++++++++++++++++++-- + drivers/watchdog/watchdog_dev.c | 4 ++ + 8 files changed, 200 insertions(+), 16 deletions(-) +Merging iommu/next (3c6a2bac15247 Merge branches 'arm/smmu/updates', 'arm/smmu/bindings', 'mediatek', 'qualcomm/msm', 'rockchip', 'ti/omap', 'riscv', 'intel/vt-d', 'amd/amd-vi', 'core' and 'typos' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/iommu/linux.git iommu/next +Already up to date. +Merging audit/next (ca12826b4aac8 audit: Fix filter rule accounting after automatic removal) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/audit.git audit/next +Auto-merging kernel/auditsc.c +Merge made by the 'ort' strategy. + include/linux/audit.h | 6 ++-- + kernel/audit.c | 2 +- + kernel/audit.h | 5 +++ + kernel/audit_tree.c | 1 + + kernel/audit_watch.c | 2 ++ + kernel/auditfilter.c | 86 +++++++++++++++++++++++++-------------------------- + kernel/auditsc.c | 2 ++ + 7 files changed, 56 insertions(+), 48 deletions(-) +Merging devicetree/for-next (1e151de2d11ec dt-bindings: arm: ti,omap-mpu: Convert to DT schema) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git devicetree/for-next +Auto-merging Documentation/devicetree/bindings/i2c/xlnx,xps-iic-2.00.a.yaml +Auto-merging Documentation/devicetree/bindings/trivial-devices.yaml +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/arm,coresight-cti.yaml | 2 +- + Documentation/devicetree/bindings/arm/omap/mpu.txt | 54 -- + .../devicetree/bindings/arm/ti/ti,omap-mpu.yaml | 58 ++ + .../devicetree/bindings/bus/omap-ocp2scp.txt | 29 - + .../devicetree/bindings/bus/ti,omap-ocp2scp.yaml | 74 ++ + .../bindings/display/bridge/sil,sii9022.yaml | 2 +- + .../devicetree/bindings/gpio/delta,tn48m-gpio.yaml | 2 +- + .../devicetree/bindings/hwmon/adi,ltc2991.yaml | 2 +- + .../bindings/i2c/xlnx,xps-iic-2.00.a.yaml | 4 +- + .../bindings/iio/light/upisemi,us5182.yaml | 2 +- + .../devicetree/bindings/input/matrix-keymap.yaml | 3 +- + .../devicetree/bindings/input/ti,tca8418.yaml | 21 +- + .../devicetree/bindings/media/nxp,imx8-jpeg.yaml | 4 +- + .../devicetree/bindings/mfd/ti,tps65910.yaml | 2 +- + .../bindings/nvmem/zii,rave-sp-eeprom.yaml | 2 +- + .../bindings/pci/hisilicon,kirin-pcie.yaml | 4 +- + .../bindings/pinctrl/sunplus,sp7021-pinctrl.yaml | 2 +- + .../bindings/spi/aspeed,ast2600-fmc.yaml | 6 +- + .../devicetree/bindings/thermal/thermal-idle.yaml | 2 +- + .../devicetree/bindings/trivial-devices.yaml | 4 +- + MAINTAINERS | 2 + + scripts/dtc/dt-check-style | 854 ++++++++++++--------- + .../bad/dts-child-name-order.dtso | 33 + + .../dtc/dt-style-selftest/bad/dts-cont-align.dts | 26 + + .../dt-style-selftest/bad/dts-digit-node-order.dts | 40 + + .../bad/dts-digit-node-order.dtso | 41 + + .../bad/dts-extend-node-child-name-order.dtso | 26 + + .../bad/dts-extend-node-digit-node-order.dtso | 34 + + .../dtc/dt-style-selftest/bad/dts-line-length.dts | 21 + + .../dtc/dt-style-selftest/bad/dts-node-name.dts | 60 ++ + .../dt-style-selftest/bad/dts-property-name.dts | 28 + + .../dt-style-selftest/bad/dts-property-order.dts | 15 + + .../dt-style-selftest/bad/dts-property-order.dtso | 59 ++ + .../bad/dts-redundant-ws-strict.dts | 27 + + .../dtc/dt-style-selftest/bad/dts-redundant-ws.dts | 28 + + .../dt-style-selftest/bad/dts-redundant-ws.dtso | 9 + + .../dtc/dt-style-selftest/bad/dts-trailing-ws.dts | 8 + + .../dtc/dt-style-selftest/bad/dts-unused-label.dts | 21 + + .../bad/yaml-child-addr-order.yaml | 2 +- + .../bad/yaml-child-name-order.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-cont-align.yaml | 8 +- + .../bad/yaml-digit-node-order.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-hex-case.yaml | 2 +- + .../dt-style-selftest/bad/yaml-indent-strict.yaml | 2 +- + .../bad/yaml-label-in-string.yaml | 2 +- + .../dt-style-selftest/bad/yaml-line-length.yaml | 2 +- + .../dt-style-selftest/bad/yaml-mixed-indent.yaml | 4 +- + .../dt-style-selftest/bad/yaml-multi-close.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-node-close.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-node-name.yaml | 54 ++ + .../bad/yaml-prop-order-device-type.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-prop-order.yaml | 2 +- + .../dt-style-selftest/bad/yaml-prop-pairing.yaml | 2 +- + .../dt-style-selftest/bad/yaml-property-name.yaml | 46 ++ + .../bad/yaml-redundant-ws-strict.yaml | 31 + + .../dt-style-selftest/bad/yaml-redundant-ws.yaml | 35 + + .../dt-style-selftest/bad/yaml-required-blank.yaml | 2 +- + scripts/dtc/dt-style-selftest/bad/yaml-tab.yaml | 2 +- + .../bad/yaml-trailing-comment.yaml | 2 +- + .../dt-style-selftest/bad/yaml-trailing-ws.yaml | 2 +- + .../bad/yaml-unclosed-comment.yaml | 2 +- + .../bad/yaml-unit-addr-prefix.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-unit-addr.yaml | 2 +- + .../dt-style-selftest/bad/yaml-unused-label.yaml | 2 +- + .../bad/yaml-value-ws-multiline.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-value-ws.yaml | 2 +- + .../expected/dts-child-name-order.dts.txt | 1 + + .../expected/dts-child-name-order.dtso.txt | 3 + + .../expected/dts-cont-align.dts.txt | 10 + + .../expected/dts-digit-node-order.dts.txt | 2 + + .../expected/dts-digit-node-order.dtso.txt | 2 + + .../dts-extend-node-child-name-order.dtso.txt | 2 + + .../dts-extend-node-digit-node-order.dtso.txt | 2 + + .../expected/dts-line-length.dts.txt | 2 + + .../expected/dts-mixed-indent.dts.txt | 1 + + .../expected/dts-node-name.dts.txt | 13 + + .../expected/dts-property-name.dts.txt | 14 + + .../expected/dts-property-order.dts.txt | 15 +- + .../expected/dts-property-order.dtso.txt | 11 + + .../expected/dts-redundant-ws-strict.dts.txt | 13 + + .../expected/dts-redundant-ws.dts.txt | 10 + + .../expected/dts-redundant-ws.dtso.txt | 2 + + .../expected/dts-trailing-ws.dts.txt | 2 + + .../expected/dts-unused-label.dts.txt | 2 + + .../expected/yaml-cont-align.yaml.txt | 3 +- + .../expected/yaml-mixed-indent.yaml.txt | 1 + + .../expected/yaml-node-name.yaml.txt | 7 + + .../expected/yaml-property-name.yaml.txt | 14 + + .../expected/yaml-redundant-ws-strict.yaml.txt | 5 + + .../expected/yaml-redundant-ws.yaml.txt | 4 + + .../expected/yaml-value-ws-multiline.yaml.txt | 1 + + .../good/dts-child-name-order.dtso | 44 ++ + .../dtc/dt-style-selftest/good/dts-cont-align.dts | 13 +- + .../good/dts-digit-node-order.dts | 3 - + .../good/dts-digit-node-order.dtso | 59 ++ + .../good/dts-extend-node-child-name-order.dtso | 26 + + .../good/dts-extend-node-digit-node-order.dtso | 34 + + .../dt-style-selftest/good/dts-property-order.dts | 5 + + .../dt-style-selftest/good/dts-property-order.dtso | 47 ++ + .../dtc/dt-style-selftest/good/yaml-4space.yaml | 2 +- + .../dt-style-selftest/good/yaml-cont-align.yaml | 32 + + .../good/yaml-tricky-parsing.yaml | 2 +- + scripts/dtc/dt-style-selftest/run.sh | 2 +- + 103 files changed, 1734 insertions(+), 510 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/arm/omap/mpu.txt + create mode 100644 Documentation/devicetree/bindings/arm/ti/ti,omap-mpu.yaml + delete mode 100644 Documentation/devicetree/bindings/bus/omap-ocp2scp.txt + create mode 100644 Documentation/devicetree/bindings/bus/ti,omap-ocp2scp.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-cont-align.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-digit-node-order.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-extend-node-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-extend-node-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-line-length.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-node-name.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-property-name.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-property-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-redundant-ws-strict.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-redundant-ws.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-redundant-ws.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-trailing-ws.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-unused-label.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-node-name.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-property-name.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-redundant-ws-strict.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-redundant-ws.yaml + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-child-name-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-cont-align.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-digit-node-order.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-digit-node-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-extend-node-child-name-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-extend-node-digit-node-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-line-length.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-node-name.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-property-name.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-property-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-redundant-ws-strict.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-redundant-ws.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-redundant-ws.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-trailing-ws.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-unused-label.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-node-name.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-property-name.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-redundant-ws-strict.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-redundant-ws.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-extend-node-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-extend-node-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-property-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/yaml-cont-align.yaml +Merging dt-krzk/for-next (6bd34d56818cc Merge branches 'next/dt' and 'next/dt64' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git dt-krzk/for-next +Merge made by the 'ort' strategy. +Merging mailbox/for-next (14af7a96afa39 mailbox: add Axiado AX3005 mailbox driver) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jassibrar/mailbox.git mailbox/for-next +Already up to date. +Merging spi/for-next (5f01cb141169f spi: omap2-mcspi: Remove unbalanced pm_runtime_put_sync() calls) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git spi/for-next +Merge made by the 'ort' strategy. + drivers/spi/Kconfig | 364 +++++++++++++++++++-------------------- + drivers/spi/spi-geni-qcom.c | 15 +- + drivers/spi/spi-omap2-mcspi.c | 2 - + drivers/spi/spi-sh-msiof.c | 5 +- + drivers/spi/spi-sunplus-sp7021.c | 13 +- + drivers/spi/spi-virtio.c | 2 - + 6 files changed, 197 insertions(+), 204 deletions(-) +Merging tip/master (461735aa6e8e3 Merge branch into tip/master: 'x86/bugs') +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git tip/master +Auto-merging arch/Kconfig +Auto-merging arch/arm64/Kconfig +Auto-merging arch/loongarch/Kconfig +Auto-merging arch/powerpc/Kconfig +Auto-merging arch/riscv/Kconfig +Auto-merging arch/s390/Kconfig +Auto-merging arch/x86/Kconfig +Auto-merging include/linux/sched.h +Merge made by the 'ort' strategy. + arch/Kconfig | 38 ------ + arch/arm64/Kconfig | 1 - + arch/arm64/include/asm/preempt.h | 10 -- + arch/arm64/kernel/paravirt.c | 4 +- + arch/loongarch/Kconfig | 1 - + arch/loongarch/kernel/paravirt.c | 4 +- + arch/powerpc/Kconfig | 1 - + arch/powerpc/platforms/pseries/setup.c | 4 +- + arch/riscv/Kconfig | 1 - + arch/riscv/kernel/paravirt.c | 4 +- + arch/s390/Kconfig | 1 - + arch/s390/include/asm/preempt.h | 11 -- + arch/x86/Kconfig | 1 - + arch/x86/include/asm/cpufeatures.h | 1 + + arch/x86/include/asm/preempt.h | 28 ----- + arch/x86/kernel/cpu/bugs.c | 18 ++- + arch/x86/kernel/cpu/scattered.c | 1 + + arch/x86/kernel/cpu/vmware.c | 4 +- + arch/x86/kernel/kvm.c | 4 +- + drivers/xen/time.c | 4 +- + include/asm-generic/preempt.h | 10 -- + include/linux/irq-entry-common.h | 17 +-- + include/linux/kernel.h | 20 --- + include/linux/preempt.h | 20 +-- + include/linux/sched.h | 31 +---- + include/linux/sched/cputime.h | 6 +- + kernel/Kconfig.preempt | 9 +- + kernel/entry/common.c | 17 +-- + kernel/sched/core.c | 221 +++++---------------------------- + kernel/sched/cputime.c | 4 +- + kernel/sched/deadline.c | 2 +- + kernel/sched/debug.c | 20 ++- + kernel/sched/fair.c | 16 +-- + kernel/sched/sched.h | 38 +++++- + 34 files changed, 132 insertions(+), 440 deletions(-) +Merging kexec/kexec-next (c6ed7331aad89 Merge branch 'kexec-7.4.hwpoison' into kexec-next) +$ git merge -m Merge branch 'kexec-next' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git kexec/kexec-next +Auto-merging include/linux/mm.h +Merge made by the 'ort' strategy. + include/linux/mm.h | 14 ++++++++++++++ + kernel/kexec_core.c | 10 ++++++++++ + kernel/kexec_file.c | 20 +++++++++++++++++++- + mm/memory-failure.c | 40 ++++++++++++++++++++++++++++++++++++++++ + 4 files changed, 83 insertions(+), 1 deletion(-) +Merging liveupdate/next (c65687600ed07 Merge branch 'luo-internal-api' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git liveupdate/next +Merge made by the 'ort' strategy. + include/linux/liveupdate.h | 22 ++++++++++++ + kernel/liveupdate/kexec_handover.c | 7 ++++ + kernel/liveupdate/luo_file.c | 69 ++++++++++++++++++++++++++++++++++++++ + kernel/liveupdate/luo_internal.h | 17 ++++++++++ + 4 files changed, 115 insertions(+) +Merging clockevents/timers/drivers/next (1b8b356b4b06e clocksource/drivers/armada: Unwind timer clock on init failure) +$ git merge -m Merge branch 'timers/drivers/next' of https://git.kernel.org/pub/scm/linux/kernel/git/daniel.lezcano/linux.git clockevents/timers/drivers/next +Auto-merging drivers/pwm/pwm-samsung.c +Merge made by the 'ort' strategy. +Merging edac/edac-for-next (9fa765b3ed33d Merge ras/edac-misc into for-next) +$ git merge -m Merge branch 'edac-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ras/ras.git edac/edac-for-next +Merge made by the 'ort' strategy. + drivers/edac/altera_edac.c | 10 +++++----- + 1 file changed, 5 insertions(+), 5 deletions(-) +Merging ftrace/for-next (3df3ae98c30c2 Merge tracefs/for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git ftrace/for-next +Merge made by the 'ort' strategy. +$ git am -3 ../patches/0001-ftrace-Fix-semantic-conflict-with-mm-tree.patch +Applying: ftrace: Fix semantic conflict with mm tree +Using index info to reconstruct a base tree... +M kernel/trace/trace_printk.c +Falling back to patching base and 3-way merge... +Auto-merging kernel/trace/trace_printk.c +No changes -- Patch already applied. +Merging rcu/next (9cc63f8bcd560 Merge branches 'expcb.2026.07.24a', 'misc.2026.07.30a', 'rcu-tasks.2026.07.30a', 'srcu.2026.08.11a' and 'torture.2026.08.14a' into HEAD) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/rcu/linux rcu/next +Already up to date. +Merging paulmck/non-rcu/next (85f4ab0da7887 Merge branch 'csd-lock.2026.08.23a', tag 'scftorture.2026.08.18a' into HEAD) +$ git merge -m Merge branch 'non-rcu/next' of https://git.kernel.org/pub/scm/linux/kernel/git/paulmck/linux-rcu.git paulmck/non-rcu/next +Auto-merging kernel/smp.c +CONFLICT (content): Merge conflict in kernel/smp.c +Auto-merging lib/Kconfig.debug +Auto-merging lib/Makefile +Resolved 'kernel/smp.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 46f9db5ce1a59] Merge branch 'non-rcu/next' of https://git.kernel.org/pub/scm/linux/kernel/git/paulmck/linux-rcu.git +$ git diff -M --stat --summary HEAD^.. + kernel/smp.c | 77 +++++++++++++++-------- + lib/Kconfig.debug | 12 ++++ + lib/Makefile | 1 + + lib/test_csd_lock.c | 173 ++++++++++++++++++++++++++++++++++++++++++++++++++++ + 4 files changed, 237 insertions(+), 26 deletions(-) + create mode 100644 lib/test_csd_lock.c +Merging kvm/next (76671054f9a1f Merge tag 'kvmarm-7.3' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm into HEAD) +$ git merge -m Merge branch 'next' of git://git.kernel.org/pub/scm/virt/kvm/kvm.git kvm/next +Already up to date. +$ git am -3 ../patches/kvm-x86-static-cpu-has +Applying: Signed-off-by: Mark Brown +Using index info to reconstruct a base tree... +M arch/x86/kvm/msrs.c +Falling back to patching base and 3-way merge... +Auto-merging arch/x86/kvm/msrs.c +No changes -- Patch already applied. +Merging kvm-arm/next (aa8e5dc6a7a2a Merge branch 'kvm-arm64/misc-7.3' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git kvm-arm/next +Already up to date. +Merging kvms390/next (dd6f4ef6f8a37 s390/vfio-ap: Fix NULL deref in status_show() during queue probe) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git kvms390/next +Already up to date. +Merging kvm-ppc/topic/ppc-kvm (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'topic/ppc-kvm' of https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git kvm-ppc/topic/ppc-kvm +Already up to date. +Merging kvm-riscv/riscv_kvm_next (b5060a4aa33d7 RISC-V: KVM: fix vcpu vector context handling for kernel-mode vector) +$ git merge -m Merge branch 'riscv_kvm_next' of https://github.com/kvm-riscv/linux.git kvm-riscv/riscv_kvm_next +Already up to date. +Merging kvm-x86/next (76671054f9a1f Merge tag 'kvmarm-7.3' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm into HEAD) +$ git merge -m Merge branch 'next' of https://github.com/kvm-x86/linux.git kvm-x86/next +Already up to date. +$ git am -3 ../patches/0001-KVM-selftests-Fix-up-semantic-changes.patch +Applying: KVM: selftests: Fix up semantic changes +Using index info to reconstruct a base tree... +M tools/testing/selftests/kvm/lib/kvm_util.c +Falling back to patching base and 3-way merge... +Auto-merging tools/testing/selftests/kvm/lib/kvm_util.c +No changes -- Patch already applied. +Merging xen-tip/linux-next (d330fb86a7170 xenbus: Unregister reboot notifier on init failure) +$ git merge -m Merge branch 'linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/xen/tip.git xen-tip/linux-next +Already up to date. +Merging percpu/for-next (8f0b4cce4481f Linux 6.19-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dennis/percpu.git percpu/for-next +Already up to date. +Merging workqueues/for-next (ab85b68e630e8 workqueue: rename the pwq slot helpers shared with percpu) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/wq.git workqueues/for-next +Auto-merging kernel/workqueue.c +Merge made by the 'ort' strategy. + kernel/workqueue.c | 157 ++++++++++++++++++++++++----------------------------- + 1 file changed, 72 insertions(+), 85 deletions(-) +Merging sched-ext/for-next (c43ea8a85f652 Merge branch 'for-7.4' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext.git sched-ext/for-next +Merge made by the 'ort' strategy. + include/linux/sched/ext.h | 4 +-- + kernel/sched/ext/ext.c | 64 +++++++++++++++++++++++++++------------- + kernel/sched/ext/internal.h | 19 ++++++++++++ + kernel/sched/ext/sub.c | 5 ++++ + tools/sched_ext/scx_flatcg.bpf.c | 2 +- + 5 files changed, 71 insertions(+), 23 deletions(-) +Merging drivers-x86/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git drivers-x86/for-next +Already up to date. +Merging chrome-platform/for-next (c55749415623a platform/chrome: cros_ec_proto: Fix deferred response) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git chrome-platform/for-next +Merge made by the 'ort' strategy. + drivers/platform/chrome/cros_ec_proto.c | 32 ++++- + drivers/platform/chrome/cros_ec_proto_test.c | 158 ++++++++++++++++++++++++- + include/linux/platform_data/cros_ec_commands.h | 72 +++++++++++ + 3 files changed, 255 insertions(+), 7 deletions(-) +Merging chrome-platform-firmware/for-firmware-next (44e33a5aaa2de firmware: Rename google firmware directory to coreboot) +$ git merge -m Merge branch 'for-firmware-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git chrome-platform-firmware/for-firmware-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 20 +++--- + drivers/firmware/Kconfig | 2 +- + drivers/firmware/Makefile | 2 +- + drivers/firmware/{google => coreboot}/Kconfig | 74 ++++++++++++++++------ + drivers/firmware/{google => coreboot}/Makefile | 10 +-- + drivers/firmware/{google => coreboot}/cbmem.c | 0 + .../firmware/{google => coreboot}/coreboot_table.c | 0 + .../firmware/{google => coreboot}/coreboot_table.h | 0 + .../{google => coreboot}/framebuffer-coreboot.c | 0 + drivers/firmware/{google => coreboot}/gsmi.c | 0 + .../{google => coreboot}/memconsole-coreboot.c | 0 + .../{google => coreboot}/memconsole-x86-legacy.c | 0 + drivers/firmware/{google => coreboot}/memconsole.c | 0 + drivers/firmware/{google => coreboot}/memconsole.h | 6 +- + drivers/firmware/{google => coreboot}/vpd.c | 0 + drivers/firmware/{google => coreboot}/vpd_decode.c | 0 + drivers/firmware/{google => coreboot}/vpd_decode.h | 0 + 17 files changed, 76 insertions(+), 38 deletions(-) + rename drivers/firmware/{google => coreboot}/Kconfig (64%) + rename drivers/firmware/{google => coreboot}/Makefile (51%) + rename drivers/firmware/{google => coreboot}/cbmem.c (100%) + rename drivers/firmware/{google => coreboot}/coreboot_table.c (100%) + rename drivers/firmware/{google => coreboot}/coreboot_table.h (100%) + rename drivers/firmware/{google => coreboot}/framebuffer-coreboot.c (100%) + rename drivers/firmware/{google => coreboot}/gsmi.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole-coreboot.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole-x86-legacy.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole.h (82%) + rename drivers/firmware/{google => coreboot}/vpd.c (100%) + rename drivers/firmware/{google => coreboot}/vpd_decode.c (100%) + rename drivers/firmware/{google => coreboot}/vpd_decode.h (100%) +Merging hsi/for-next (e81250ec6b692 hsi: omap_ssi_core: fix missing DMA mask setup for SSI controller device) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-hsi.git hsi/for-next +Already up to date. +Merging leds-lj/for-leds-next (f8cca63a0a4f3 leds: is31fl319x: Modernize registration) +$ git merge -m Merge branch 'for-leds-next' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/leds.git leds-lj/for-leds-next +Already up to date. +Merging ipmi/for-next (89a312991dc6e Merge tag 'cifs-fixes-7.3-rc2' of https://git.manguebit.org/linux) +$ git merge -m Merge branch 'for-next' of https://github.com/cminyard/linux-ipmi.git ipmi/for-next +Already up to date. +Merging driver-core/driver-core-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'driver-core-next' of https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git driver-core/driver-core-next +Already up to date. +Merging usb/usb-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'usb-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git usb/usb-next +Already up to date. +Merging thunderbolt/next (48e989e33b715 thunderbolt: dma_test: Tear down DMA paths before stopping the rings) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git thunderbolt/next +Auto-merging drivers/thunderbolt/switch.c +Merge made by the 'ort' strategy. + drivers/thunderbolt/dma_test.c | 10 +++++----- + drivers/thunderbolt/eeprom.c | 23 ++++++++++++++++------- + drivers/thunderbolt/path.c | 2 +- + drivers/thunderbolt/quirks.c | 18 ++++++++++++++++++ + drivers/thunderbolt/switch.c | 4 ++++ + drivers/thunderbolt/tb.h | 3 +++ + 6 files changed, 47 insertions(+), 13 deletions(-) +Merging usb-serial/usb-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'usb-next' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git usb-serial/usb-next +Already up to date. +Merging tty/tty-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'tty-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git tty/tty-next +Already up to date. +Merging char-misc/char-misc-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'char-misc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git char-misc/char-misc-next +Already up to date. +$ git am -3 ../patches/rust-vs-char-misc-fixup +Applying: pin-init changes for v7.3-rc1 +Using index info to reconstruct a base tree... +M rust/kernel/serdev.rs +Falling back to patching base and 3-way merge... +Auto-merging rust/kernel/serdev.rs +No changes -- Patch already applied. +$ git am -3 ../merge-fixes/rust_binder-use-LocalModule-for-THIS_MODULE +Applying: rust_binder: use `LocalModule` for `THIS_MODULE` +Using index info to reconstruct a base tree... +M drivers/android/binder/netlink.rs +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +Merging coresight/next (9e3604d7369cf coresight: etm4x: remove redundant fields in etmv4_save_state) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/coresight/linux.git coresight/next +Already up to date. +Merging fastrpc/for-next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/srini/fastrpc.git fastrpc/for-next +Already up to date. +Merging fpga/for-next (4216e549a2657 fpga: dfl: fix spelling in sysfs-platform-dfl-port ABI documentation) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git fpga/for-next +Already up to date. +Merging icc/icc-next (9621c81fd2eb7 Merge branch 'icc-misc' into icc-next) +$ git merge -m Merge branch 'icc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/djakov/icc.git icc/icc-next +Already up to date. +Merging iio/togreg (183f05a300eab iio: light: vcnl4000: use sysfs_emit() for near_level) +$ git merge -m Merge branch 'togreg' of https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git iio/togreg +Auto-merging Documentation/devicetree/bindings/trivial-devices.yaml +Auto-merging MAINTAINERS +Auto-merging drivers/iio/accel/kionix-kx022a.c +Auto-merging drivers/iio/adc/ade9000.c +CONFLICT (content): Merge conflict in drivers/iio/adc/ade9000.c +Auto-merging drivers/iio/buffer/industrialio-buffer-dmaengine.c +Auto-merging drivers/iio/pressure/rohm-bm1390.c +Auto-merging drivers/iio/proximity/aw96103.c +Resolved 'drivers/iio/adc/ade9000.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 0ce15836e3afa] Merge branch 'togreg' of https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git +$ git diff -M --stat --summary HEAD^.. + .../devicetree/bindings/iio/adc/adi,ade9000.yaml | 30 +- + .../bindings/iio/adc/axiado,ax3000-saradc.yaml | 63 +++ + .../devicetree/bindings/iio/adc/ti,ads1100.yaml | 10 +- + .../devicetree/bindings/iio/adc/ti,ads112c04.yaml | 149 ++++++ + .../bindings/iio/imu/invensense,icm42600.yaml | 24 +- + .../bindings/iio/light/liteon,ltr501.yaml | 32 +- + .../iio/proximity/pulsedlight,lidar-lite-v2.yaml | 71 +++ + .../devicetree/bindings/trivial-devices.yaml | 4 +- + Documentation/iio/ade9000.rst | 26 +- + Documentation/iio/adxl380.rst | 8 +- + MAINTAINERS | 17 + + drivers/iio/accel/adxl367.c | 7 + + drivers/iio/accel/adxl380.c | 7 + + drivers/iio/accel/bma400_core.c | 2 - + drivers/iio/accel/kionix-kx022a.c | 20 +- + drivers/iio/accel/mma8452.c | 145 +++--- + drivers/iio/adc/Kconfig | 46 +- + drivers/iio/adc/Makefile | 2 + + drivers/iio/adc/ad4080.c | 13 +- + drivers/iio/adc/ade9000.c | 180 ++++--- + drivers/iio/adc/axiado_saradc.c | 277 +++++++++++ + drivers/iio/adc/bcm_iproc_adc.c | 46 +- + drivers/iio/adc/meson_saradc.c | 2 +- + drivers/iio/adc/pac1934.c | 9 +- + drivers/iio/adc/rockchip_saradc.c | 16 + + drivers/iio/adc/sophgo-cv1800b-adc.c | 2 + + drivers/iio/adc/ti-ads1100.c | 188 +++++++- + drivers/iio/adc/ti-ads112c04.c | 524 +++++++++++++++++++++ + drivers/iio/buffer/industrialio-buffer-dmaengine.c | 3 - + drivers/iio/chemical/sps30.c | 11 +- + .../iio/common/cros_ec_sensors/cros_ec_sensors.c | 2 +- + .../common/cros_ec_sensors/cros_ec_sensors_core.c | 5 +- + drivers/iio/frequency/adf4350.c | 2 +- + drivers/iio/gyro/adxrs290.c | 4 +- + drivers/iio/gyro/bmg160_core.c | 4 +- + drivers/iio/gyro/itg3200_buffer.c | 3 +- + drivers/iio/humidity/Kconfig | 8 +- + drivers/iio/humidity/am2315.c | 34 +- + drivers/iio/humidity/ens210.c | 4 +- + drivers/iio/humidity/hts221_core.c | 168 +++---- + drivers/iio/humidity/hts221_i2c.c | 11 +- + drivers/iio/humidity/hts221_spi.c | 11 +- + drivers/iio/imu/inv_icm42600/inv_icm42600.h | 4 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_accel.c | 16 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_buffer.c | 112 ++--- + drivers/iio/imu/inv_icm42600/inv_icm42600_core.c | 17 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_gyro.c | 16 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_i2c.c | 16 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_spi.c | 16 +- + drivers/iio/imu/inv_mpu6050/inv_mpu_core.c | 6 +- + drivers/iio/industrialio-core.c | 6 +- + drivers/iio/industrialio-gts-helper.c | 2 +- + drivers/iio/light/apds9999.c | 12 +- + drivers/iio/light/cros_ec_light_prox.c | 2 +- + drivers/iio/light/isl29018.c | 7 +- + drivers/iio/light/lm3533-als.c | 11 +- + drivers/iio/light/ltr390.c | 6 - + drivers/iio/light/ltr501.c | 109 +++-- + drivers/iio/light/tsl2583.c | 15 +- + drivers/iio/light/tsl2772.c | 16 +- + drivers/iio/light/vcnl4000.c | 3 +- + drivers/iio/light/veml3328.c | 53 ++- + drivers/iio/pressure/cros_ec_baro.c | 2 +- + drivers/iio/pressure/rohm-bm1390.c | 4 +- + drivers/iio/proximity/aw96103.c | 2 +- + drivers/iio/temperature/tmp117.c | 9 +- + drivers/staging/iio/frequency/ad9832.c | 6 + + drivers/staging/iio/frequency/ad9834.c | 6 + + include/linux/iio/iio-gts-helper.h | 1 + + 69 files changed, 2082 insertions(+), 583 deletions(-) + create mode 100644 Documentation/devicetree/bindings/iio/adc/axiado,ax3000-saradc.yaml + create mode 100644 Documentation/devicetree/bindings/iio/adc/ti,ads112c04.yaml + create mode 100644 Documentation/devicetree/bindings/iio/proximity/pulsedlight,lidar-lite-v2.yaml + create mode 100644 drivers/iio/adc/axiado_saradc.c + create mode 100644 drivers/iio/adc/ti-ads112c04.c +Merging nfc/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://codeberg.org/linux-nfc/linux.git nfc/for-next +Already up to date. +Merging phy-next/next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git phy-next/next +Already up to date. +Merging soundwire/next (1d4d3198ccce1 soundwire: intel_auxdevice: add tac5xx2 family to wake_capable_list) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git soundwire/next +Auto-merging drivers/soundwire/dmi-quirks.c +Merge made by the 'ort' strategy. + drivers/soundwire/bus_type.c | 16 +++++++++++----- + drivers/soundwire/dmi-quirks.c | 7 +++++++ + drivers/soundwire/intel_auxdevice.c | 4 +++- + 3 files changed, 21 insertions(+), 6 deletions(-) +Merging extcon/extcon-next (8d3ae59288f1e Linux 7.2) +$ git merge -m Merge branch 'extcon-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/extcon.git extcon/extcon-next +Already up to date. +Merging gnss/gnss-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'gnss-next' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git gnss/gnss-next +Already up to date. +Merging vfio/next (4e3c1fc8abcb8 vfio/pci: Remove the pcie check for VFIO_PCI_ERR_IRQ_INDEX) +$ git merge -m Merge branch 'next' of https://github.com/awilliam/linux-vfio.git vfio/next +Already up to date. +Merging w1/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git w1/for-next +Already up to date. +Merging spmi/spmi-next (8cdeaa50eae8d Linux 7.2-rc2) +$ git merge -m Merge branch 'spmi-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sboyd/spmi.git spmi/spmi-next +Already up to date. +Merging staging/staging-next (758be99c46265 staging: rtl8723bs: avoid CamelCase in rtw_efuse) +$ git merge -m Merge branch 'staging-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git staging/staging-next +Auto-merging drivers/staging/rtl8723bs/core/rtw_mlme.c +Auto-merging drivers/staging/sm750fb/sm750.c +Auto-merging drivers/staging/sm750fb/sm750_accel.h +Merge made by the 'ort' strategy. + drivers/staging/axis-fifo/axis-fifo.c | 6 +- + drivers/staging/fbtft/fb_bd663474.c | 4 +- + drivers/staging/greybus/loopback.c | 5 +- + drivers/staging/most/video/video.c | 1 - + drivers/staging/octeon/ethernet-tx.c | 2 +- + drivers/staging/rtl8723bs/core/rtw_ap.c | 2 +- + drivers/staging/rtl8723bs/core/rtw_cmd.c | 25 +- + drivers/staging/rtl8723bs/core/rtw_efuse.c | 147 +--- + drivers/staging/rtl8723bs/core/rtw_mlme.c | 966 +++++++++++---------- + drivers/staging/rtl8723bs/core/rtw_mlme_ext.c | 13 +- + drivers/staging/rtl8723bs/core/rtw_recv.c | 49 +- + drivers/staging/rtl8723bs/core/rtw_wlan_util.c | 6 +- + drivers/staging/rtl8723bs/core/rtw_xmit.c | 48 +- + drivers/staging/rtl8723bs/hal/HalPhyRf.c | 3 +- + drivers/staging/rtl8723bs/hal/hal_com.c | 6 +- + drivers/staging/rtl8723bs/hal/hal_intf.c | 9 +- + drivers/staging/rtl8723bs/hal/odm.h | 9 - + drivers/staging/rtl8723bs/hal/odm_DynamicTxPower.h | 1 - + drivers/staging/rtl8723bs/hal/odm_precomp.h | 1 - + drivers/staging/rtl8723bs/hal/rtl8723b_rf6052.c | 2 +- + drivers/staging/rtl8723bs/hal/rtl8723bs_recv.c | 3 +- + drivers/staging/rtl8723bs/hal/sdio_halinit.c | 24 - + drivers/staging/rtl8723bs/include/basic_types.h | 3 - + drivers/staging/rtl8723bs/include/cmd_osdep.h | 6 +- + drivers/staging/rtl8723bs/include/hal_com.h | 2 +- + drivers/staging/rtl8723bs/include/hal_data.h | 3 - + drivers/staging/rtl8723bs/include/hal_intf.h | 2 +- + drivers/staging/rtl8723bs/include/hal_phy.h | 16 - + drivers/staging/rtl8723bs/include/osdep_service.h | 4 +- + .../rtl8723bs/include/osdep_service_linux.h | 2 +- + drivers/staging/rtl8723bs/include/rtl8723b_hal.h | 2 +- + drivers/staging/rtl8723bs/include/rtw_cmd.h | 58 +- + drivers/staging/rtl8723bs/include/rtw_efuse.h | 5 +- + drivers/staging/rtl8723bs/include/rtw_io.h | 14 +- + drivers/staging/rtl8723bs/include/rtw_ioctl_set.h | 2 - + drivers/staging/rtl8723bs/include/rtw_mlme.h | 78 +- + drivers/staging/rtl8723bs/include/rtw_mlme_ext.h | 36 +- + drivers/staging/rtl8723bs/include/rtw_pwrctrl.h | 20 +- + drivers/staging/rtl8723bs/include/rtw_recv.h | 69 +- + drivers/staging/rtl8723bs/include/rtw_xmit.h | 8 +- + drivers/staging/rtl8723bs/include/sdio_ops.h | 21 +- + drivers/staging/rtl8723bs/include/sta_info.h | 84 +- + drivers/staging/rtl8723bs/include/wlan_bssdef.h | 8 +- + drivers/staging/rtl8723bs/include/xmit_osdep.h | 14 +- + drivers/staging/rtl8723bs/os_dep/ioctl_cfg80211.c | 86 +- + drivers/staging/rtl8723bs/os_dep/sdio_intf.c | 4 +- + drivers/staging/sm750fb/sm750.c | 12 +- + drivers/staging/sm750fb/sm750_accel.h | 4 +- + drivers/staging/vme_user/vme_tsi148.c | 3 +- + 49 files changed, 879 insertions(+), 1019 deletions(-) +Merging counter-next/counter-next (353b2e09f44a9 counter: ti-eqep: Remove redundant dev_err_probe()) +$ git merge -m Merge branch 'counter-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git counter-next/counter-next +Already up to date. +Merging mux/for-next (ac7bde3c53166 mux: Add driver for Renesas RZ/V2H VBENCTL VBUS_SEL mux) +$ git merge -m Merge branch 'for-next' of https://gitlab.com/peda-linux/mux.git mux/for-next +Merge made by the 'ort' strategy. + drivers/mux/Kconfig | 13 ++++++++ + drivers/mux/Makefile | 2 ++ + drivers/mux/rzv2h-vbenctl.c | 81 +++++++++++++++++++++++++++++++++++++++++++++ + 3 files changed, 96 insertions(+) + create mode 100644 drivers/mux/rzv2h-vbenctl.c +Merging dmaengine/next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git dmaengine/next +Already up to date. +Merging cgroup/for-next (75f8b845df69a Merge branch 'for-7.3-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/cgroup.git cgroup/for-next +Merge made by the 'ort' strategy. +Merging scsi/for-next (af8c27375733f scsi: megaraid_sas: Limit NVMe request size to the PRP chain frame) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jejb/scsi.git scsi/for-next +Already up to date. +Merging scsi-mkp/for-next (e83b47309f733 scsi: core: Drop Scsi_Host.default_lock) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git scsi-mkp/for-next +Auto-merging drivers/scsi/ibmvscsi/ibmvfc-core.c +Auto-merging drivers/scsi/megaraid/megaraid_sas_base.c +Merge made by the 'ort' strategy. + Documentation/scsi/scsi_mid_low_api.rst | 8 +- + drivers/ata/libata-eh.c | 4 +- + drivers/message/fusion/mptfc.c | 8 +- + drivers/s390/scsi/zfcp_erp.c | 30 +-- + drivers/s390/scsi/zfcp_sysfs.c | 4 +- + drivers/scsi/3w-9xxx.c | 20 +- + drivers/scsi/3w-sas.c | 28 +-- + drivers/scsi/3w-xxxx.c | 24 +-- + drivers/scsi/53c700.c | 16 +- + drivers/scsi/BusLogic.c | 20 +- + drivers/scsi/a100u2w.c | 4 +- + drivers/scsi/a2091.c | 4 +- + drivers/scsi/a3000.c | 4 +- + drivers/scsi/aacraid/commsup.c | 8 +- + drivers/scsi/advansys.c | 8 +- + drivers/scsi/aha1542.c | 20 +- + drivers/scsi/aha1740.c | 12 +- + drivers/scsi/arm/eesox.c | 8 +- + drivers/scsi/arm/fas216.c | 20 +- + drivers/scsi/atp870u.c | 12 +- + drivers/scsi/bfa/bfad_bsg.c | 4 +- + drivers/scsi/csiostor/csio_attr.c | 4 +- + drivers/scsi/csiostor/csio_scsi.c | 4 +- + drivers/scsi/dc395x.c | 8 +- + drivers/scsi/esp_scsi.c | 32 +-- + drivers/scsi/fcoe/fcoe.c | 4 +- + drivers/scsi/fdomain.c | 16 +- + drivers/scsi/gvp11.c | 4 +- + drivers/scsi/hosts.c | 17 +- + drivers/scsi/hptiop.c | 8 +- + drivers/scsi/ibmvscsi/ibmvfc-core.c | 262 ++++++++++++------------ + drivers/scsi/ibmvscsi/ibmvfc-nvme.c | 8 +- + drivers/scsi/ibmvscsi/ibmvscsi.c | 86 ++++---- + drivers/scsi/imm.c | 4 +- + drivers/scsi/initio.c | 8 +- + drivers/scsi/ipr.c | 324 +++++++++++++++--------------- + drivers/scsi/ips.c | 26 +-- + drivers/scsi/libfc/fc_fcp.c | 8 +- + drivers/scsi/libsas/sas_scsi_host.c | 10 +- + drivers/scsi/lpfc/lpfc_els.c | 30 +-- + drivers/scsi/lpfc/lpfc_hbadisc.c | 44 ++-- + drivers/scsi/lpfc/lpfc_init.c | 16 +- + drivers/scsi/lpfc/lpfc_scsi.c | 8 +- + drivers/scsi/lpfc/lpfc_sli.c | 4 +- + drivers/scsi/mac53c94.c | 8 +- + drivers/scsi/megaraid/megaraid_sas_base.c | 18 +- + drivers/scsi/mesh.c | 24 +-- + drivers/scsi/mvumi.c | 26 +-- + drivers/scsi/nsp32.c | 8 +- + drivers/scsi/pcmcia/sym53c500_cs.c | 8 +- + drivers/scsi/pmcraid.c | 84 ++++---- + drivers/scsi/qla1280.c | 44 ++-- + drivers/scsi/qla2xxx/qla_attr.c | 4 +- + drivers/scsi/qla2xxx/qla_init.c | 4 +- + drivers/scsi/qlogicfas408.c | 8 +- + drivers/scsi/qlogicpti.c | 20 +- + drivers/scsi/scsi.c | 12 +- + drivers/scsi/scsi_debugfs.c | 2 +- + drivers/scsi/scsi_devinfo.c | 5 +- + drivers/scsi/scsi_error.c | 70 +++---- + drivers/scsi/scsi_lib.c | 46 ++--- + drivers/scsi/scsi_proc.c | 9 +- + drivers/scsi/scsi_scan.c | 16 +- + drivers/scsi/scsi_sysfs.c | 28 +-- + drivers/scsi/scsi_transport_fc.c | 136 ++++++------- + drivers/scsi/sgiwd93.c | 4 +- + drivers/scsi/snic/snic_disc.c | 16 +- + drivers/scsi/stex.c | 48 ++--- + drivers/scsi/sym53c8xx_2/sym_glue.c | 56 +++--- + drivers/scsi/sym53c8xx_2/sym_hipd.h | 4 +- + drivers/scsi/wd33c93.c | 4 +- + drivers/scsi/wd719x.c | 24 +-- + drivers/scsi/xen-scsifront.c | 40 ++-- + drivers/target/target_core_file.c | 4 +- + drivers/target/target_core_pscsi.c | 16 +- + drivers/ufs/core/ufs-debugfs.c | 4 +- + drivers/ufs/core/ufs-sysfs.c | 12 +- + drivers/ufs/core/ufshcd.c | 152 +++++++------- + drivers/ufs/host/ufs-mediatek.c | 4 +- + drivers/ufs/host/ufs-sprd.c | 4 +- + drivers/usb/storage/uas.c | 12 +- + drivers/usb/storage/usb.h | 4 +- + include/scsi/scsi_host.h | 9 +- + 83 files changed, 1097 insertions(+), 1101 deletions(-) +$ git am -3 ../patches/device-id-zorro +Applying: foofof +Using index info to reconstruct a base tree... +M include/linux/device-id/zorro.h +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +Merging vhost/linux-next (ae864da7d762b vduse: Add suspend) +$ git merge -m Merge branch 'linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mst/vhost.git vhost/linux-next +Auto-merging MAINTAINERS +Auto-merging drivers/vhost/net.c +Auto-merging drivers/vhost/vhost.c +Auto-merging drivers/virtio/virtio_balloon.c +Auto-merging lib/iov_iter.c +Auto-merging net/vmw_vsock/virtio_transport_common.c +Merge made by the 'ort' strategy. + drivers/vhost/vhost.c | 11 ++++++ + drivers/vhost/vsock.c | 80 +++++++++++++++++++++++++++++++---------- + drivers/virtio/virtio_balloon.c | 51 ++++++++++++++++---------- + 3 files changed, 105 insertions(+), 37 deletions(-) +Merging rpmsg/for-next (c0bdd460b8fd5 Merge branches 'rproc-next', 'rproc-fixes' and 'rpmsg-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/remoteproc/linux.git rpmsg/for-next +Merge made by the 'ort' strategy. + .../bindings/remoteproc/ti,am3352-wkup-m3.yaml | 1 - + drivers/remoteproc/imx_rproc.c | 33 +--------------------- + drivers/remoteproc/qcom_q6v5_adsp.c | 10 +++---- + drivers/remoteproc/qcom_q6v5_mss.c | 11 +++++++- + drivers/remoteproc/qcom_q6v5_pas.c | 12 ++++---- + drivers/remoteproc/remoteproc_elf_loader.c | 23 +++++++++++++-- + drivers/remoteproc/stm32_rproc.c | 2 +- + drivers/remoteproc/ti_k3_dsp_remoteproc.c | 4 +++ + drivers/remoteproc/ti_k3_r5_remoteproc.c | 4 +++ + drivers/remoteproc/xlnx_r5_remoteproc.c | 13 +++++++-- + 10 files changed, 63 insertions(+), 50 deletions(-) +Merging gpio-brgl/gpio/for-next (a2cfc48fee416 gpio: altera: Fix build failure caused by undeclared 'irq') +$ git merge -m Merge branch 'gpio/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git gpio-brgl/gpio/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/gpio/fsl,qoriq-gpio.yaml | 7 ++ + .../bindings/gpio/nvidia,tegra186-gpio.yaml | 9 ++ + drivers/gpio/gpio-adp5585.c | 20 +--- + drivers/gpio/gpio-altera.c | 5 +- + drivers/gpio/gpio-eic-sprd.c | 17 +--- + drivers/gpio/gpio-mlxbf2.c | 4 +- + drivers/gpio/gpio-mlxbf3.c | 4 +- + drivers/gpio/gpio-mmio.c | 5 + + drivers/gpio/gpio-omap.c | 11 +-- + drivers/gpio/gpio-pcf857x.c | 16 +++- + drivers/gpio/gpio-realtek-otto.c | 16 +++- + drivers/gpio/gpio-rtd1625.c | 26 ++--- + drivers/gpio/gpio-sloppy-logic-analyzer.c | 4 +- + drivers/gpio/gpio-tegra186.c | 74 +++++++++++++++ + drivers/gpio/gpiolib-acpi-quirks.c | 13 +++ + drivers/gpio/gpiolib-cdev.c | 2 + + drivers/gpio/gpiolib-kunit.c | 14 +-- + include/linux/notifier.h | 7 ++ + kernel/notifier.c | 105 +++++++++++++++++++++ + 19 files changed, 285 insertions(+), 74 deletions(-) +Merging gpio-intel/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git gpio-intel/for-next +Already up to date. +Merging pinctrl/for-next (d067f0f4c96cb Merge tag 'pinctrl-qcom-updates-for-v7.3-rc1' of git://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux into devel) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/linusw/linux-pinctrl.git pinctrl/for-next +Already up to date. +Merging pinctrl-intel/for-next (ed1608d5e3c44 Merge patch series "upboard pinctrl support for device id INTC1055") +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git pinctrl-intel/for-next +Already up to date. +Merging pinctrl-renesas/renesas-pinctrl (d48a308fabd6d pinctrl: renesas: rzg2l: Add SD channel POC support for RZ/G3L) +$ git merge -m Merge branch 'renesas-pinctrl' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git pinctrl-renesas/renesas-pinctrl +Merge made by the 'ort' strategy. + drivers/pinctrl/renesas/pinctrl-rzg2l.c | 74 ++++++++++++++++++++++----------- + 1 file changed, 50 insertions(+), 24 deletions(-) +Merging pinctrl-samsung/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git pinctrl-samsung/for-next +Already up to date. +Merging pinctrl-qcom/pinctrl-qcom/for-next (d0344c6510d05 pinctrl: qcom: Add Kuno pinctrl driver) +$ git merge -m Merge branch 'pinctrl-qcom/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pinctrl-qcom/pinctrl-qcom/for-next +Merge made by the 'ort' strategy. + .../pinctrl/qcom,hawi-lpass-lpi-pinctrl.yaml | 109 +++ + .../bindings/pinctrl/qcom,kuno-tlmm.yaml | 110 +++ + .../bindings/pinctrl/qcom,pmic-gpio.yaml | 3 + + drivers/pinctrl/qcom/Kconfig | 10 + + drivers/pinctrl/qcom/Kconfig.msm | 11 + + drivers/pinctrl/qcom/Makefile | 2 + + drivers/pinctrl/qcom/pinctrl-hawi-lpass-lpi.c | 244 +++++++ + drivers/pinctrl/qcom/pinctrl-kuno.c | 801 +++++++++++++++++++++ + drivers/pinctrl/qcom/pinctrl-lpass-lpi.h | 17 + + drivers/pinctrl/qcom/pinctrl-spmi-gpio.c | 1 + + 10 files changed, 1308 insertions(+) + create mode 100644 Documentation/devicetree/bindings/pinctrl/qcom,hawi-lpass-lpi-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/qcom,kuno-tlmm.yaml + create mode 100644 drivers/pinctrl/qcom/pinctrl-hawi-lpass-lpi.c + create mode 100644 drivers/pinctrl/qcom/pinctrl-kuno.c +Merging pwm/pwm/for-next (6b0c6ff76795f pwm: iqs620a: Use devm_blocking_notifier_chain_register()) +$ git merge -m Merge branch 'pwm/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git pwm/pwm/for-next +Merge made by the 'ort' strategy. + drivers/pwm/pwm-iqs620a.c | 23 +++-------------------- + drivers/pwm/pwm-renesas-tpu.c | 23 ++++++++++++++++++----- + 2 files changed, 21 insertions(+), 25 deletions(-) +Merging ktest/for-next (932cdaf3e273a ktest: Add logfile to failure directory) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rostedt/linux-ktest.git ktest/for-next +Already up to date. +Merging kselftest/next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kselftest/next +Already up to date. +Merging kunit/test (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'test' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kunit/test +Already up to date. +Merging kunit-next/kunit (e38f53f048246 kunit: Return void from kunit_run_all_tests()) +$ git merge -m Merge branch 'kunit' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kunit-next/kunit +Merge made by the 'ort' strategy. + include/kunit/test.h | 5 ++--- + lib/kunit/executor.c | 5 ++--- + 2 files changed, 4 insertions(+), 6 deletions(-) +Merging livepatching/for-next (26260251022fb Merge tag 'livepatching-for-7.3' of git://git.kernel.org/pub/scm/linux/kernel/git/livepatching/livepatching) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/livepatching/livepatching.git livepatching/for-next +Already up to date. +Merging rtc/rtc-next (afce9701d6423 MAINTAINERS: update rtc subsystem patchwork location) +$ git merge -m Merge branch 'rtc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git rtc/rtc-next +Already up to date. +Merging nvdimm/libnvdimm-for-next (e99cb3ecd8334 nvdimm-btt: clean up kernel-doc warnings) +$ git merge -m Merge branch 'libnvdimm-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git nvdimm/libnvdimm-for-next +Already up to date. +Merging at24/at24/for-next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'at24/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git at24/at24/for-next +Already up to date. +Merging ntb/ntb-next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'ntb-next' of https://github.com/jonmason/ntb.git ntb/ntb-next +Already up to date. +Merging seccomp/for-next/seccomp (41fa043273841 selftests/seccomp: Add hard-coded __NR_uprobe for x86_64) +$ git merge -m Merge branch 'for-next/seccomp' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git seccomp/for-next/seccomp +Already up to date. +Merging slimbus/for-next (c922423ce66bc slimbus: qcom-ngd-ctrl: Use the unified QMI service ID instead of defining it locally) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/srini/slimbus.git slimbus/for-next +Auto-merging drivers/slimbus/qcom-ngd-ctrl.c +Merge made by the 'ort' strategy. + drivers/slimbus/qcom-ngd-ctrl.c | 5 ++--- + 1 file changed, 2 insertions(+), 3 deletions(-) +Merging nvmem/for-next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/srini/nvmem.git nvmem/for-next +Already up to date. +Merging hyperv/hyperv-next (be0cfab740e58 clocksource: hyper-v: Remove support for stimer interrupts in message mode) +$ git merge -m Merge branch 'hyperv-next' of https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git hyperv/hyperv-next +Already up to date. +Merging auxdisplay/for-next (c868291c2b3d7 auxdisplay: arm-charlcd: Use DEFINE_SIMPLE_DEV_PM_OPS for power management) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git auxdisplay/for-next +Merge made by the 'ort' strategy. + drivers/auxdisplay/arm-charlcd.c | 7 ++----- + 1 file changed, 2 insertions(+), 5 deletions(-) +Merging kgdb/kgdb/for-next (fdbdd0ccb30af kdb: remove redundant check for scancode 0xe0) +$ git merge -m Merge branch 'kgdb/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/danielt/linux.git kgdb/kgdb/for-next +Already up to date. +Merging hmm/hmm (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'hmm' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git hmm/hmm +Already up to date. +Merging cfi/cfi/next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'cfi/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git cfi/cfi/next +Already up to date. +Merging mhi/mhi-next (83c29a55b89e0 bus: mhi: host: pci_generic: Add IP_CTRL channel for Sierra EM919x/EM929x) +$ git merge -m Merge branch 'mhi-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mani/mhi.git mhi/mhi-next +Merge made by the 'ort' strategy. + drivers/bus/mhi/host/pci_generic.c | 2 ++ + 1 file changed, 2 insertions(+) +$ git am -3 ../patches/0001-fix-up-for-net-qrtr-Drop-the-MHI-auto_queue-feature-.patch +Applying: fix up for "net: qrtr: Drop the MHI auto_queue feature for IPCR DL channels" +Using index info to reconstruct a base tree... +M drivers/net/wireless/ath/ath12k/wifi7/mhi.c +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +Merging cxl/next (274b30592196c Merge branch 'for-7.4/cxl-misc' into cxl-for-next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git cxl/next +Merge made by the 'ort' strategy. + drivers/cxl/core/mce.c | 8 +++++++- + drivers/cxl/core/region.c | 5 +++-- + tools/testing/cxl/Kbuild | 16 ++++++++++------ + tools/testing/cxl/test/cxl.c | 8 ++++---- + 4 files changed, 24 insertions(+), 13 deletions(-) +Merging zstd/zstd-next (65d1f5507ed2c zstd: Import upstream v1.5.7) +$ git merge -m Merge branch 'zstd-next' of https://github.com/terrelln/linux.git zstd/zstd-next +Already up to date. +Merging efi/next (42bf9d266e528 efi: add dynamic control interface for EFI runtime services) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git efi/next +Merge made by the 'ort' strategy. + drivers/firmware/efi/capsule.c | 4 ++-- + drivers/firmware/efi/efi.c | 31 +++++++++++++++++++++++++++++++ + drivers/firmware/efi/runtime-wrappers.c | 28 ++++++++++++++++++++++++++++ + include/linux/efi.h | 1 + + 4 files changed, 62 insertions(+), 2 deletions(-) +Merging unicode/for-next (a511442085c14 unicode: Properly reject invalid encoding version strings) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krisman/unicode.git unicode/for-next +Already up to date. +Merging random/master (26f5abb98a9cd virt: vmgenid: move to using dev_set/get_drvdata) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/crng/random.git random/master +Merge made by the 'ort' strategy. + drivers/virt/vmgenid.c | 11 +++++++---- + 1 file changed, 7 insertions(+), 4 deletions(-) +Merging landlock/next (172b6a6d84635 landlock: Document tracepoints) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/mic/linux.git landlock/next +Already up to date. +Merging sysctl/sysctl-next (8d75c338f0bce sysctl: remove CONFIG_PROC_SYSCTL, it just mirrors CONFIG_SYSCTL) +$ git merge -m Merge branch 'sysctl-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sysctl/sysctl.git sysctl/sysctl-next +Already up to date. +$ git am -3 ../patches/0001-jifies-Fix-up-merge.patch +Applying: jifies: Fix up merge +Using index info to reconstruct a base tree... +M kernel/time/jiffies.c +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +Merging execve/for-next/execve (ab11176bd3a76 x86/elf: Correct comment for STACK_RND_MASK()) +$ git merge -m Merge branch 'for-next/execve' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git execve/for-next/execve +Already up to date. +Merging bitmap/bitmap-for-next (cf72cbb39da84 Merge tag 'io_uring-7.3-20260828' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux) +$ git merge -m Merge branch 'bitmap-for-next' of https://github.com/norov/linux.git bitmap/bitmap-for-next +Already up to date. +Merging hte/for-next (1329abe1bae41 hte: tegra194: Initialize slice locks before registering chip) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pateldipen1984/linux.git hte/for-next +Merge made by the 'ort' strategy. + drivers/hte/hte-tegra194.c | 14 ++++++-------- + 1 file changed, 6 insertions(+), 8 deletions(-) +Merging kspp/for-next/kspp (1b501c6afc7a6 Merge branch 'for-next/hardening' into for-next/kspp) +$ git merge -m Merge branch 'for-next/kspp' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git kspp/for-next/kspp +Merge made by the 'ort' strategy. + drivers/misc/lkdtm/core.c | 16 ++++++++-------- + fs/signalfd.c | 28 ++++++++++++++++++++++------ + include/linux/fortify-string.h | 2 -- + 3 files changed, 30 insertions(+), 16 deletions(-) +Merging nolibc/for-next (6ef847683e7d2 tools/nolibc: validate directory with O_DIRECTORY in opendir()) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/nolibc/linux-nolibc.git nolibc/for-next +Merge made by the 'ort' strategy. + tools/include/nolibc/Makefile | 19 ++- + tools/include/nolibc/arch-hexagon.h | 164 +++++++++++++++++++++++++ + tools/include/nolibc/arch.h | 2 + + tools/include/nolibc/dirent.h | 18 ++- + tools/include/nolibc/nolibc.h | 1 + + tools/include/nolibc/sys/sendfile.h | 41 +++++++ + tools/testing/selftests/nolibc/Makefile.nolibc | 10 +- + tools/testing/selftests/nolibc/nolibc-test.c | 70 +++++++++++ + tools/testing/selftests/nolibc/run-tests.sh | 24 +++- + 9 files changed, 341 insertions(+), 8 deletions(-) + create mode 100644 tools/include/nolibc/arch-hexagon.h + create mode 100644 tools/include/nolibc/sys/sendfile.h +Merging iommufd/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git iommufd/for-next +Already up to date. +Merging turbostat/next (1c996a37fd244 tools/power turbostat: pmt: Improve sscanf() hygiene) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/lenb/linux.git turbostat/next +Merge made by the 'ort' strategy. + tools/power/x86/turbostat/turbostat.8 | 6 +- + tools/power/x86/turbostat/turbostat.c | 567 ++++++++++++++++------------------ + 2 files changed, 266 insertions(+), 307 deletions(-) +Merging pwrseq/pwrseq/for-next (3b04e9b8056e8 power: sequencing: Add Renesas RZ/G3L Power Ready driver) +$ git merge -m Merge branch 'pwrseq/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pwrseq/pwrseq/for-next +Auto-merging drivers/power/sequencing/Kconfig +Merge made by the 'ort' strategy. + drivers/power/sequencing/Kconfig | 9 ++ + drivers/power/sequencing/Makefile | 1 + + drivers/power/sequencing/pwrseq-renesas-pwrrdy.c | 142 +++++++++++++++++++++++ + 3 files changed, 152 insertions(+) + create mode 100644 drivers/power/sequencing/pwrseq-renesas-pwrrdy.c +Merging capabilities-next/caps-next (9b0ee13b80798 capability: remove non-kernel-doc comments) +$ git merge -m Merge branch 'caps-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sergeh/linux.git capabilities-next/caps-next +Already up to date. +$ git am -3 ../patches/0001-sign-file-Fix-up-merge-issue.patch +Applying: sign-file: Fix up merge issue +Using index info to reconstruct a base tree... +M scripts/sign-file.c +Falling back to patching base and 3-way merge... +Auto-merging scripts/sign-file.c +No changes -- Patch already applied. +Merging ipe/next (028ef9c96e961 Linux 7.0) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/wufan/ipe.git ipe/next +Already up to date. +Merging kcsan/next (a8488ecbd7ba4 kcsan: avoid unintended access checking in NMIs) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/melver/linux.git kcsan/next +Already up to date. +Merging crc/crc-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'crc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git crc/crc-next +Already up to date. +Merging keys-next/keys-next (965e9a2cf23b0 pkcs7: Change a pr_warn() to pr_warn_once()) +$ git merge -m Merge branch 'keys-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dhowells/linux-fs.git keys-next/keys-next +Already up to date. +Merging fwctl/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git fwctl/for-next +Already up to date. +$ git am -3 ../patches/rust-fwctl-replace-__pinned_init-with-raw_try_init +Applying: rust: fwctl: replace `__pinned_init` with `raw_try_init` +Using index info to reconstruct a base tree... +M rust/kernel/fwctl.rs +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +Merging devsec-tsm/next (3177779ae17db virt: coco: change tsm_class to a const struct) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git devsec-tsm/next +Already up to date. +Merging hisilicon/for-next (a5db65458a911 Merge branch 'next/dt64' into for-next) +$ git merge -m Merge branch 'for-next' of https://github.com/hisilicon/linux-hisi.git hisilicon/for-next +Merge made by the 'ort' strategy. +Merging device-id/device-id-rework (995832b2cebe6 Replace by more specific (c files)) +$ git merge -m Merge branch 'device-id-rework' of https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git device-id/device-id-rework +Already up to date. +Merging kthread/for-next (fa39ec4f89f26 doc: Add housekeeping documentation) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/frederic/linux-dynticks.git kthread/for-next +Already up to date. +Merging pagemap-headers/headers (e02cb91d4644d ksm: Remove pagemap.h include) +$ git merge -m Merge branch 'headers' of git://git.infradead.org/users/willy/pagecache.git pagemap-headers/headers +Auto-merging arch/alpha/include/asm/pgtable.h +Auto-merging arch/arc/include/asm/pgtable-levels.h +Auto-merging arch/microblaze/include/asm/pgtable.h +Auto-merging arch/sparc/include/asm/pgtable_64.h +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gpuvm.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c +Auto-merging drivers/gpu/drm/amd/amdkfd/kfd_migrate.c +Auto-merging drivers/gpu/drm/msm/msm_fb.c +Auto-merging drivers/gpu/drm/nouveau/nouveau_dmem.c +Auto-merging drivers/gpu/drm/panthor/panthor_gem.c +Auto-merging drivers/hwmon/pmbus/pmbus_core.c +Auto-merging drivers/i2c/busses/i2c-amd-asf-plat.c +Auto-merging drivers/i2c/busses/i2c-gxp.c +Auto-merging drivers/i2c/busses/i2c-k1.c +Auto-merging drivers/iio/accel/fxls8962af-core.c +Auto-merging drivers/iio/adc/ti-ads1298.c +Auto-merging drivers/iio/light/rohm-bu27034.c +Auto-merging drivers/iio/pressure/bmp280-core.c +Auto-merging drivers/mfd/88pm886.c +Auto-merging drivers/mfd/cs42l43.c +Auto-merging drivers/pinctrl/pinctrl-sx150x.c +Auto-merging drivers/regulator/fixed.c +Auto-merging drivers/regulator/tps65185.c +Auto-merging drivers/ufs/host/ufs-mediatek.c +Auto-merging drivers/ufs/host/ufs-qcom.c +Auto-merging fs/aio.c +Auto-merging fs/buffer.c +Auto-merging fs/inode.c +Auto-merging fs/nfs/nfstrace.h +Auto-merging fs/nfs/pnfs.c +Auto-merging fs/nfsd/nfs4proc.c +Auto-merging include/linux/ceph/libceph.h +Auto-merging include/linux/i3c/master.h +Auto-merging include/linux/nfs_fs.h +Auto-merging include/linux/swap.h +Auto-merging kernel/power/snapshot.c +Auto-merging net/ceph/osd_client.c +CONFLICT (content): Merge conflict in net/ceph/osd_client.c +Auto-merging net/ceph/pagevec.c +Auto-merging net/sunrpc/xprtsock.c +Auto-merging security/selinux/hooks.c +Auto-merging security/selinux/selinuxfs.c +Resolved 'net/ceph/osd_client.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 9f308e750b949] Merge branch 'headers' of git://git.infradead.org/users/willy/pagecache.git +$ git diff -M --stat --summary HEAD^.. + arch/alpha/include/asm/pgtable.h | 2 +- + arch/arc/include/asm/Kbuild | 1 + + arch/arc/include/asm/pgtable-levels.h | 2 ++ + arch/arc/include/asm/tlb.h | 12 ---------- + arch/arm/include/asm/pgalloc.h | 2 -- + arch/arm/include/asm/tlb.h | 2 -- + arch/arm/mach-pxa/pxa3xx.c | 1 + + arch/arm64/include/asm/tlb.h | 3 --- + arch/hexagon/include/asm/tlb.h | 1 - + arch/microblaze/include/asm/pgtable.h | 2 +- + arch/nios2/include/asm/tlb.h | 1 - + arch/openrisc/include/asm/Kbuild | 1 + + arch/openrisc/include/asm/tlb.h | 26 ---------------------- + arch/powerpc/include/asm/tlb.h | 2 -- + arch/sh/include/asm/pgtable.h | 2 +- + arch/sh/include/asm/tlb.h | 1 - + arch/sparc/include/asm/pgtable_64.h | 2 +- + arch/sparc/include/asm/tlb_64.h | 1 - + arch/x86/include/asm/pgalloc.h | 1 - + arch/x86/virt/vmx/tdx/tdx.c | 1 + + arch/xtensa/kernel/hibernate.c | 1 + + block/fops.c | 1 + + drivers/char/agp/backend.c | 1 - + drivers/char/agp/generic.c | 1 - + drivers/char/agp/intel-agp.c | 1 - + drivers/char/agp/intel-gtt.c | 1 - + drivers/char/agp/uninorth-agp.c | 3 ++- + drivers/dma/ste_dma40.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gpuvm.c | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_cs.c | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c | 3 +-- + drivers/gpu/drm/amd/amdkfd/kfd_migrate.c | 1 + + drivers/gpu/drm/amd/amdkfd/kfd_priv.h | 1 - + drivers/gpu/drm/display/drm_dp_cec.c | 1 + + drivers/gpu/drm/etnaviv/etnaviv_gpu.c | 1 + + drivers/gpu/drm/i915/display/intel_display_power.c | 1 + + drivers/gpu/drm/i915/gem/i915_gem_userptr.c | 1 + + drivers/gpu/drm/msm/msm_fb.c | 2 ++ + drivers/gpu/drm/msm/msm_gem_shrinker.c | 2 ++ + drivers/gpu/drm/nouveau/nouveau_dmem.c | 1 + + drivers/gpu/drm/omapdrm/dss/hdmi4.c | 1 + + drivers/gpu/drm/omapdrm/dss/hdmi5.c | 1 + + drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.c | 1 + + drivers/gpu/drm/panfrost/panfrost_gem.c | 2 ++ + drivers/gpu/drm/panthor/panthor_gem.c | 3 +++ + drivers/hwmon/pmbus/pmbus_core.c | 1 + + drivers/i2c/busses/i2c-amd-asf-plat.c | 1 + + drivers/i2c/busses/i2c-gxp.c | 1 + + drivers/i2c/busses/i2c-k1.c | 1 + + drivers/i2c/busses/i2c-pasemi-platform.c | 1 + + drivers/iio/accel/fxls8962af-core.c | 1 + + drivers/iio/adc/nct7201.c | 1 + + drivers/iio/adc/ti-ads1298.c | 1 + + drivers/iio/light/rohm-bu27034.c | 1 + + drivers/iio/pressure/bmp280-core.c | 1 + + drivers/iio/pressure/bmp280.h | 1 + + drivers/input/misc/aw86927.c | 1 + + drivers/input/touchscreen/goodix_berlin_core.c | 1 + + drivers/input/touchscreen/hynitron_cstxxx.c | 1 + + drivers/input/touchscreen/imagis.c | 1 + + drivers/leds/leds-turris-omnia.c | 1 + + drivers/media/common/videobuf2/frame_vector.c | 1 - + drivers/media/pci/cx18/cx18-driver.h | 1 - + drivers/media/pci/ivtv/ivtv-driver.h | 2 +- + drivers/media/platform/st/stm32/stm32-csi.c | 1 + + drivers/media/platform/ti/omap3isp/ispvideo.c | 1 - + drivers/media/usb/go7007/go7007-v4l2.c | 1 - + drivers/media/usb/gspca/gspca.c | 1 - + drivers/mfd/88pm886.c | 1 + + drivers/mfd/abx500-core.c | 1 + + drivers/mfd/adp5585.c | 1 + + drivers/mfd/cs40l50-core.c | 1 + + drivers/mfd/cs42l43.c | 1 + + drivers/mfd/rt5120.c | 1 + + drivers/mfd/tps65219.c | 1 + + drivers/mfd/twl-core.c | 2 +- + drivers/mmc/core/core.c | 1 - + drivers/mmc/core/host.c | 1 - + drivers/mmc/host/renesas_sdhi_internal_dmac.c | 1 - + drivers/mmc/host/renesas_sdhi_sys_dmac.c | 1 - + drivers/mmc/host/sh_mmcif.c | 1 - + drivers/mmc/host/tmio_mmc.h | 1 - + drivers/mmc/host/tmio_mmc_core.c | 1 - + drivers/mmc/host/usdhi6rol0.c | 1 - + drivers/net/ethernet/atheros/atl1c/atl1c.h | 1 - + drivers/net/ethernet/atheros/atl1e/atl1e.h | 1 - + drivers/net/ethernet/intel/e1000/e1000.h | 1 - + drivers/net/ethernet/intel/e1000e/netdev.c | 1 - + drivers/net/ethernet/intel/igb/igb_main.c | 1 - + drivers/net/ethernet/intel/igbvf/netdev.c | 1 - + drivers/phy/st/phy-stm32-combophy.c | 1 + + drivers/pinctrl/pinctrl-sx150x.c | 1 + + drivers/platform/arm64/huawei-gaokun-ec.c | 1 + + drivers/platform/arm64/lenovo-yoga-c630.c | 2 +- + .../platform/x86/intel/int3472/clk_and_regulator.c | 1 + + drivers/power/supply/bq25630_charger.c | 1 + + drivers/power/supply/rk817_charger.c | 2 +- + drivers/regulator/fixed.c | 1 + + drivers/regulator/fp9931.c | 1 + + drivers/regulator/mt6360-regulator.c | 1 + + drivers/regulator/qcom-labibb-regulator.c | 1 + + drivers/regulator/rtq2208-regulator.c | 1 + + drivers/regulator/tps65185.c | 1 + + drivers/regulator/tps65219-regulator.c | 1 + + drivers/regulator/tps6594-regulator.c | 1 + + drivers/soc/xilinx/zynqmp_power.c | 1 + + drivers/ufs/host/ufs-mediatek.c | 1 + + drivers/ufs/host/ufs-qcom.c | 1 + + drivers/usb/core/hcd.c | 1 + + drivers/usb/typec/mux/it5205.c | 1 + + drivers/usb/typec/wusb3801.c | 1 + + drivers/video/fbdev/core/fb_procfs.c | 1 + + drivers/xen/balloon.c | 1 + + fs/aio.c | 1 + + fs/buffer.c | 1 + + fs/file_table.c | 1 + + fs/inode.c | 1 + + fs/nfs/blocklayout/blocklayout.c | 1 + + fs/nfs/nfs42proc.c | 1 + + fs/nfs/nfstrace.h | 1 + + fs/nfs/pnfs.c | 1 + + fs/nfsd/nfs4proc.c | 1 + + include/drm/drm_print.h | 1 + + include/drm/ttm/ttm_tt.h | 1 - + include/linux/balloon.h | 1 - + include/linux/ceph/libceph.h | 1 - + include/linux/i3c/master.h | 1 + + include/linux/ksm.h | 1 - + include/linux/mempolicy.h | 1 - + include/linux/nfs_fs.h | 1 - + include/linux/nfs_page.h | 1 - + include/linux/suspend.h | 1 - + include/linux/swap.h | 1 - + include/sound/cs35l41.h | 1 + + kernel/power/snapshot.c | 1 + + net/ceph/osd_client.c | 1 - + net/ceph/pagevec.c | 1 + + net/core/datagram.c | 1 - + net/rds/rdma.c | 1 - + net/sunrpc/auth_gss/auth_gss.c | 1 - + net/sunrpc/auth_gss/gss_krb5_crypto.c | 1 - + net/sunrpc/auth_gss/gss_krb5_wrap.c | 1 - + net/sunrpc/auth_gss/svcauth_gss.c | 1 - + net/sunrpc/cache.c | 1 - + net/sunrpc/rpc_pipe.c | 1 - + net/sunrpc/socklib.c | 1 - + net/sunrpc/xdr.c | 1 - + net/sunrpc/xprtsock.c | 1 - + security/commoncap.c | 3 --- + security/inode.c | 1 - + security/selinux/hooks.c | 1 - + security/selinux/selinuxfs.c | 1 - + security/smack/smack_lsm.c | 1 - + sound/soc/codecs/cs35l41-lib.c | 1 + + 155 files changed, 98 insertions(+), 118 deletions(-) + delete mode 100644 arch/arc/include/asm/tlb.h + delete mode 100644 arch/openrisc/include/asm/tlb.h diff --git a/localversion-next b/localversion-next new file mode 100644 index 00000000000000..00309da46a3316 --- /dev/null +++ b/localversion-next @@ -0,0 +1 @@ +-next-20260903 From 3e79edaff7e606a84f6527476b0023937a318578 Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Tue, 18 Aug 2026 20:14:06 +0000 Subject: [PATCH 832/857] ogc: linux-unstable: first commit -- 2.47.3 (cherry picked from commit 536526d1b587ace66924207d8dcbfd0a50279aec) (cherry picked from commit 0ebc551a74d0ee721a6f59870a5a88ff88e42268) (cherry picked from commit 2e5980a43d81039ee5a4b02fc8ff6637fead084d) -- 2.47.3 -- 2.47.3 (cherry picked from commit 82fa59412110c4bc4906a7f0e76c12fb533a7ec2) From 82c2a9719cf58da1731e6bebeaaf8e4862b3ff0a Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Tue, 1 Sep 2026 19:15:25 +0000 Subject: [PATCH 833/857] [NOT-FOR-UPSTREAM] ogc: linux-unstable: add github workflow (cherry picked from commit 64a16187d10090ade4d9134f4047f343e9856346) --- .github/packaging/PKGBUILD | 262 ++++++++++ .github/packaging/config.fragment | 124 +++++ .github/packaging/fedora/kernel.spec | 279 ++++++++++ .github/packaging/merge-fragments.sh | 62 +++ .github/workflows/build-kernel.yml | 721 ++++++++++++++++++++++++++ .github/workflows/sync-linux-next.yml | 137 +++++ .github/workflows/test-pr.yml | 227 ++++++++ 7 files changed, 1812 insertions(+) create mode 100644 .github/packaging/PKGBUILD create mode 100644 .github/packaging/config.fragment create mode 100644 .github/packaging/fedora/kernel.spec create mode 100644 .github/packaging/merge-fragments.sh create mode 100644 .github/workflows/build-kernel.yml create mode 100644 .github/workflows/sync-linux-next.yml create mode 100644 .github/workflows/test-pr.yml diff --git a/.github/packaging/PKGBUILD b/.github/packaging/PKGBUILD new file mode 100644 index 00000000000000..85dcd59ff2abe7 --- /dev/null +++ b/.github/packaging/PKGBUILD @@ -0,0 +1,262 @@ +# SPDX-License-Identifier: GPL-2.0-only +# Maintainer: OpenGamingCollective +# +# PKGBUILD for the linux-unstable-ogc kernel. +# +# It is designed to be built by the CI job defined in +# .github/workflows/build-kernel.yml, which stages the following next to this +# file before invoking makepkg: +# - linux.tar.gz -> tarball of the checked-out kernel tree (no VCS data) +# - config -> Arch linux-headers .config with the OGC +# kernel-packages config fragments already applied +# - config.fragment -> repo-local overrides merged on top of it +# +# Based on the official Arch Linux kernel PKGBUILD and on the (known to work +# with linux-next) https://github.com/NeroReflex/linux-bisector PKGBUILD. + +pkgbase=linux-unstable-ogc +# Placeholder: makepkg refuses an empty pkgver before pkgver() runs; the real +# version is computed dynamically by pkgver() once sources are extracted. +pkgver=0.0.0 +pkgrel=1 +pkgdesc='linux-next kernel for the Open Gaming Collective' +url='https://github.com/OpenGamingCollective/linux-unstable' +arch=(x86_64) +license=(GPL2) +makedepends=( + bc + cpio + gettext + libelf + pahole + perl + python + tar + xz + gcc + ccache + git + + llvm + clang + lld + + # CONFIG_RUST=y in the base config + rust + rust-bindgen +) +options=('!strip') +source=( + 'linux.tar.gz' + 'config' + 'config.fragment' +) +b2sums=( + 'SKIP' + 'SKIP' + 'SKIP' +) + +export KBUILD_BUILD_HOST=archlinux +export KBUILD_BUILD_USER=$pkgbase +export KBUILD_BUILD_TIMESTAMP="" +export CC="ccache clang" +export MAKEFLAGS="-j$(nproc)" + +_make() { + test -s version + LLVM=1 LLVM_IAS=1 WERROR=0 KBUILD_BUILD_TIMESTAMP="" make CC="$CC" KERNELRELEASE="$(.. + $(cat localversion-next) + # + the git short sha recorded by CI in .build_commit + # e.g. "7.2.0" + "-next-20260818" + ".g4ca2fc86" -> "7.2.0-next-20260818.g4ca2fc86". + # Pacman does not allow hyphens in pkgver, so they become underscores: + # 7.2.0-next-20260818.g4ca2fc86 -> 7.2.0_next_20260818.g4ca2fc86 + cd "$srcdir/linux" + local ver + ver="$(make -s kernelversion)$(cat localversion-next 2>/dev/null || true)" + if [[ -s .build_commit ]]; then + ver+=".g$(<.build_commit)" + fi + printf '%s\n' "${ver//-/_}" +} + +prepare() { + cd "$srcdir/linux" + + # glibc >= 2.42 made strstr/strchr const-correct; some trees hardcode + # -Werror in tools/lib/bpf which then fails to compile. Keep builds green. + if [[ -f tools/lib/bpf/Makefile ]]; then + sed -i 's/ -Werror -Wall/ -Wall/' tools/lib/bpf/Makefile || true + fi + + echo "Setting version..." + # localversion* files are picked up sorted by name, so the final kernel + # release becomes e.g.: 7.2.0-next-20260818-unstable-ogc-g4ca2fc86-1 + # The -g suffix comes from the .build_commit file the CI writes into + # the tarball (same trick linux-bisector uses with .bisector_commit). + local _commit="" + if [[ -s .build_commit ]]; then + _commit="-g$(<.build_commit)" + fi + echo "${pkgbase#linux}${_commit}" > localversion.10-pkgname + echo "-$pkgrel" > localversion.20-pkgrel + LLVM=1 LLVM_IAS=1 make defconfig + LLVM=1 LLVM_IAS=1 make -s kernelrelease > version + LLVM=1 LLVM_IAS=1 make mrproper + + echo "Setting config..." + cp ../config .config + _make olddefconfig + + echo "Applying OGC config fragment..." + scripts/kconfig/merge_config.sh -m .config ../config.fragment + _make olddefconfig + + diff -u ../config .config || : + + echo "Prepared $pkgbase version $(-unstable-ogc-g-1 (the suffix +# itself comes from the localversion.10/20 files written by the packaging). +CONFIG_LOCALVERSION="" +# CONFIG_LOCALVERSION_AUTO is not set + +# --- Build fixes ------------------------------------------------------------ +# drivers/net/ethernet/cadence/macb_main.c fails to compile in linux-next +# (implicit declarations of macb_alloc_tieoff/macb_free_tieoff). Cadence +# MACB/GEM is an ARM SoC NIC, never present on x86_64 gaming hardware. +# MACB_PCI and MACB_USE_HWSTAMP depend on MACB and drop out automatically. +# CONFIG_MACB is not set + +# --- Build speed ------------------------------------------------------------ +# Keep DEBUG_INFO/DEBUG_INFO_BTF=y (OGC requires them for BPF/sched_ext) but +# skip running pahole on every module (thousands of invocations saved). +# CONFIG_DEBUG_INFO_BTF_MODULES is not set + +# --- Xen guest support (not running inside a Xen VM) ------------------------ +# CONFIG_XEN is not set + +# --- Ancient graphics / buses ------------------------------------------------ +# AGP is a pre-PCIe bus; radeon covers pre-GCN AMD GPUs (12+ years old). +# CONFIG_AGP is not set +# CONFIG_DRM_RADEON is not set +# PCMCIA/CardBus slots (2000s laptops) +# CONFIG_PCCARD is not set +# FireWire (IEEE 1394) ports +# CONFIG_FIREWIRE is not set +# Parallel port (printers/scanners from the 90s) +# CONFIG_PARPORT is not set +# IndustryPack carrier boards +# CONFIG_IPACK_BUS is not set +# 1-Wire bus +# CONFIG_W1 is not set +# Analog gameport joysticks (pre-USB) +# CONFIG_GAMEPORT is not set +# Floppy disk controller +# CONFIG_BLK_DEV_FD is not set +# Server BMC management +# CONFIG_IPMI_HANDLER is not set + +# --- Enterprise / datacenter storage ---------------------------------------- +# Fibre Channel HBAs and enterprise RAID controllers +# CONFIG_SCSI_LPFC is not set +# CONFIG_SCSI_QLA_FC is not set +# CONFIG_QEDF is not set +# CONFIG_FCOE is not set +# CONFIG_LIBFC is not set +# CONFIG_MEGARAID_SAS is not set +# CONFIG_MEGARAID_LEGACY is not set +# CONFIG_MEGARAID_MAILBOX is not set +# CONFIG_FUSION is not set +# CONFIG_SCSI_HPSA is not set +# CONFIG_SCSI_SMARTPQI is not set + +# --- Ancient Ethernet NICs --------------------------------------------------- +# Vendor menus for 10/100 hardware from the 90s/00s (3Com, Tulip, NatSemi, +# NE2000-era, SiS 900, VIA Rhine/Velocity, Adaptec starfire). Modern NICs +# (Intel/Realtek/Marvell/Broadcom/Aquantia) are untouched. +# CONFIG_NET_VENDOR_3COM is not set +# CONFIG_NET_VENDOR_ADAPTEC is not set +# CONFIG_NET_TULIP is not set +# CONFIG_NET_VENDOR_NATSEMI is not set +# CONFIG_NET_VENDOR_8390 is not set +# CONFIG_NET_VENDOR_SIS is not set +# CONFIG_NET_VENDOR_VIA is not set + +# --- Specialized/legacy networking ------------------------------------------ +# CONFIG_ATM is not set +# CONFIG_RDS is not set +# CONFIG_TIPC is not set +# CONFIG_PHONET is not set +# CONFIG_IEEE802154 is not set +# CONFIG_CAN is not set +# CONFIG_NFC is not set +# CONFIG_INFINIBAND is not set + +# --- Analog/digital TV, radio, legacy webcams -------------------------------- +# UVC (USB_VIDEO_CLASS) webcams stay enabled — handhelds use them. +# CONFIG_MEDIA_ANALOG_TV_SUPPORT is not set +# CONFIG_MEDIA_DIGITAL_TV_SUPPORT is not set +# CONFIG_MEDIA_RADIO_SUPPORT is not set +# CONFIG_USB_GSPCA is not set + +# --- Ancient/cluster filesystems ---------------------------------------------- +# Desktop FS (ext4/btrfs/xfs/f2fs/exfat/ntfs3/vfat/nfs/cifs) untouched. +# CONFIG_MINIX_FS is not set +# CONFIG_UFS_FS is not set +# CONFIG_BFS_FS is not set +# CONFIG_GFS2_FS is not set +# CONFIG_OCFS2_FS is not set +# CONFIG_AFS_FS is not set + +# --- Chemical / gas / air-quality sensors ------------------------------------ +# IMU/gyro/accelerometer/pressure/temp/humidity IIO drivers are KEPT +# (handheld controllers need them); only gas/VOC/CO2/PM sensors removed. +# CONFIG_CCS811 is not set +# CONFIG_SPS30 is not set +# CONFIG_PMS7003 is not set +# CONFIG_SENSIRION_SGP30 is not set +# CONFIG_SENSIRION_SGP40 is not set +# CONFIG_SCD30_CORE is not set +# CONFIG_SCD4X is not set +# CONFIG_VZ89X is not set +# CONFIG_BME680 is not set + +# --- Staging drivers ---------------------------------------------------------- +# All OGC handheld drivers live in mainline trees (drivers/hid, +# drivers/platform/x86), not staging. +# CONFIG_STAGING is not set \ No newline at end of file diff --git a/.github/packaging/fedora/kernel.spec b/.github/packaging/fedora/kernel.spec new file mode 100644 index 00000000000000..2165fefce635a9 --- /dev/null +++ b/.github/packaging/fedora/kernel.spec @@ -0,0 +1,279 @@ +# SPDX-License-Identifier: GPL-2.0-only +# +# RPM spec for the linux-unstable-ogc kernel (linux-next based). +# +# Built by .github/workflows/build-kernel.yml ("fedora" job) which, before +# invoking rpmbuild: +# - substitutes @@KBASEVER@@ / @@KVERDOTTED@@ / @@SHA8@@ placeholders below +# - stages SOURCES/linux.tar.gz (kernel tree contents, root dir "linux/", +# no VCS data, localversion-next included) +# - stages SOURCES/config (Fedora kernel-core .config with the OGC +# kernel-packages fragments and the local +# config.fragment already applied) +# +# Derived from the OpenGamingCollective/kernel-packages fedora/kernel.spec +# (itself based on CachyOS/Nobara), trimmed down to core/modules/devel only. +# The kernel is compiled with clang (LLVM=1) exactly like the Arch packages. + +%global _default_patch_fuzz 2 + +# See https://fedoraproject.org/wiki/Changes/SetBuildFlagsBuildCheck +%if 0%{?fedora} >= 37 +%undefine _auto_set_build_flags +%endif + +%define _build_id_links none +%define _disable_source_fetch 1 +# no debuginfo generation, no brp strip/mangle of kernel binaries +%define debug_package %{nil} +%define __spec_install_post /usr/lib/rpm/brp-compress || : + +# ---- substituted by CI ------------------------------------------------------ +%define kbasever @@KBASEVER@@ +%define sha8 @@SHA8@@ +# ---------------------------------------------------------------------------- + +Version: @@KVERDOTTED@@ +Release: 1.g%{sha8}%{?dist} + +%define rpmver %{version}-%{release} +# Kernel release string (uname -r). Identical to what the localversion* files +# below produce during the build, and identical to the Arch packages built +# from the same commit: +# -unstable-ogc-g-1 +%define kverstr %{kbasever}-unstable-ogc-g%{sha8}-1 +# RPM dependency versions may contain at most ONE hyphen (V-R separator), so +# the full kernel release string cannot be used as a Provides: version. Use a +# 1:1 dotted translation for the *-uname-r provides instead. +%define kverdot %(echo "%{kverstr}" | sed -e "s/-/./g") + +Name: kernel-unstable-ogc +Summary: The linux-next kernel for the Open Gaming Collective +License: GPLv2 +URL: https://github.com/OpenGamingCollective/linux-unstable +Group: System Environment/Kernel +ExclusiveArch: x86_64 +Source0: linux.tar.gz +Source1: config + +BuildRequires: bash, coreutils, make, tar, findutils, gawk, diffutils, m4 +BuildRequires: bc, bison, flex, perl-interpreter, perl-Carp, binutils +BuildRequires: xz, zstd, kmod, python3 +BuildRequires: elfutils-libelf-devel, elfutils-devel +BuildRequires: openssl, openssl-devel +BuildRequires: dwarves, hmaccalc +# clang toolchain (like the Arch packages) +BuildRequires: clang, lld, llvm, ccache +# needed when the base config enables CONFIG_RUST. The bindgen binary +# package was renamed from rust-bindgen to bindgen in newer Fedora releases; +# accept either so this spec works across distro versions. +BuildRequires: rust, rust-src +BuildRequires: (bindgen or rust-bindgen) + +# All kernel make invocations: clang via ccache, deterministic version strings +%define kmake make CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc KBUILD_BUILD_TIMESTAMP="" + +%description +This package is a meta package that pulls in the linux-unstable-ogc kernel +(a linux-next snapshot for the Open Gaming Collective) and its matching +modules. + +%package core +Summary: The linux-unstable-ogc kernel (vmlinuz and core files) +Group: System Environment/Kernel +Provides: installonlypkg(kernel) +Provides: %{name}-core-uname-r = %{kverdot} +Requires: bash, coreutils, kmod +Requires: /usr/bin/kernel-install +Requires: %{name}-modules = %{rpmver} +Recommends: linux-firmware +%description core +This package contains the linux-unstable-ogc kernel image (vmlinuz), +System.map, the build configuration and the module symbol version file. + +%package modules +Summary: Kernel modules to match the linux-unstable-ogc core kernel +Group: System Environment/Kernel +Provides: installonlypkg(kernel-module) +Provides: %{name}-modules-uname-r = %{kverdot} +Supplements: %{name}-core = %{rpmver} +# kmod needed for depmod in %%post +Requires: kmod +%description modules +This package provides the kernel modules for the linux-unstable-ogc kernel. + +%package devel +Summary: Development files for building external modules +Group: Development/System +AutoReqProv: no +Requires: findutils, make, perl-interpreter, flex, bison +Requires: elfutils-libelf-devel, openssl-devel, gcc +Requires: clang, llvm, lld +Provides: %{name}-devel-uname-r = %{kverdot} +Enhances: akmods +Enhances: dkms +%description devel +This package provides the headers, scripts and tooling (objtool, +resolve_btfids) needed to build out-of-tree kernel modules against the +linux-unstable-ogc kernel (%{kverstr}). + +%prep +%setup -q -n linux + +cp %{SOURCE1} .config + +# The Fedora distro config references Fedora-only certificate files that do +# not exist in this tree; use the ephemeral in-tree key instead. +scripts/config --set-str SYSTEM_TRUSTED_KEYS "" +scripts/config --set-str SYSTEM_REVOCATION_KEYS "" +# CONFIG_MODULE_SIG_KEY points at the Red Hat signing cert in the distro +# config, which does not exist in this tree. Reset it to the kbuild default +# ("certs/signing_key.pem"): an ephemeral self-signed key generated during +# the build and trusted by the kernel itself. The Fedora config sets +# CONFIG_MODULE_SIG_ALL=y, and with an empty key string scripts/Makefile.modinst +# resolves sig-key to "./", making sign-file read a directory as the private +# key (SSL DECODER error) on every module. +scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + +# Deterministic / distro-agnostic build identity +scripts/config -u DEFAULT_HOSTNAME +scripts/config --set-str BUILD_SALT "%{kverstr}" + +# Kernel release suffix, same scheme as the Arch packages: +# -unstable-ogc-g-1 +echo "-unstable-ogc-g%{sha8}" > localversion.10-pkgname +echo "-1" > localversion.20-pkgrel + +# glibc >= 2.42 const-correctness vs -Werror in tools/lib/bpf (used by +# resolve_btfids when DEBUG_INFO_BTF=y); keep the build green. +if [ -f tools/lib/bpf/Makefile ]; then + sed -i 's/ -Werror -Wall/ -Wall/' tools/lib/bpf/Makefile || true +fi + +%{kmake} olddefconfig + +# Fail fast if the release string ever drifts from the spec +REL="$(make -s kernelrelease)" +if [ "$REL" != "%{kverstr}" ]; then + echo "kernelrelease '$REL' does not match spec kverstr '%{kverstr}'" >&2 + exit 1 +fi +cp .config config-linux-unstable-ogc + +%build +%{kmake} %{?_smp_mflags} all + +%install +MODDIR="%{buildroot}/lib/modules/%{kverstr}" +DEVEL="%{buildroot}%{_prefix}/src/kernels/%{kverstr}" + +mkdir -p "%{buildroot}/boot" "$MODDIR" + +echo "Installing boot image..." +ImageName="$(make -s image_name | tail -n 1)" +install -m 0644 "$ImageName" "$MODDIR/vmlinuz" +chmod 0755 "$MODDIR/vmlinuz" + +echo "Installing modules..." +# Modules are compressed by modules_install itself (CONFIG_MODULE_COMPRESS_*). +# depmod runs from %%post at install time. +%{kmake} %{?_smp_mflags} KERNELRELEASE=%{kverstr} \ + INSTALL_MOD_PATH=%{buildroot} INSTALL_MOD_STRIP=1 \ + DEPMOD=/doesnt/exist modules_install + +echo "Installing core files..." +cp System.map "$MODDIR/System.map" +cp .config "$MODDIR/config" +gzip -c9 < Module.symvers > "$MODDIR/symvers.gz" +(cd "$MODDIR" && sha512hmac vmlinuz > .vmlinuz.hmac) + +# ---- kernel-devel ----------------------------------------------------------- +echo "Preparing kernel-devel..." +rm -f "$MODDIR"/build "$MODDIR"/source +ln -s "%{_prefix}/src/kernels/%{kverstr}" "$MODDIR/build" +(cd "$MODDIR" && ln -s build source) +mkdir -p "$MODDIR"/updates "$MODDIR"/weak-updates "$DEVEL" + +find . -type f \( -name 'Makefile*' -o -name 'Kconfig*' \) -print0 \ + | xargs -0 cp --parents -t "$DEVEL" +cp -a include "$DEVEL"/ +cp -a arch/x86/include "$DEVEL"/arch/x86/ +if [ -f arch/x86/kernel/module.lds ]; then + cp -a --parents arch/x86/kernel/module.lds "$DEVEL"/ +fi +cp -a scripts "$DEVEL"/ +rm -rf "$DEVEL"/scripts/tracing +rm -f "$DEVEL"/scripts/spdxcheck.py +cp Module.symvers System.map .config "$DEVEL"/ + +mkdir -p "$DEVEL"/tools/{objtool,bpf/resolve_btfids,lib,build} +cp -a tools/objtool/objtool "$DEVEL"/tools/objtool/ || : +cp -a tools/bpf/resolve_btfids/resolve_btfids "$DEVEL"/tools/bpf/resolve_btfids/ || : +cp -a tools/include "$DEVEL"/tools/ +cp -a tools/lib/subcmd "$DEVEL"/tools/lib/ +cp -a tools/lib/bpf "$DEVEL"/tools/lib/ +cp -a tools/build/Build.include tools/build/fixdep.c "$DEVEL"/tools/build/ +cp -a tools/scripts/utilities.mak "$DEVEL"/tools/scripts/ 2>/dev/null || : +cp -a --parents arch/x86/entry/syscalls/syscall_32.tbl "$DEVEL"/ +cp -a --parents arch/x86/entry/syscalls/syscall_64.tbl "$DEVEL"/ +cp -a arch/x86/tools "$DEVEL"/arch/x86/ + +# Drop intermediate build artifacts from devel tree +find "$DEVEL" \( -name '*.o' -o -name '*.cmd' -o -name '.*.cmd' \) -delete + +# Timestamps must line up so external module builds do not rerun kconfig +touch -r "$DEVEL"/Makefile \ + "$DEVEL"/include/generated/uapi/linux/version.h \ + "$DEVEL"/include/config/auto.conf + +%post core +# nothing to do at this point + +%posttrans core +# Runs after ALL packages of this transaction have been installed and their +# %%post scriptlets (incl. depmod from -modules) have run. +if [ -x /usr/bin/kernel-install ]; then + /usr/bin/kernel-install add %{kverstr} /lib/modules/%{kverstr}/vmlinuz || exit $? +fi +if [ -x /usr/sbin/grubby ]; then + grubby --set-default="/boot/vmlinuz-%{kverstr}" || : +fi + +%preun core +if [ "$1" = "0" ] && [ -x /usr/bin/kernel-install ]; then + /usr/bin/kernel-install remove %{kverstr} /lib/modules/%{kverstr}/vmlinuz || exit $? +fi + +%post modules +/sbin/depmod -a %{kverstr} + +%files +# meta package: everything lives in the subpackages + +%files core +%ghost /boot/vmlinuz-%{kverstr} +%ghost /boot/initramfs-%{kverstr}.img +/lib/modules/%{kverstr}/vmlinuz +/lib/modules/%{kverstr}/.vmlinuz.hmac +/lib/modules/%{kverstr}/System.map +/lib/modules/%{kverstr}/config +/lib/modules/%{kverstr}/symvers.gz + +%files modules +/lib/modules/%{kverstr}/ +%exclude /lib/modules/%{kverstr}/vmlinuz +%exclude /lib/modules/%{kverstr}/.vmlinuz.hmac +%exclude /lib/modules/%{kverstr}/System.map +%exclude /lib/modules/%{kverstr}/config +%exclude /lib/modules/%{kverstr}/symvers.gz +%exclude /lib/modules/%{kverstr}/build +%exclude /lib/modules/%{kverstr}/source + +%files devel +/usr/src/kernels/%{kverstr} +/lib/modules/%{kverstr}/build +/lib/modules/%{kverstr}/source + +%changelog +* Thu Aug 20 2026 OpenGamingCollective CI +- Initial linux-unstable-ogc spec, generated by CI from linux-next. \ No newline at end of file diff --git a/.github/packaging/merge-fragments.sh b/.github/packaging/merge-fragments.sh new file mode 100644 index 00000000000000..e8525b2d86ce8e --- /dev/null +++ b/.github/packaging/merge-fragments.sh @@ -0,0 +1,62 @@ +#!/usr/bin/env bash +# Textually merge kernel config fragments into a base .config file. +# +# Usage: merge-fragments.sh [...] +# +# Recognized fragment line formats (fragments are applied in order, later +# fragments win): +# CONFIG_X=value set X to value +# CONFIG_X unset X (kernel-configurator *.unset format) +# "# CONFIG_X is not set" unset X (kconfig fragment format) +# Blank lines and other comments are ignored. +# +# This replicates the semantics of the OpenGamingCollective +# kernel-configurator action (*.config.set / *.config.unset fragments) and +# additionally accepts regular kconfig fragment files, so both the OGC +# fragments and the repo-local config.fragment can be handled uniformly. +# The final `make olddefconfig` (run by the PKGBUILD / RPM spec) turns the +# textual result into a consistent kconfig. + +set -euo pipefail + +if [ "$#" -lt 2 ]; then + echo "usage: $0 ..." >&2 + exit 2 +fi + +CFG="$(realpath "$1")" +shift + +set_key() { + local key="$1" value="$2" + if grep -q "^${key}=" "$CFG"; then + sed -i "s|^${key}=.*|${key}=${value}|" "$CFG" + elif grep -q "^# ${key} is not set$" "$CFG"; then + sed -i "s|^# ${key} is not set\$|${key}=${value}|" "$CFG" + else + printf '%s=%s\n' "${key}" "${value}" >> "$CFG" + fi +} + +unset_key() { + local key="$1" + if grep -q "^${key}=" "$CFG"; then + sed -i "s|^${key}=.*|# ${key} is not set|" "$CFG" + fi +} + +for frag in "$@"; do + frag="$(realpath "$frag")" + while IFS= read -r line || [ -n "$line" ]; do + line="${line#"${line%%[![:space:]]*}"}" + line="${line%"${line##*[![:space:]]}"}" + [ -z "$line" ] && continue + if [[ "$line" =~ ^#\ (CONFIG_[A-Za-z0-9_]+)\ is\ not\ set$ ]]; then + unset_key "${BASH_REMATCH[1]}" + elif [[ "$line" =~ ^(CONFIG_[A-Za-z0-9_]+)=(.*)$ ]]; then + set_key "${BASH_REMATCH[1]}" "${BASH_REMATCH[2]}" + elif [[ "$line" =~ ^(CONFIG_[A-Za-z0-9_]+)$ ]]; then + unset_key "${BASH_REMATCH[1]}" + fi + done < "$frag" +done \ No newline at end of file diff --git a/.github/workflows/build-kernel.yml b/.github/workflows/build-kernel.yml new file mode 100644 index 00000000000000..623cd5e89c2561 --- /dev/null +++ b/.github/workflows/build-kernel.yml @@ -0,0 +1,721 @@ +name: Build & release linux-unstable-ogc + +on: + push: + branches: [master] + workflow_dispatch: {} + +permissions: + contents: write + +concurrency: + group: kernel-build-${{ github.ref }} + cancel-in-progress: true + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # Distro base configs are extracted from the official distro packages + # (Arch: linux-headers, Fedora: kernel-core) — same approach as the + # kernel-packages arch.yaml / fedora.yaml workflows. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + prepare: + name: Prepare sources and version + runs-on: ubuntu-latest + outputs: + kver: ${{ steps.kernel.outputs.kver }} + next: ${{ steps.kernel.outputs.next }} + sha8: ${{ steps.kernel.outputs.sha8 }} + krel: ${{ steps.kernel.outputs.krel }} + tag: ${{ steps.kernel.outputs.tag }} + kbase: ${{ steps.kernel.outputs.kbase }} + kverdot: ${{ steps.kernel.outputs.kverdot }} + steps: + - name: Checkout kernel sources + uses: actions/checkout@v7 + with: + fetch-depth: 1 + + - name: Compute kernel version + id: kernel + run: | + set -euxo pipefail + KVER="$(make -s kernelversion)" + NEXT="$(cat localversion-next 2>/dev/null || true)" + SHA8="$(git rev-parse --short=8 HEAD)" + KREL="${KVER}${NEXT}-unstable-ogc-g${SHA8}-1" + TAG="v${KVER}${NEXT}-g${SHA8}" + # RPM Version: field must not contain hyphens + KVERDOT="$(printf '%s%s' "${KVER}" "${NEXT}" | tr '-' '.')" + echo "kver=${KVER}" >> "$GITHUB_OUTPUT" + echo "next=${NEXT}" >> "$GITHUB_OUTPUT" + echo "sha8=${SHA8}" >> "$GITHUB_OUTPUT" + echo "krel=${KREL}" >> "$GITHUB_OUTPUT" + echo "tag=${TAG}" >> "$GITHUB_OUTPUT" + echo "kbase=${KVER}${NEXT}" >> "$GITHUB_OUTPUT" + echo "kverdot=${KVERDOT}" >> "$GITHUB_OUTPUT" + echo "Kernel release: ${KREL}" + echo "Release tag: ${TAG}" + + - name: Stage sources and packaging files + run: | + set -euxo pipefail + # Commit marker consumed by the Arch PKGBUILD (same trick as + # linux-bisector). Written before the tarball is created so it is + # included in it. + git rev-parse --short=8 HEAD > .build_commit + + STAGE="/tmp/stage" + mkdir -p "${STAGE}" + cp .github/packaging/PKGBUILD "${STAGE}/PKGBUILD" + cp .github/packaging/merge-fragments.sh "${STAGE}/merge-fragments.sh" + cp .github/packaging/config.fragment "${STAGE}/config.fragment" + cp .github/packaging/fedora/kernel.spec "${STAGE}/kernel.spec" + + # Source tarball: contents of the repo as ./linux/, without VCS data. + tar --exclude-vcs -I 'gzip -1' -cf "${STAGE}/linux.tar.gz" \ + --transform 's|^\./|linux/|' -C "$GITHUB_WORKSPACE" . + ls -lh "${STAGE}" + + - name: Upload staged sources + uses: actions/upload-artifact@v7 + with: + name: kernel-sources + path: /tmp/stage + retention-days: 3 + + arch: + name: Build Arch Linux packages + needs: prepare + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/archlinux:base-devel + env: + PKGDIR: /tmp/pkgbuild + DISTDIR: /tmp/dist + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ needs.prepare.outputs.krel }} + steps: + - name: Show disk space + run: df -h / + + - name: Bootstrap Arch Linux build environment + run: | + set -euxo pipefail + # Refresh keyring first to avoid signature failures on stale images + pacman -Sy --needed --noconfirm archlinux-keyring + pacman -Su --needed --noconfirm + # Everything makepkg needs (mirrors the makedepends of the PKGBUILD). + # rust + rust-bindgen are needed because the Arch config enables + # CONFIG_RUST=y. + pacman -S --needed --noconfirm \ + bc cpio gettext libelf pahole perl python tar xz zstd \ + gcc clang llvm lld ccache pigz file curl git \ + rust rust-bindgen + # makepkg refuses to run as root + useradd -m build + install -d -o build -g build -m 0777 /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-arch-${{ github.sha }} + restore-keys: | + ccache-arch- + + - name: Prepare ccache for the build user + run: chown -R build:build /ccache && su build -c 'ccache -s' || true + + - name: Assemble Arch kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: extract the official Arch Linux kernel .config from + # the distro's `linux-headers` package (same method as the OGC + # kernel-packages arch.yaml workflow). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + pacman -Sw --noconfirm --cachedir "$CACHE" linux-headers + PKG="$(find "$CACHE" -maxdepth 1 -name 'linux-headers-*.pkg.tar.zst' -type f)" + if [ -z "$PKG" ]; then + echo "::error::linux-headers package not found in cache"; exit 1 + fi + CFG="$(tar --zstd -tf "$PKG" | grep -E '^usr/lib/modules/[^/]+/build/\.config$' || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel .config not found inside $PKG"; exit 1 + fi + tar --zstd -xOf "$PKG" "$CFG" > config + test -s config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # Layer the OGC config fragments from kernel-packages on top, then + # the repo-local config.fragment (which wins). Order is critical: + # unsets apply first (removing things we explicitly don't want), + # then sets apply (enabling things we do want), so explicit enables + # can override explicit disables. + # ------------------------------------------------------------------ + for f in arch.config.set ogc.config.set arch.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + arch.config.unset ogc.config.unset \ + arch.config.set ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Stage PKGBUILD directory + run: | + set -euxo pipefail + mkdir -p "${PKGDIR}" "${DISTDIR}" + cd /tmp/stage + cp PKGBUILD config.fragment linux.tar.gz config "${PKGDIR}/" + chown -R build:build "${PKGDIR}" + + - name: Build packages with makepkg + run: | + set -euxo pipefail + cd "${PKGDIR}" + runuser -u build -- env HOME=/home/build CCACHE_DIR=/ccache \ + makepkg -f --noconfirm --noprogressbar + + ls -lh ./*.pkg.tar.zst + cp -v ./*.pkg.tar.zst "${DISTDIR}/" + # Ship the exact .config used for the build as well + cp -v "src/linux/.config" "${DISTDIR}/config-arch-${KREL}" + + - name: Clean build tree (free disk before cache save) + run: rm -rf "${PKGDIR}/src" "${PKGDIR}/pkg" ; df -h / + + - name: Upload Arch packages + uses: actions/upload-artifact@v7 + with: + name: arch-packages + path: /tmp/dist + retention-days: 3 + + fedora: + name: Build Fedora RPM packages + needs: prepare + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/library/fedora:43 + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ needs.prepare.outputs.krel }} + KBASE: ${{ needs.prepare.outputs.kbase }} + KVERDOT: ${{ needs.prepare.outputs.kverdot }} + SHA8: ${{ needs.prepare.outputs.sha8 }} + steps: + - name: Show disk space + run: df -h / + + - name: Install build tools + run: | + set -euxo pipefail + dnf -y install dnf5-plugins rpm-build + dnf -y install cpio curl findutils tar gzip + mkdir -p /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-fedora-${{ github.sha }} + restore-keys: | + ccache-fedora- + + - name: Assemble Fedora kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: extract the official Fedora kernel .config from the + # distro's `kernel-core` package (same method as the OGC + # kernel-packages fedora.yaml workflow). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + dnf download --destdir "$CACHE" kernel-core + RPM="$(find "$CACHE" -maxdepth 1 -name 'kernel-core-*.rpm' -type f)" + if [ -z "$RPM" ]; then + echo "::error::kernel-core package not found in cache"; exit 1 + fi + CFG="$(rpm -qlp "$RPM" | grep -E '^/lib/modules/[^/]+/config$' | head -n1 || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel config not found inside $RPM"; exit 1 + fi + rpm2cpio "$RPM" | cpio -i --to-stdout ".${CFG}" > config + test -s config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # Layer the OGC config fragments on top, then the repo-local + # config.fragment (which wins). Order is critical: unsets apply + # first (removing things we explicitly don't want), then sets apply + # (enabling things we do want), so explicit enables can override + # explicit disables. + # ------------------------------------------------------------------ + for f in fedora.config.set ogc.config.set fedora.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + fedora.config.unset ogc.config.unset \ + fedora.config.set ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Finalize spec and install build dependencies + working-directory: /tmp/stage + run: | + set -euxo pipefail + # Substitute the CI placeholders (must happen before `dnf builddep` + # so RPM can parse Version:/Release:). + sed -i \ + -e "s/@@KBASEVER@@/${KBASE}/" \ + -e "s/@@KVERDOTTED@@/${KVERDOT}/" \ + -e "s/@@SHA8@@/${SHA8}/" \ + kernel.spec + grep -n '^Version:\|^Release:\|%define kbasever\|%define sha8' kernel.spec + ! grep -q '@@' kernel.spec + + # Installs the toolchain from the spec BuildRequires, including the + # distro dwarves/pahole. Run BEFORE the pahole check below so a + # source-built pahole is not overwritten by builddep. + dnf -y builddep kernel.spec + + - name: Ensure pahole >= 1.31 (sched_ext/BTF) + working-directory: /tmp + run: | + set -euxo pipefail + VER="$(pahole --version | tr -d 'v')" + MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" + echo "Installed pahole: ${VER}" + if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then + echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" + dnf -y install cmake make gcc elfutils-devel zlib-devel + curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz + echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c + tar -xf dwarves-1.31.tar.xz + cmake -B dwarves-1.31/build -S dwarves-1.31 \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -D__LIB=lib + make -C dwarves-1.31/build -j"$(nproc)" install + fi + pahole --version + + - name: Build RPMs with rpmbuild + working-directory: /tmp + run: | + set -euxo pipefail + TOPDIR=/tmp/rpmbuild + mkdir -p "${TOPDIR}"/{BUILD,BUILDROOT,RPMS,SOURCES,SPECS,SRPMS} + cp /tmp/stage/linux.tar.gz "${TOPDIR}/SOURCES/" + cp /tmp/stage/config "${TOPDIR}/SOURCES/config" + cp /tmp/stage/kernel.spec "${TOPDIR}/SPECS/kernel.spec" + + rpmbuild --define "_topdir ${TOPDIR}" -ba "${TOPDIR}/SPECS/kernel.spec" + + ls -lh "${TOPDIR}"/RPMS/x86_64/ + + - name: Collect Fedora artifacts + run: | + set -euxo pipefail + DISTDIR=/tmp/dist + mkdir -p "${DISTDIR}" + cp -v /tmp/rpmbuild/RPMS/x86_64/*.rpm "${DISTDIR}/" + cp -v /tmp/stage/config "${DISTDIR}/config-fedora-${KREL}" + ls -lh "${DISTDIR}" + + - name: Clean build tree (free disk before cache save) + run: rm -rf /tmp/rpmbuild/BUILD /tmp/rpmbuild/BUILDROOT ; df -h / + + - name: Upload Fedora packages + uses: actions/upload-artifact@v7 + with: + name: fedora-packages + path: /tmp/dist + retention-days: 3 + + debian: + name: Build Debian packages + needs: prepare + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/library/debian:trixie + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ needs.prepare.outputs.krel }} + KVERDOT: ${{ needs.prepare.outputs.kverdot }} + SHA8: ${{ needs.prepare.outputs.sha8 }} + steps: + - name: Show disk space + run: df -h / + + - name: Install build tools + run: | + set -euxo pipefail + apt-get update + # Satisfies the Build-Depends generated by scripts/package/mkdebian + # (debhelper-compat, bc, bison, flex, kmod, libdw/libelf/libssl-dev, + # python3, rsync) plus the clang toolchain used by all OGC builds. + apt-get install -y --no-install-recommends \ + build-essential debhelper rsync \ + bc bison flex python3 kmod \ + libelf-dev libdw-dev libssl-dev zlib1g-dev \ + dwarves zstd xz-utils \ + clang llvm lld ccache curl ca-certificates git \ + openssl + # dpkg-buildpackage refuses to run as root + useradd -m builder + install -d -o builder -g builder -m 0777 /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-debian-${{ github.sha }} + restore-keys: | + ccache-debian- + + - name: Assemble Debian kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: the official Debian kernel configuration. It ships in + # the small linux-config- package; fall back to /boot/config-* + # of the linux-image- package if that is unavailable. + # ------------------------------------------------------------------ + ABI="$(apt-cache depends linux-image-amd64 \ + | awk '/^ *Depends: *linux-image-[0-9]/ {print $2; exit}' \ + | sed 's/^linux-image-//')" + test -n "${ABI}" + echo "Debian kernel ABI: ${ABI}" + # linux-config is versioned by major.minor only (e.g. 6.12) and + # stores the per-flavour config xz-compressed, e.g. + # /usr/src/linux-config-6.12/config.amd64_none_amd64.xz + KMAJMIN="$(printf '%s' "${ABI}" | cut -d. -f1,2)" + mkdir -p /tmp/pkgcfg + if apt-get download "linux-config-${KMAJMIN}"; then + dpkg-deb -x linux-config-"${KMAJMIN}"_*.deb /tmp/pkgcfg + BASE="$(find /tmp/pkgcfg/usr/src -name 'config.amd64_none_amd64.xz' | head -n1)" + test -s "${BASE}" + xz -dc "${BASE}" > config + else + apt-get download "linux-image-${ABI}" + dpkg-deb -x linux-image-"${ABI}"_*.deb /tmp/pkgcfg + BASE="$(find /tmp/pkgcfg/boot -maxdepth 1 -name 'config-*' | head -n1)" + test -s "${BASE}" + cp "${BASE}" config + fi + test -s config + rm -rf /tmp/pkgcfg + + # ------------------------------------------------------------------ + # Layer the OGC config fragments and the repo-local config.fragment + # (which wins). kernel-packages ships no debian-specific fragments. + # Order is critical: unsets apply first, then sets, so explicit + # enables can override explicit disables. + # ------------------------------------------------------------------ + for f in ogc.config.set ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + ogc.config.unset \ + ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Ensure pahole >= 1.31 (sched_ext/BTF) + run: | + set -euxo pipefail + VER="$(pahole --version | tr -d 'v')" + MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" + echo "Installed pahole: ${VER}" + if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then + echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" + apt-get install -y --no-install-recommends cmake pkg-config + cd /tmp + curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz + echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c + tar -xf dwarves-1.31.tar.xz + cmake -B dwarves-1.31/build -S dwarves-1.31 \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr + make -C dwarves-1.31/build -j"$(nproc)" install + fi + pahole --version + + - name: Build .deb packages with the in-tree packaging + run: | + set -euxo pipefail + mkdir -p /build + tar -xzf /tmp/stage/linux.tar.gz -C /build + cd /build/linux + + # Config adjustments + kernel release suffix, identical scheme to + # the Arch and Fedora packages (uname -r == ${KREL}). + cp /tmp/stage/config .config + # Debian's config references distro-only certificate files that do + # not exist in this tree; use the ephemeral in-tree key instead. + scripts/config --set-str SYSTEM_TRUSTED_KEYS "" + scripts/config --set-str SYSTEM_REVOCATION_KEYS "" + # CONFIG_MODULE_SIG_KEY points at the Debian signing cert, which + # does not exist in this tree. Reset it to the kbuild default + # ("certs/signing_key.pem"): an ephemeral self-signed key generated + # during the build and trusted by the kernel itself. The Debian + # config sets CONFIG_MODULE_SIG_ALL=y, and with an empty key string + # scripts/Makefile.modinst resolves sig-key to "./", making + # sign-file read a directory as the private key (SSL DECODER error) + # on every module. + scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + scripts/config -u DEFAULT_HOSTNAME + scripts/config --set-str BUILD_SALT "${KREL}" + echo "-unstable-ogc-g${SHA8}" > localversion.10-pkgname + echo "-1" > localversion.20-pkgrel + # The staged tarball carries no VCS data, but mkdebian (via + # gen-diff-patch) runs "git diff HEAD" and aborts on failure. + # A throwaway repo with everything committed makes that a no-op. + git init -q + git config user.email "ci@opengamingcollective.org" + git config user.name "OpenGamingCollective CI" + git add -A + git commit -qm "linux-unstable-ogc ${KREL}" + chown -R builder:builder /build + + # scripts/setlocalversion appends a "+" whenever a git repo exists + # and LOCALVERSION is unset and the HEAD is not at an annotated + # version tag. The throwaway repo below would therefore corrupt the + # release string. Setting LOCALVERSION to the empty string (as + # documented in the script) suppresses that suffix everywhere. + runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ + LOCALVERSION= \ + make CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 \ + KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc \ + olddefconfig + + # Fail fast if the release string ever drifts from the other distros + REL="$(runuser -u builder -- env HOME=/home/builder LOCALVERSION= make -s kernelrelease)" + if [ "$REL" != "${KREL}" ]; then + echo "kernelrelease '$REL' does not match expected '${KREL}'" >&2 + exit 1 + fi + + # Generate the debian/ directory with the tree's own packaging. + # mkdebian is invoked directly as a script: a bare "make debian" is + # NOT a top-level make target in this tree (only *-pkg patterns are + # delegated to scripts/Makefile.package), and going through make + # without the exact same CC flags as the olddefconfig above made + # kbuild re-sync the config interactively. As a plain sh script it + # touches nothing kbuild-related. + # KDEB_PKGVERSION must not contain hyphens except the final revision + # separator (Debian policy), hence the dotted translation. + KDEBVER="${KVERDOT}.unstable.ogc.g${SHA8}-1" + runuser -u builder -- env HOME=/home/builder \ + srctree="$PWD" \ + ARCH=x86_64 SRCARCH=x86 UTS_MACHINE=x86_64 \ + KERNELRELEASE="${KREL}" \ + KCONFIG_CONFIG=.config \ + KDEB_SOURCENAME=linux-unstable-ogc \ + KDEB_PKGVERSION="${KDEBVER}" \ + KDEB_CHANGELOG_DIST=trixie \ + DEBFULLNAME="OpenGamingCollective CI" \ + DEBEMAIL="ci@opengamingcollective.org" \ + sh scripts/package/mkdebian + echo "debian arch: $(cat debian/arch)" + + # Kbuild only honours command-line variables, so inject the + # clang/ccache toolchain into the generated debian/rules (this is + # what "make bindeb-pkg" would otherwise lose). LOCALVERSION= (set, + # empty) keeps setlocalversion from appending "+" inside the build. + sed -i 's|^make-opts = |make-opts = CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 LOCALVERSION= KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc |' debian/rules + grep -n '^make-opts' debian/rules + + # Same invocation as "make bindeb-pkg" (scripts/Makefile.package), + # but with parallel jobs, --no-check-builddeps (the build deps are + # preinstalled above; the generated Build-Depends-Arch also names a + # distro-only cross-gcc package that need not exist), and without + # the multi-GB debug-symbol package (all other distro jobs ship no + # debug packages either). + runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ + LOCALVERSION= \ + DEB_BUILD_PROFILES="pkg.linux-unstable-ogc.nokerneldbg" \ + dpkg-buildpackage --build=binary --no-pre-clean --unsigned-changes \ + --no-check-builddeps \ + -R'make -f debian/rules' -j"$(nproc)" -a"$(cat debian/arch)" + + ls -lh /build/*.deb + + - name: Collect Debian artifacts + run: | + set -euxo pipefail + DISTDIR=/tmp/dist + mkdir -p "${DISTDIR}" + cp -v "/build/linux-image-${KREL}"_*.deb "${DISTDIR}/" + cp -v "/build/linux-headers-${KREL}"_*.deb "${DISTDIR}/" + cp -v /build/linux/.config "${DISTDIR}/config-debian-${KREL}" + # linux-libc-dev is intentionally NOT shipped: installing it would + # replace the distribution's own linux-libc-dev package. + ls -lh "${DISTDIR}" + + - name: Clean build tree (free disk before cache save) + run: rm -rf /build ; df -h / + + - name: Upload Debian packages + uses: actions/upload-artifact@v7 + with: + name: debian-packages + path: /tmp/dist + retention-days: 3 + + release: + name: Create GitHub release + needs: [prepare, arch, fedora, debian] + runs-on: ubuntu-latest + env: + TAG: ${{ needs.prepare.outputs.tag }} + KREL: ${{ needs.prepare.outputs.krel }} + KVER: ${{ needs.prepare.outputs.kver }} + NEXT: ${{ needs.prepare.outputs.next }} + steps: + - name: Download package artifacts + uses: actions/download-artifact@v8 + with: + path: /tmp/dist + pattern: "*-packages" + + - name: Flatten artifact directory + run: | + set -euxo pipefail + cd /tmp/dist + find . -mindepth 2 -maxdepth 2 -type f -exec mv -t . {} + + find . -mindepth 1 -type d -delete + ls -lh + + - name: Generate checksums + run: | + set -euxo pipefail + cd /tmp/dist + sha256sum * > SHA256SUMS + cat SHA256SUMS + + - name: Create GitHub release and upload packages + env: + GH_TOKEN: ${{ github.token }} + run: | + set -euxo pipefail + + # Idempotency: if a stale release exists for this tag (e.g. re-run + # of the same commit), remove it so this run can recreate it. + if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" 2>/dev/null; then + echo "Deleting existing release $TAG, will recreate it" + gh release delete "$TAG" --repo "$GITHUB_REPOSITORY" --yes --cleanup-tag + fi + # In case a tag without a release is left over, drop it too. + gh api -X DELETE "repos/${GITHUB_REPOSITORY}/git/refs/tags/${TAG}" >/dev/null 2>&1 || true + + NOTES="$(mktemp)" + { + echo "Automated build of linux-unstable-ogc." + echo + echo "- Source commit: https://github.com/${GITHUB_REPOSITORY}/commit/${GITHUB_SHA}" + echo "- Kernel release: ${KREL}" + echo "- Upstream version: ${KVER}${NEXT}" + echo "- Base configs: Arch \`linux-headers\` + Fedora \`kernel-core\`, plus [OGC kernel-packages fragments](${KERNEL_PACKAGES_RAW}/config)" + echo "- Compiler: clang / LLVM=1 (with ccache)" + echo + echo "### Artifacts" + echo + echo "#### Arch Linux" + echo + echo '- `linux-unstable-ogc` — kernel image and modules' + echo '- `linux-unstable-ogc-headers` — headers for building external modules' + echo "- \`config-arch-${KREL}\` — the exact .config used for this build" + echo + echo "#### Fedora" + echo + echo '- `kernel-unstable-ogc-core` — kernel image (vmlinuz) and core files' + echo '- `kernel-unstable-ogc-modules` — kernel modules' + echo '- `kernel-unstable-ogc-devel` — headers for building external modules' + echo "- \`config-fedora-${KREL}\` — the exact .config used for this build" + echo + echo "#### Debian (and derivatives)" + echo + echo "- \`linux-image-${KREL}\` — kernel image and modules" + echo "- \`linux-headers-${KREL}\` — headers for building external modules" + echo "- \`config-debian-${KREL}\` — the exact .config used for this build" + echo + echo '- `SHA256SUMS` — checksums of all artifacts' + echo + echo "### Install" + echo + echo "Arch Linux:" + echo + echo '```sh' + echo 'sudo pacman -U linux-unstable-ogc-headers-*.pkg.tar.zst linux-unstable-ogc-*.pkg.tar.zst' + echo '```' + echo + echo "Fedora:" + echo + echo '```sh' + echo 'sudo dnf install ./kernel-unstable-ogc-core-*.rpm ./kernel-unstable-ogc-modules-*.rpm' + echo '```' + echo + echo "Debian and derivatives:" + echo + echo '```sh' + echo 'sudo apt install ./linux-image-*.deb ./linux-headers-*.deb' + echo '```' + echo + echo "> The initramfs is generated automatically on install (mkinitcpio hooks" + echo "> on Arch, kernel-install/dracut on Fedora, initramfs-tools hooks on Debian)." + } > "$NOTES" + + cd /tmp/dist + # Upload every artifact produced by the distro jobs (package files + # plus the config-* files). SHA256SUMS itself is included by ./*. + gh release create "$TAG" \ + --repo "$GITHUB_REPOSITORY" \ + --target "$GITHUB_SHA" \ + --title "linux-unstable-ogc ${KREL}" \ + --notes-file "$NOTES" \ + ./* + + - name: Summary + run: | + { + echo "## linux-unstable-ogc build" + echo + echo "- Kernel release: \`${KREL}\`" + echo "- Release: https://github.com/${GITHUB_REPOSITORY}/releases/tag/${TAG}" + } >> "$GITHUB_STEP_SUMMARY" \ No newline at end of file diff --git a/.github/workflows/sync-linux-next.yml b/.github/workflows/sync-linux-next.yml new file mode 100644 index 00000000000000..a83fec1d5c146e --- /dev/null +++ b/.github/workflows/sync-linux-next.yml @@ -0,0 +1,137 @@ +name: Sync linux-next -> master (replay fork commits) + +on: + schedule: + - cron: "0 2 * * *" # every day at 02:00 UTC + workflow_dispatch: {} + +permissions: + contents: write + +concurrency: + group: sync-linux-next + cancel-in-progress: false + +jobs: + sync: + runs-on: ubuntu-latest + steps: + - name: Checkout fork + uses: actions/checkout@v7 + with: + ref: master + fetch-depth: 0 + + - name: Configure upstream + run: | + git remote remove upstream || true + git remote add upstream https://git.kernel.org/pub/scm/linux/kernel/git/next/linux-next.git + git fetch --no-tags upstream --prune + + - name: Replay fork commits on top of upstream/master + run: | + set -euo pipefail + + git checkout master + + # REQUIRED for cherry-pick commit creation (runner has no identity by default) + git config user.name "github-actions[bot]" + git config user.email "github-actions[bot]@users.noreply.github.com" + + UP_BASE="upstream/master" + git show -s --oneline "$UP_BASE" >/dev/null + + # ------------------------------------------------------------------ + # Find the upstream snapshot that master's fork commits sit on. + # + # linux-next is rebuilt from scratch every day: commits merged into + # yesterday's tree are regularly dropped from today's history + # (subsystem trees get rebased), so: + # - "upstream/master..master" contains hundreds of unrelated + # upstream commits, not just our fork commits, and + # - a hardcoded marker hash gets orphaned by the very replay this + # workflow performs (run #2 then dies with "unknown revision", + # which is how this workflow broke in the first place). + # + # Anchor: walk first-parent history from master and stop at the + # second consecutive commit whose subject does not start with + # "ogc:" (the reserved prefix of every fork commit; cherry-pick + # preserves subjects, so this survives replays and force-pushes). + # The first of those two commits is the upstream snapshot the fork + # was built on. + # ------------------------------------------------------------------ + BASE="" + NONOGC="" + while read -r c; do + case "$(git show -s --format=%s "$c")" in + ogc:*) NONOGC="" ;; + *) + if [ -n "$NONOGC" ]; then BASE="$NONOGC"; break; fi + NONOGC="$c" + ;; + esac + done < <(git rev-list --first-parent master) + + if [ -z "$BASE" ]; then + echo "::error::Could not determine the upstream base of the fork; refusing to touch master." + exit 1 + fi + echo "Upstream base: $(git show -s --oneline "$BASE")" + + # If master already sits on the current upstream tip there is + # nothing to sync: replaying would only churn the fork commits' + # hashes for identical content. + if [ "$(git rev-parse "$BASE")" = "$(git rev-parse "$UP_BASE")" ]; then + echo "master is already based on the current linux-next tip; nothing to do." + exit 0 + fi + + # Everything on master on top of the upstream base = the fork + # commits (including non-"ogc:"-prefixed ones, e.g. squash-merged + # PRs). --no-merges matches the cherry-pick loop below and makes + # merge commits replay as their constituent commits. + MY_COMMITS="$(git rev-list --reverse --no-merges "${BASE}..master")" + if [ -z "${MY_COMMITS// }" ]; then + echo "::error::No commits found on top of the upstream base; refusing to reset master (that would delete the fork)." + exit 1 + fi + + # Safety net: a mis-detected base would replay days of upstream + # history. The fork accumulates its own commits over time (and PR + # squash-merges add to that), so allow a generous 150; anything + # beyond that is far more likely a mis-detected base than real fork + # commits, so bail out and ask for a manual look instead of + # corrupting master. + N="$(echo "$MY_COMMITS" | wc -l)" + if [ "$N" -gt 150 ]; then + echo "::error::Refusing to replay ${N} commits (expected only fork commits). Check master's history (all fork commits must carry the 'ogc:' subject prefix)." + exit 1 + fi + + echo "Replaying ${N} commit(s):" + for c in $MY_COMMITS; do + echo " $c $(git show -s --format=%s "$c")" + done + echo + + # Reset master to the new upstream tip, then replay the fork commits. + git reset --hard "$UP_BASE" + + for c in $MY_COMMITS; do + echo "Cherry-picking: $c $(git show -s --format=%s "$c")" + + if ! git cherry-pick -x --allow-empty "$c"; then + git status || true + # If we're left mid-cherry-pick, abort to avoid a broken + # working state. Nothing was pushed, so master on origin is + # still intact. + if [ -f .git/CHERRY_PICK_HEAD ]; then + git cherry-pick --abort || true + fi + echo "::error::Cherry-pick failed for $c; nothing was pushed." + exit 1 + fi + done + + echo "Done. Pushing updated master." + git push --force-with-lease origin master \ No newline at end of file diff --git a/.github/workflows/test-pr.yml b/.github/workflows/test-pr.yml new file mode 100644 index 00000000000000..29c4e2fb8f2dbe --- /dev/null +++ b/.github/workflows/test-pr.yml @@ -0,0 +1,227 @@ +name: Test PR (checkpatch + config gate + gcc build) + +on: + pull_request: + branches: [master] + +permissions: + contents: read + +concurrency: + group: pr-test-${{ github.event.pull_request.number }} + cancel-in-progress: true + +env: + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + checks: + name: Static checks (checkpatch, new-driver symbols) + runs-on: ubuntu-latest + outputs: + new_symbols: ${{ steps.syms.outputs.syms }} + steps: + - name: Checkout PR merge result + uses: actions/checkout@v7 + with: + # Default ref for pull_request is the merge commit refs/pull/N/merge. + # depth 2 fetches the merge commit AND its parents, so HEAD^1 + # (current master tip) is present and the PR diff is exactly + # "what landing this PR changes on master". + fetch-depth: 2 + + - name: Compute PR patch and new driver symbols + id: syms + run: | + set -euo pipefail + # For pull_request, checkout gets the merge commit refs/pull/N/merge: + # parent 1 = master tip, parent 2 = PR head. Fall back to + # origin/master if HEAD is not a merge commit for any reason. + BASE="$(git rev-parse --verify -q HEAD^1 || true)" + if [ -z "$BASE" ]; then + echo "HEAD^1 unresolvable; falling back to origin/master" + git fetch --depth=1 origin master + BASE="$(git rev-parse --verify -q FETCH_HEAD || true)" + fi + if [ -z "$BASE" ]; then + echo "::error::Could not determine the base commit for the PR diff." + exit 1 + fi + git diff "$BASE" HEAD > /tmp/pr.patch + + if [ ! -s /tmp/pr.patch ]; then + echo "Empty patch; nothing to check." + echo "syms=" >> "$GITHUB_OUTPUT" + exit 0 + fi + + # Selectable symbols introduced under drivers/ (added + # "config FOO" / "menuconfig FOO" lines in any Kconfig file). + # NOTE: grep exits 1 on zero matches; under `set -euo pipefail` + # that would abort the step, so it is neutralised here only + # (git diff / awk failures still propagate). + SYMS="$(git diff "$BASE" HEAD -- drivers/ \ + | { grep -E '^\+[[:space:]]*(menu)?config[[:space:]]+[A-Z0-9_]+' || true; } \ + | awk '{print $NF}' | sort -u | tr '\n' ' ')" + echo "syms=${SYMS}" >> "$GITHUB_OUTPUT" + { + echo "### PR checks" + echo + echo "- Changed files: $(git diff --name-only "$BASE" HEAD | wc -l)" + echo "- New driver CONFIG symbols: ${SYMS:-none}" + } >> "$GITHUB_STEP_SUMMARY" + + - name: checkpatch (fail on errors) + run: | + set -uo pipefail + if [ ! -s /tmp/pr.patch ]; then + echo "Empty patch; skipping checkpatch." + exit 0 + fi + # --no-signoff: internal fork PRs do not require Signed-off-by. + perl scripts/checkpatch.pl --no-signoff /tmp/pr.patch \ + > /tmp/checkpatch.out 2>&1 || true + cat /tmp/checkpatch.out + + if grep -q '^ERROR:' /tmp/checkpatch.out; then + NERRS="$(grep -c '^ERROR:' /tmp/checkpatch.out)" + echo "::error::checkpatch reported ${NERRS} error(s); fix them before merge (warnings do not block)." + exit 1 + fi + echo "checkpatch: no errors." + + build: + name: Build with GCC (Arch config + fragments) + needs: checks + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/archlinux:base-devel + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + NEW_SYMBOLS: ${{ needs.checks.outputs.new_symbols }} + steps: + - name: Show disk space + run: df -h / + + - name: Bootstrap build environment + run: | + set -euxo pipefail + pacman -Sy --needed --noconfirm archlinux-keyring + pacman -Su --needed --noconfirm + # base-devel provides gcc/make/perl; the rest mirrors the makedepends + # of the release PKGBUILD minus clang and the rust toolchain. + pacman -S --needed --noconfirm \ + bc cpio gettext libelf pahole perl python tar xz zstd \ + file curl git ccache + + - name: Checkout PR merge result + uses: actions/checkout@v7 + with: + fetch-depth: 1 + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-pr-gcc-${{ github.sha }} + restore-keys: | + ccache-pr-gcc- + + - name: Assemble test config (same as the Arch release build) + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: official Arch Linux kernel .config extracted from the + # distro's linux-headers package (same method as the release build). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + pacman -Sw --noconfirm --cachedir "$CACHE" linux-headers + PKG="$(find "$CACHE" -maxdepth 1 -name 'linux-headers-*.pkg.tar.zst' -type f)" + if [ -z "$PKG" ]; then + echo "::error::linux-headers package not found in cache"; exit 1 + fi + CFG="$(tar --zstd -tf "$PKG" | grep -E '^usr/lib/modules/[^/]+/build/\.config$' || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel .config not found inside $PKG"; exit 1 + fi + tar --zstd -xOf "$PKG" "$CFG" > .config + test -s .config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # OGC kernel-packages fragments + the repo-local config.fragment + # (merged last, wins). Same order as the release build. + # ------------------------------------------------------------------ + for f in arch.config.set ogc.config.set arch.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash .github/packaging/merge-fragments.sh .config \ + arch.config.set ogc.config.set \ + arch.config.unset ogc.config.unset \ + .github/packaging/config.fragment + + # Same fixups as the release packaging... + scripts/config --set-str SYSTEM_TRUSTED_KEYS "" + scripts/config --set-str SYSTEM_REVOCATION_KEYS "" + scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + scripts/config --set-str CONFIG_LOCALVERSION "" || true + # ...plus test-only tweaks: + # No rust toolchain is installed in this job; the release builds + # (clang) compile the rust bits, this job focuses on the C parts. + scripts/config -d RUST + make olddefconfig + + - name: "Gate: new driver symbols must be enabled in the config" + run: | + set -euo pipefail + if [ -z "${NEW_SYMBOLS// }" ]; then + echo "No new driver CONFIG symbols in this PR; skipping gate." + exit 0 + fi + echo "Gate symbols: ${NEW_SYMBOLS}" + + FAIL=0 + for s in ${NEW_SYMBOLS}; do + if grep -qE "^CONFIG_${s}=[ym]$" .config; then + echo "OK: CONFIG_${s} is enabled." + elif grep -q "^# CONFIG_${s} is not set$" .config; then + echo "::error::CONFIG_${s} was added by this PR but is disabled in the merged config." + echo "::error::Enable it in .github/packaging/config.fragment (CONFIG_${s}=y or =m) if it should ship." + FAIL=1 + elif grep -q "^CONFIG_${s}=" .config; then + echo "::error::CONFIG_${s} is set but not to y/m: $(grep "^CONFIG_${s}=" .config)" + FAIL=1 + else + echo "::error::CONFIG_${s} is absent from the final .config (its dependencies or its vendor menu are off in the merged config)." + echo "::error::If this driver should ship, enable it (and its dependencies) in .github/packaging/config.fragment." + FAIL=1 + fi + done + if [ "$FAIL" -ne 0 ]; then + echo "::error::config gate failed: new drivers must be enabled in the OGC config." + exit 1 + fi + echo "config gate passed." + + - name: Build kernel with GCC + run: | + set -euxo pipefail + ccache -s || true + make CC="ccache gcc" WERROR=0 -j"$(nproc)" all + echo "Kernel release: $(make -s kernelrelease)" + ccache -s + df -h / + + - name: Summary + if: always() + run: | + { + echo "### GCC compile test" + echo + echo "- Compiler: gcc (Arch Linux) via ccache" + echo "- Config: Arch linux-headers base + OGC fragments + repo config.fragment (Rust disabled for this test)" + echo "- New driver symbols checked: ${NEW_SYMBOLS:-none}" + } >> "$GITHUB_STEP_SUMMARY" \ No newline at end of file From 2aaecd3850e803cd6121322aa46bea4b48c7e7d6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tomasz=20Paku=C5=82a?= Date: Tue, 3 Feb 2026 18:56:14 +0000 Subject: [PATCH 834/857] [FROM-ML] drm/amd/display: Add CH7218 PCON ID MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit [Why] Chrontel CH7218 found in Ugreen DP -> HDMI 2.1 adapter (model 85564) works perfectly with VRR after testing. VRR and FreeSync compatibility is explicitly advertised as a feature so it's addition is a formality. Support FreeSync info packet passthrough and "generic" HDMI VRR. [How] Add CH7218's ID to dm_helpers_is_vrr_pcon_allowed() Closes: https://gitlab.freedesktop.org/drm/amd/-/issues/4773 Signed-off-by: Tomasz Pakuła (cherry picked from commit 7b2436287ed953496ad1c9eb5820f00b75db597f) (cherry picked from commit a9c75486e9c655c604e05d55e1546f4e44b1bfd2) --- drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c | 1 + drivers/gpu/drm/amd/display/include/ddc_service_types.h | 1 + 2 files changed, 2 insertions(+) diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c index ab15ceec4daa0e..10472ae367e99c 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c @@ -1670,6 +1670,7 @@ STATIC_IFN_KUNIT const uint32_t dm_freesync_pcon_whitelist[] = { DP_BRANCH_DEVICE_ID_90CC24, DP_BRANCH_DEVICE_ID_001CF8, DP_BRANCH_DEVICE_ID_001FF2, + DP_BRANCH_DEVICE_ID_2B02F0, }; EXPORT_IF_KUNIT(dm_freesync_pcon_whitelist); diff --git a/drivers/gpu/drm/amd/display/include/ddc_service_types.h b/drivers/gpu/drm/amd/display/include/ddc_service_types.h index 827e9bd7c5cff3..d2a8e712d4a712 100644 --- a/drivers/gpu/drm/amd/display/include/ddc_service_types.h +++ b/drivers/gpu/drm/amd/display/include/ddc_service_types.h @@ -37,6 +37,7 @@ #define DP_BRANCH_DEVICE_ID_001CF8 0x001CF8 #define DP_BRANCH_DEVICE_ID_0060AD 0x0060AD #define DP_BRANCH_DEVICE_ID_001FF2 0x001FF2 +#define DP_BRANCH_DEVICE_ID_2B02F0 0x2B02F0 /* Chrontel CH7218 */ #define DP_BRANCH_HW_REV_10 0x10 #define DP_BRANCH_HW_REV_20 0x20 From 789eb18d263ce8ef74688504c9bc95f84fb73a4c Mon Sep 17 00:00:00 2001 From: Fangzhi Zuo Date: Fri, 14 Aug 2026 12:43:41 -0400 Subject: [PATCH 835/857] [FROM-ML] drm/amd/display: Add 2.1 FreeSync support for AMD VSDB EDID Block why: HDMI FRL sinks were not parsed for the AMD VSDB and no VTEM info packet was emitted for them, so 2.1 FreeSync over HDMI FRL did not work. It is backward-compatible with 2.0 FreeSync. how: - Accept SIGNAL_TYPE_HDMI_FRL alongside SIGNAL_TYPE_HDMI_TYPE_A when parsing the AMD VSDB in amdgpu_dm_update_freesync_caps(). - Build and send the VTEM info packet via mod_build_infopacket_vtem() when the stream signal is HDMI FRL during the freesync state update. - Set the VTEM Data_Set_Length to 0 when no VTEM feature is enabled. build_infopacket_header_vtem() hardcodes Data_Set_Length = 4, so a VTEM with Data_Set_Length = 4 would be transmitted even when no VTEM feature is enabled (VRR_EN = 0 and no FVA), e.g. when the sink advertises VRRMIN = 0 and vrr_capable is false. This fails HDMI GCTS HF1-58 step 6.2. The VTEM must keep being transmitted every MTW while VRR is enabled (HF1-58 steps 8.1 and 8.3), so it cannot simply be suppressed per frame. Instead, follow the MLDS option in HDMI 2.1 10.10.2.4: keep transmitting the VTEM but set Data_Set_Length = 0 when no feature is enabled. When VRR becomes active the full Data_Set_Length = 4 payload with VRR_EN = 1 is sent as before. Signed-off-by: Fangzhi Zuo Reviewed-by: Harry Wentland (cherry picked from commit 62ac74defba77c84daee2adc56b19b2c0b4e9afd) (cherry picked from commit 7484e04135e2269407dbff1f4952dff6da260b26) --- .../display/amdgpu_dm/amdgpu_dm_connector.c | 4 +- .../display/amdgpu_dm/amdgpu_dm_freesync.c | 3 + .../amd/display/modules/inc/mod_info_packet.h | 4 + .../display/modules/info_packet/info_packet.c | 109 ++++++++++++++++++ 4 files changed, 119 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c index 35960636bd0fd5..94ad7ecd50a8ca 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c @@ -3902,7 +3902,9 @@ void amdgpu_dm_update_freesync_caps(struct drm_connector *connector, amdgpu_dm_connector->as_type = ADAPTIVE_SYNC_TYPE_EDP; } - } else if (drm_edid && sink->sink_signal == SIGNAL_TYPE_HDMI_TYPE_A) { + } else if (drm_edid && + (sink->sink_signal == SIGNAL_TYPE_HDMI_TYPE_A || + sink->sink_signal == SIGNAL_TYPE_HDMI_FRL)) { i = parse_hdmi_amd_vsdb(amdgpu_dm_connector, edid, &vsdb_info); if (i >= 0) { amdgpu_dm_connector->vsdb_info = vsdb_info; diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c index 46674b83183cac..0d89159cc9c59c 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c @@ -229,6 +229,9 @@ void amdgpu_dm_update_freesync_state_on_stream( &vrr_infopacket, pack_sdp_v1_3); + if (new_stream->sink->sink_signal == SIGNAL_TYPE_HDMI_FRL) + mod_build_infopacket_vtem(new_stream, &vrr_params, 0, &vrr_infopacket); + new_crtc_state->freesync_vrr_info_changed |= (memcmp(&new_crtc_state->vrr_infopacket, &vrr_infopacket, diff --git a/drivers/gpu/drm/amd/display/modules/inc/mod_info_packet.h b/drivers/gpu/drm/amd/display/modules/inc/mod_info_packet.h index eee8206bc531a8..5181d889fe7fed 100644 --- a/drivers/gpu/drm/amd/display/modules/inc/mod_info_packet.h +++ b/drivers/gpu/drm/amd/display/modules/inc/mod_info_packet.h @@ -67,6 +67,10 @@ struct AS_Df_params { struct frame_duration_op decrease; }; +void mod_build_infopacket_vtem(const struct dc_stream_state *stream, + const struct mod_vrr_params *vrr, int fva_factor, + struct dc_info_packet *infopacket); + void mod_build_adaptive_sync_infopacket(const struct dc_stream_state *stream, enum adaptive_sync_type asType, const struct AS_Df_params *param, struct dc_info_packet *info_packet); diff --git a/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c b/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c index f5ac4bf32a784c..32b697f46788b5 100644 --- a/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c +++ b/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c @@ -291,6 +291,21 @@ void set_vsc_packet_colorimetry_data( info_packet->sb[18] = 0; } +static void set_field_with_mask(unsigned char *dest, unsigned int mask, unsigned int value) +{ + unsigned int shift = 0; + + if (!mask || !dest) + return; + + while (!((mask >> shift) & 1)) + shift++; + + *dest = *dest & ~mask; + value = value & (mask >> shift); + *dest = *dest | (value << shift); +} + void mod_build_vsc_infopacket(const struct dc_stream_state *stream, struct dc_info_packet *info_packet, enum dc_color_space cs, @@ -644,6 +659,100 @@ void mod_build_hf_vsif_infopacket(const struct dc_stream_state *stream, info_packet->valid = true; } +static void build_vtem_infopacket_data(const struct dc_stream_state *stream, + const struct mod_vrr_params *vrr, int fva_factor, + struct dc_info_packet *infopacket) +{ + unsigned int field_rate_in_hz; + + /* FVA Factor setting */ + set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__FVA_FACTOR_M1, + (fva_factor > 0) ? (fva_factor - 1) : 0); + /* VRR Parameters */ + if (vrr->state == VRR_STATE_ACTIVE_VARIABLE || + vrr->state == VRR_STATE_ACTIVE_FIXED) { + set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__VRR_EN, 1); + } else { + set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__VRR_EN, 0); + } + + if (vrr->state == VRR_STATE_ACTIVE_FIXED) + set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__M_CONST, vrr->m_const); + + if (!stream->timing.vic) { + set_field_with_mask(&infopacket->sb[VTEM_MD1], MASK_VTEM_MD1__BASE_VFRONT, + stream->timing.v_front_porch); + + + /* TODO: In dal2, we check mode flags for a reduced blanking timing. + * Need a way to relay that information to this function. + * if("ReducedBlanking") + * { + * set_field_with_mask(&infopacket->sb[VRR_VTEM_MD2], MASK__VRR_VTEM_MD2__RB, 1; + * } + */ + + field_rate_in_hz = stream->timing.pix_clk_100hz * 100; + field_rate_in_hz /= stream->timing.h_total; + field_rate_in_hz = (field_rate_in_hz + stream->timing.v_total / 2) + / stream->timing.v_total; + + set_field_with_mask(&infopacket->sb[VTEM_MD2], MASK_VTEM_MD2__BASE_REFRESH_RATE_98, + field_rate_in_hz >> 8); + set_field_with_mask(&infopacket->sb[VTEM_MD3], MASK_VTEM_MD3__BASE_REFRESH_RATE_07, + field_rate_in_hz); + + } + + /* + * When no VTEM feature is enabled (neither VRR nor FVA), signal a + * zero-length data set (MLDS) by clearing Data_Set_Length. HDMI 2.1 + * 10.10.2.4 requires the Source to either stop transmitting the VTEM + * or set Data_Set_Length = 0 when no feature is enabled; keeping the + * VTEM with Data_Set_Length = 0 preserves the every-MTW cadence while + * staying compliant (e.g. HDMI GCTS HF1-58 step 6.2). + */ + if (vrr->state != VRR_STATE_ACTIVE_VARIABLE && + vrr->state != VRR_STATE_ACTIVE_FIXED && fva_factor == 0) + set_field_with_mask(&infopacket->sb[VTEM_PB6], + MASK_VTEM_PB6__DATA_SET_LENGTH_LSB, 0); + + infopacket->valid = true; +} + +static void build_infopacket_header_vtem(enum signal_type signal, + struct dc_info_packet *infopacket) +{ + /* HEADER */ + + /* HB0, HB1, HB2 indicates PacketType VTEMPacket */ + infopacket->hb0 = 0x7F; + infopacket->hb1 = 0xC0; + infopacket->hb2 = 0x00; /* sequence_index */ + + set_field_with_mask(&infopacket->sb[VTEM_PB0], MASK_VTEM_PB0__VFR, 1); + set_field_with_mask(&infopacket->sb[VTEM_PB2], MASK_VTEM_PB2__ORGANIZATION_ID, 1); + set_field_with_mask(&infopacket->sb[VTEM_PB3], MASK_VTEM_PB3__DATA_SET_TAG_MSB, 0); + set_field_with_mask(&infopacket->sb[VTEM_PB4], MASK_VTEM_PB4__DATA_SET_TAG_LSB, 1); + set_field_with_mask(&infopacket->sb[VTEM_PB5], MASK_VTEM_PB5__DATA_SET_LENGTH_MSB, 0); + set_field_with_mask(&infopacket->sb[VTEM_PB6], MASK_VTEM_PB6__DATA_SET_LENGTH_LSB, 4); +} + +void mod_build_infopacket_vtem(const struct dc_stream_state *stream, + const struct mod_vrr_params *vrr, int fva_factor, + struct dc_info_packet *infopacket) +{ + /* VTEM info packet for HdmiVrr */ + + memset(infopacket, 0, sizeof(struct dc_info_packet)); + + /* VTEM Packet is structured differently */ + build_infopacket_header_vtem(stream->signal, infopacket); + build_vtem_infopacket_data(stream, vrr, fva_factor, infopacket); + + infopacket->valid = true; +} + void mod_build_adaptive_sync_infopacket(const struct dc_stream_state *stream, enum adaptive_sync_type asType, const struct AS_Df_params *param, From e4a664a66da15b73cca7a7c0cfbe48d88f74f06f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tomasz=20Paku=C5=82a?= Date: Fri, 14 Aug 2026 12:43:42 -0400 Subject: [PATCH 836/857] [FROM-ML] drm/edid: parse HDMI 2.1 gaming (ALLM/VRR) capabilities from HF-VSDB MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Parse the HDMI 2.1 gaming-related capabilities advertised in the HDMI Forum VSDB (HF-VSDB) and expose them through struct drm_hdmi_info so drivers can consume them. Add struct drm_hdmi_vrr_cap describing the sink's VRR capabilities: Fast VActive (Quick Frame Transport), Negative M VRR, Cinema VRR, MDelta, and the VRRmin/VRRmax range, together with a "supported" flag derived from that range. Add the fapa_start_location and allm (Auto Low Latency Mode) flags to struct drm_hdmi_info. drm_parse_hdmi_gaming_info() reads byte 8 of the HF-VSDB for the FAPA/ALLM/FVA/CNMVRR/CinemaVRR/MDelta flags and bytes 9-10 for VRRmin/VRRmax. Per HDMI 2.1, VRR is considered supported when VRRmin is within 1-48 and VRRmax is either 0 (maximum based on the video mode) or >= 100. It is invoked from drm_parse_hdmi_forum_scds(), and the parsed values are logged for debugging. Signed-off-by: Tomasz Pakuła Signed-off-by: Fangzhi Zuo Tested-by: Bernhard Berger Reviewed-by: Harry Wentland (cherry picked from commit 26e1509c3ec24f6314e972edd64d9dc18d8be779) (cherry picked from commit ae43fd66768efc4a0aade43fb4a4ba605637da15) --- drivers/gpu/drm/drm_edid.c | 42 +++++++++++++++++++++++++++++++++ include/drm/drm_connector.h | 47 +++++++++++++++++++++++++++++++++++++ 2 files changed, 89 insertions(+) diff --git a/drivers/gpu/drm/drm_edid.c b/drivers/gpu/drm/drm_edid.c index 07970e5b5f65e5..6a729b0655b48e 100644 --- a/drivers/gpu/drm/drm_edid.c +++ b/drivers/gpu/drm/drm_edid.c @@ -6182,6 +6182,33 @@ static void drm_parse_ycbcr420_deep_color_info(struct drm_connector *connector, hdmi->y420_dc_modes = dc_mask; } +static void drm_parse_hdmi_gaming_info(struct drm_hdmi_info *hdmi, const u8 *db) +{ + struct drm_hdmi_vrr_cap *vrr = &hdmi->vrr_cap; + + if (cea_db_payload_len(db) < 8) + return; + + hdmi->fapa_start_location = db[8] & DRM_EDID_FAPA_START_LOCATION; + hdmi->allm = db[8] & DRM_EDID_ALLM; + vrr->fva = db[8] & DRM_EDID_FVA; + vrr->cnmvrr = db[8] & DRM_EDID_CNMVRR; + vrr->cinema_vrr = db[8] & DRM_EDID_CINEMA_VRR; + vrr->mdelta = db[8] & DRM_EDID_MDELTA; + + if (cea_db_payload_len(db) < 9) + return; + + vrr->vrr_min = db[9] & DRM_EDID_VRR_MIN_MASK; + vrr->supported = (vrr->vrr_min > 0 && vrr->vrr_min <= 48); + + if (cea_db_payload_len(db) < 10) + return; + + vrr->vrr_max = (db[9] & DRM_EDID_VRR_MAX_UPPER_MASK) << 2 | db[10]; + vrr->supported &= (vrr->vrr_max == 0 || vrr->vrr_max >= 100); +} + static void drm_parse_dsc_info(struct drm_hdmi_dsc_cap *hdmi_dsc, const u8 *hf_scds) { @@ -6308,6 +6335,8 @@ static void drm_parse_hdmi_forum_scds(struct drm_connector *connector, drm_parse_ycbcr420_deep_color_info(connector, hf_scds); + drm_parse_hdmi_gaming_info(&connector->display_info.hdmi, hf_scds); + if (cea_db_payload_len(hf_scds) >= 11 && hf_scds[11]) { drm_parse_dsc_info(hdmi_dsc, hf_scds); dsc_support = true; @@ -6317,6 +6346,19 @@ static void drm_parse_hdmi_forum_scds(struct drm_connector *connector, "[CONNECTOR:%d:%s] HF-VSDB: max TMDS clock: %d KHz, HDMI 2.1 support: %s, DSC 1.2 support: %s\n", connector->base.id, connector->name, max_tmds_clock, str_yes_no(max_frl_rate), str_yes_no(dsc_support)); + drm_dbg_kms(connector->dev, + "[CONNECTOR:%d:%s] FAPA in blanking: %s, ALLM support: %s, Fast Vactive support: %s\n", + connector->base.id, connector->name, str_yes_no(hdmi->fapa_start_location), + str_yes_no(hdmi->allm), str_yes_no(hdmi->vrr_cap.fva)); + drm_dbg_kms(connector->dev, + "[CONNECTOR:%d:%s] Negative M VRR support: %s, CinemaVRR support: %s, Mdelta: %d\n", + connector->base.id, connector->name, str_yes_no(hdmi->vrr_cap.cnmvrr), + str_yes_no(hdmi->vrr_cap.cinema_vrr), hdmi->vrr_cap.mdelta); + drm_dbg_kms(connector->dev, + "[CONNECTOR:%d:%s] VRRmin: %u, VRRmax: %u, VRR supported: %s\n", + connector->base.id, connector->name, hdmi->vrr_cap.vrr_min, + hdmi->vrr_cap.vrr_max, str_yes_no(hdmi->vrr_cap.supported)); + } static void drm_parse_hdmi_deep_color_info(struct drm_connector *connector, diff --git a/include/drm/drm_connector.h b/include/drm/drm_connector.h index a0cf0268de483a..1659a220f1244d 100644 --- a/include/drm/drm_connector.h +++ b/include/drm/drm_connector.h @@ -254,6 +254,44 @@ struct drm_scdc { struct drm_scrambling scrambling; }; +/** + * struct drm_hdmi_vrr_cap - Information about VRR capabilities of a HDMI sink + * + * Describes the VRR support provided by HDMI 2.1 sink. The information is + * fetched fom additional HFVSDB blocks defined for HDMI 2.1. + */ +struct drm_hdmi_vrr_cap { + /** @fva: flag for Fast VActive (Quick Frame Transport) support */ + bool fva; + + /** @mcnmvrr: flag for Negative M VRR support */ + bool cnmvrr; + + /** @mcinema_vrr: flag for Cinema VRR support */ + bool cinema_vrr; + + /** @mdelta: flag for limited frame-to-frame compensation support */ + bool mdelta; + + /** + * @vrr_min : minimum supported variable refresh rate in Hz. + * Valid values only inide 1 - 48 range + */ + u16 vrr_min; + + /** + * @vrr_max : maximum supported variable refresh rate in Hz (optional). + * Valid values are either 0 (max based on video mode) or >= 100 + */ + u16 vrr_max; + + /** + * @supported: flag for vrr support based on checking for VRRmin and + * VRRmax values having correct values. + */ + bool supported; +}; + /** * struct drm_hdmi_dsc_cap - DSC capabilities of HDMI sink * @@ -330,6 +368,15 @@ struct drm_hdmi_info { /** @max_lanes: supported by sink */ u8 max_lanes; + /** @fapa_start_location: flag for the FAPA in blanking support */ + bool fapa_start_location; + + /** @allm: flag for Auto Low Latency Mode support by sink */ + bool allm; + + /** @vrr_cap: VRR capabilities of the sink */ + struct drm_hdmi_vrr_cap vrr_cap; + /** @dsc_cap: DSC capabilities of the sink */ struct drm_hdmi_dsc_cap dsc_cap; }; From cd36354d6062a796dc416dd9ae7bc9bef2428274 Mon Sep 17 00:00:00 2001 From: Fangzhi Zuo Date: Fri, 14 Aug 2026 12:43:43 -0400 Subject: [PATCH 837/857] [FROM-ML] drm/amd/display: Add HDMI 2.1 VRR support from HF-VSDB why: HDMI 2.1 sinks advertise their VRR range in the HDMI Forum VSDB (HF-VSDB), but amdgpu derived FreeSync capability only from the AMD VSDB. Sinks that expose just the HDMI Forum VRR capability (e.g. HDMI compliance EDIDs) were therefore reported as not VRR capable. how: - In amdgpu_dm_update_freesync_caps(), when the AMD VSDB does not provide a valid FreeSync range, fall back to the HDMI 2.1 VRR range parsed by DRM core from the HF-VSDB (connector->display_info.hdmi.vrr_cap). VRRMAX = 0 means "up to the Base Refresh Rate"; when the EDID provides no monitor range maximum either, fall back to the Base Refresh Rate (the highest refresh-rate mode of the preferred timing) so a valid VRR range is still reported to userspace. - Add VRR debug logging along the FreeSync capability and config paths. Signed-off-by: Fangzhi Zuo Reviewed-by: Harry Wentland (cherry picked from commit c5010ee089293c52c6489d308f1e659ba74f6ed5) (cherry picked from commit 8c96508fbe35ffbbc3cadfc23f1c1c74634a44ec) --- .../display/amdgpu_dm/amdgpu_dm_connector.c | 67 +++++++++++++++++++ .../display/amdgpu_dm/amdgpu_dm_freesync.c | 8 +++ 2 files changed, 75 insertions(+) diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c index 94ad7ecd50a8ca..57c6b32b951671 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c @@ -3876,6 +3876,15 @@ void amdgpu_dm_update_freesync_caps(struct drm_connector *connector, if (!adev->dm.freesync_module || !dc_supports_vrr(sink->ctx->dce_version)) goto update; + drm_dbg_driver(adev_to_drm(adev), + "VRR: enter signal=%d hdmi_vrr=%d mrange[%d-%d] hdmi.vrr_cap[sup=%d min=%d max=%d]\n", + sink->sink_signal, connector->display_info.hdmi.vrr_cap.supported, + connector->display_info.monitor_range.min_vfreq, + connector->display_info.monitor_range.max_vfreq, + connector->display_info.hdmi.vrr_cap.supported, + connector->display_info.hdmi.vrr_cap.vrr_min, + connector->display_info.hdmi.vrr_cap.vrr_max); + /* FIXME: Get rid of drm_edid_raw() */ edid = drm_edid_raw(drm_edid); @@ -3920,6 +3929,59 @@ void amdgpu_dm_update_freesync_caps(struct drm_connector *connector, connector->display_info.monitor_range.max_vfreq = vsdb_info.max_refresh_rate_hz; } } + + drm_dbg_driver(adev_to_drm(adev), + "VRR: amd_vsdb i=%d fs_sup=%d min=%d max=%d fs_capable=%d\n", + i, vsdb_info.freesync_supported, + vsdb_info.min_refresh_rate_hz, + vsdb_info.max_refresh_rate_hz, freesync_capable); + + /* + * If AMD VSDB didn't provide a valid FreeSync range, fall back to + * the HDMI 2.1 VRR capability parsed from the HF-VSDB. + */ + if (!freesync_capable && connector->display_info.hdmi.vrr_cap.supported) { + struct drm_hdmi_vrr_cap *vrr_cap = + &connector->display_info.hdmi.vrr_cap; + + drm_dbg_driver(adev_to_drm(adev), + "VRR: HF-VSDB fallback: hdmi_vrr=1 vrr_cap[sup=%d min=%d max=%d] mrange_max=%d\n", + vrr_cap->supported, vrr_cap->vrr_min, vrr_cap->vrr_max, + connector->display_info.monitor_range.max_vfreq); + + if (vrr_cap->supported && vrr_cap->vrr_min > 0) { + amdgpu_dm_connector->min_vfreq = vrr_cap->vrr_min; + amdgpu_dm_connector->max_vfreq = vrr_cap->vrr_max ? + vrr_cap->vrr_max : + connector->display_info.monitor_range.max_vfreq; + + /* + * VRRMAX = 0 in the HF-VSDB means "up to the Base + * Refresh Rate". If the EDID also did not provide a + * monitor range max, fall back to the Base Refresh + * Rate (the highest refresh rate of the preferred + * timing) so a valid VRR range is still reported to + * userspace. + */ + if (!amdgpu_dm_connector->max_vfreq) { + struct drm_display_mode *brr_mode = + amdgpu_dm_get_highest_refresh_rate_mode(amdgpu_dm_connector, true); + + if (brr_mode) + amdgpu_dm_connector->max_vfreq = + drm_mode_vrefresh(brr_mode); + } + + if (amdgpu_dm_connector->max_vfreq - + amdgpu_dm_connector->min_vfreq > 10) + freesync_capable = true; + + connector->display_info.monitor_range.min_vfreq = + amdgpu_dm_connector->min_vfreq; + connector->display_info.monitor_range.max_vfreq = + amdgpu_dm_connector->max_vfreq; + } + } } if (amdgpu_dm_connector->dc_link) @@ -3975,6 +4037,11 @@ void amdgpu_dm_update_freesync_caps(struct drm_connector *connector, if (dm_con_state) dm_con_state->freesync_capable = freesync_capable; + drm_dbg_driver(adev_to_drm(adev), + "VRR: caps result: freesync_capable=%d min_vfreq=%d max_vfreq=%d\n", + freesync_capable, amdgpu_dm_connector->min_vfreq, + amdgpu_dm_connector->max_vfreq); + if (connector->state && amdgpu_dm_connector->dc_link && !freesync_capable && amdgpu_dm_connector->dc_link->replay_settings.config.replay_supported) { amdgpu_dm_connector->dc_link->replay_settings.config.replay_supported = false; diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c index 0d89159cc9c59c..4dc5494f97117c 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c @@ -151,6 +151,14 @@ void amdgpu_dm_get_freesync_config_for_crtc( } out: new_crtc_state->freesync_config = config; + + drm_dbg_driver(new_con_state->base.connector->dev, + "VRR: cfg vrr_enabled=%d vrr_supported=%d fs_capable=%d vrefresh=%d min=%d max=%d state=%d\n", + new_crtc_state->base.vrr_enabled, + new_crtc_state->vrr_supported, + new_con_state->freesync_capable, vrefresh, + aconnector->min_vfreq, aconnector->max_vfreq, + config.state); } EXPORT_IF_KUNIT(amdgpu_dm_get_freesync_config_for_crtc); From 38e057b531d055af6f663e689ada1b84343a0508 Mon Sep 17 00:00:00 2001 From: Fangzhi Zuo Date: Fri, 14 Aug 2026 12:43:44 -0400 Subject: [PATCH 838/857] [FROM-ML] drm/amd/display: Enable HDMI ALLM for Gaming-VRR why: HDMI 2.1 Auto Low-Latency Mode (ALLM) lets a Source request the Sink's low-latency mode through the HF-VSIF. HDMI 2.1 Section 7.6.6 requires that when Gaming-VRR is enabled (VRR_EN=1) and the Sink advertises ALLM in the SCDS, the Source shall transmit the HF-VSIF and set ALLM_Mode=1. amdgpu never set ALLM_Mode, so this requirement was not met. how: - In update_freesync_state_on_stream(), set ALLM_Mode=1 in the HF-VSIF when Gaming-VRR is active (vrr state ACTIVE_VARIABLE/ACTIVE_FIXED, i.e. VRR_EN=1) and the sink advertises ALLM, per HDMI 2.1 Section 7.6.6, and push the updated HF-VSIF (vsp_infopacket) as a stream update. ALLM is driven only by the mandatory Gaming-VRR case. Signed-off-by: Fangzhi Zuo (cherry picked from commit 7cafa47e65ace3daf0758553da2d6131c4290b5f) (cherry picked from commit 2d28e87a06f6203ff260ddead68d9aa22f5d5a58) --- .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c | 5 +++- .../display/amdgpu_dm/amdgpu_dm_freesync.c | 29 +++++++++++++++++++ 2 files changed, 33 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c index eb00c62c6f7244..a27c1fa7499462 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c @@ -4027,9 +4027,12 @@ static void amdgpu_dm_commit_planes(struct drm_atomic_commit *state, } if (acrtc_state->stream) { - if (acrtc_state->freesync_vrr_info_changed) + if (acrtc_state->freesync_vrr_info_changed) { bundle->stream_update.vrr_infopacket = &acrtc_state->stream->vrr_infopacket; + bundle->stream_update.vsp_infopacket = + &acrtc_state->stream->vsp_infopacket; + } } } diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c index 4dc5494f97117c..402b7cd3a87589 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c @@ -251,6 +251,35 @@ void amdgpu_dm_update_freesync_state_on_stream( new_stream->vrr_infopacket = vrr_infopacket; new_stream->allow_freesync = mod_freesync_get_freesync_enabled(&vrr_params); + /* + * HDMI ALLM: when Gaming-VRR is active (VRR_EN=1) and the sink + * advertises ALLM in the SCDS, the Source shall transmit the HF-VSIF + * with ALLM_Mode=1 (HDMI 2.1 Section 7.6.6). + */ + if (new_stream->signal == SIGNAL_TYPE_HDMI_TYPE_A || + new_stream->signal == SIGNAL_TYPE_HDMI_FRL) { + struct dc_info_packet vsp_infopacket = {0}; + bool sink_allm = aconn && aconn->base.display_info.hdmi.allm; + bool allm = sink_allm && + (vrr_params.state == VRR_STATE_ACTIVE_VARIABLE || + vrr_params.state == VRR_STATE_ACTIVE_FIXED); + bool allm_changed; + + mod_build_hf_vsif_infopacket(new_stream, &vsp_infopacket, allm, allm); + + allm_changed = memcmp(&new_stream->vsp_infopacket, &vsp_infopacket, + sizeof(vsp_infopacket)) != 0; + new_crtc_state->freesync_vrr_info_changed |= allm_changed; + new_stream->vsp_infopacket = vsp_infopacket; + + if (allm_changed) + drm_dbg_driver(adev_to_drm(adev), + "ALLM: flip on crtc=%u: sink_allm=%d vrr_state=%d -> ALLM_Mode=%d\n", + new_crtc_state->base.crtc->base.id, + sink_allm, + vrr_params.state, allm); + } + if (new_crtc_state->freesync_vrr_info_changed) drm_dbg_kms(adev_to_drm(adev), "VRR packet update: crtc=%u enabled=%d state=%d", new_crtc_state->base.crtc->base.id, From 0c3a53ad235489504f104b5c7040595fc2a7d51d Mon Sep 17 00:00:00 2001 From: Andy East <162940559+andy10115@users.noreply.github.com> Date: Mon, 24 Aug 2026 00:37:47 -0400 Subject: [PATCH 839/857] [FOR-UPSTREAM] drm/amd/display: enable FreeSync on HDMI desktop Enable freesync_on_desktop for HDMI streams so the display can keep FreeSync enabled during normal desktop use. This allows the HDMI VRR path to support fixed-refresh desktop operation while retaining FreeSync signaling for the display. Signed-off-by: Andy East (cherry picked from commit 7e64309f7e376cbcb249afefa6447b83bfda5245) --- drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c index 402b7cd3a87589..acbce05ee0eda8 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c @@ -333,6 +333,10 @@ void amdgpu_dm_update_stream_irq_parameters( VRR_STATE_ACTIVE_VARIABLE : VRR_STATE_INACTIVE; } + + /* Enable freesync on desktop for HDMI */ + if (dc_is_hdmi_signal(new_stream->signal)) + new_stream->freesync_on_desktop = true; } else { config.state = VRR_STATE_UNSUPPORTED; } From a2e37b3877f35cffbf99b67137445fcb921b10a9 Mon Sep 17 00:00:00 2001 From: Andy East <162940559+andy10115@users.noreply.github.com> Date: Mon, 24 Aug 2026 00:44:28 -0400 Subject: [PATCH 840/857] [FOR-UPSTREAM] drm/amd/display: Simplify VRR state handling in info_packet.c Treat VRR_STATE_INACTIVE as VRR-active when freesync_on_desktop is enabled so HDMI VTEM continues advertising VRR during fixed-refresh desktop use. Use a single vrr_active value for both the VTEM VRR_EN bit and the Data_Set_Length decision. This keeps VTEM signaling consistent while allowing the display to remain in its VRR mode without varying the actual refresh rate. Signed-off-by: Andy East (cherry picked from commit a3020f596a52166d40676b35ad56ba5eb2239dcb) --- .../display/modules/info_packet/info_packet.c | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c b/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c index 32b697f46788b5..fec5b0c9dc65e5 100644 --- a/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c +++ b/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c @@ -665,12 +665,22 @@ static void build_vtem_infopacket_data(const struct dc_stream_state *stream, { unsigned int field_rate_in_hz; + /* + * Enables FreeSync-like behavior by keeping HDMI VRR signalling active + * in fixed refresh rate conditions like normal desktop work/web browsing. + * Functionally behaves like non-VRR mode by keeping the actual refresh + * rate fixed. + */ + const bool vrr_active = vrr->state == VRR_STATE_ACTIVE_VARIABLE || + vrr->state == VRR_STATE_ACTIVE_FIXED || + (stream->freesync_on_desktop && + vrr->state == VRR_STATE_INACTIVE); + /* FVA Factor setting */ set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__FVA_FACTOR_M1, (fva_factor > 0) ? (fva_factor - 1) : 0); /* VRR Parameters */ - if (vrr->state == VRR_STATE_ACTIVE_VARIABLE || - vrr->state == VRR_STATE_ACTIVE_FIXED) { + if (vrr_active) { set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__VRR_EN, 1); } else { set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__VRR_EN, 0); @@ -712,8 +722,7 @@ static void build_vtem_infopacket_data(const struct dc_stream_state *stream, * VTEM with Data_Set_Length = 0 preserves the every-MTW cadence while * staying compliant (e.g. HDMI GCTS HF1-58 step 6.2). */ - if (vrr->state != VRR_STATE_ACTIVE_VARIABLE && - vrr->state != VRR_STATE_ACTIVE_FIXED && fva_factor == 0) + if (!vrr_active && fva_factor == 0) set_field_with_mask(&infopacket->sb[VTEM_PB6], MASK_VTEM_PB6__DATA_SET_LENGTH_LSB, 0); From 236eb7c456b28feffc3d4e272b78fa41f91dc87d Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Tue, 18 Aug 2026 06:53:12 +0500 Subject: [PATCH 841/857] [FOR-UPSTREAM] HID: asus: add ROG Zephyrus Duo GX651AR keyboard The detachable keyboard shipped with the ROG Zephyrus Duo GX651AR (0b05:1ce6) is a ROG N-Key keyboard, but it is not listed in asus_devices[], so its interfaces are left to hid-generic and its vendor usages are never mapped by asus_input_mapping(). Add it with QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD, matching the other ROG N-Key keyboards. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit 8d70b5f90b41b98f9fa6293e03be78c31bac8c36) --- drivers/hid/hid-asus.c | 3 +++ drivers/hid/hid-ids.h | 1 + 2 files changed, 4 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index ec966fc0a411c5..900521614232c4 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1693,6 +1693,9 @@ static const struct hid_device_id asus_devices[] = { { HID_I2C_DEVICE(USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD | QUIRK_HID_FN_LOCK }, + { HID_USB_DEVICE(USB_VENDOR_ID_ASUSTEK, + USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3), + QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, { HID_USB_DEVICE(USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_Z13_LIGHTBAR), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, diff --git a/drivers/hid/hid-ids.h b/drivers/hid/hid-ids.h index b3aca5aa917677..969ea428a5bebe 100644 --- a/drivers/hid/hid-ids.h +++ b/drivers/hid/hid-ids.h @@ -227,6 +227,7 @@ #define USB_DEVICE_ID_ASUSTEK_ROG_KEYBOARD3 0x1822 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD 0x1866 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2 0x19b6 +#define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3 0x1ce6 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_FOLIO 0x1a30 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_LIGHTBAR 0x18c6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY 0x1abe From a7c212f39dea4310b30d33bd5e8c33a9b441151e Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Tue, 18 Aug 2026 09:46:22 +0500 Subject: [PATCH 842/857] [FOR-UPSTREAM] HID: asus: force input connection on vendor-only N-Key interfaces On the ROG Zephyrus Duo GX651AR (0b05:1ce6) the hotkeys live on report 0x5a on an interface whose descriptor holds nothing but two ASUS vendor collections. Neither satisfies IS_INPUT_APPLICATION(), so hidinput_connect() creates no input device, asus_input_mapping() never runs and every hotkey is dropped by asus_event() as unmapped. Set HID_QUIRK_HIDINPUT_FORCE on ROG N-Key interfaces that carry an ASUS vendor input report so those usages get mapped. Interfaces left with no mapped usage are still discarded by hidinput_has_been_populated(). The vendor check reads report_enum[HID_INPUT_REPORT], so interfaces with no input reports, such as the RGB control interface, are unaffected. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit 69887e3b53350a792c34272d0b101a21131a7ed7) --- drivers/hid/hid-asus.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 900521614232c4..8cd6e1fc578f70 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1480,6 +1480,14 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) is_vendor = true; } + /* + * A vendor collection may be the only application collection on the + * interface, which hidinput_connect() otherwise skips, leaving the + * hotkey usages unmapped. Unpopulated inputs are dropped later. + */ + if (is_vendor && (drvdata->quirks & QUIRK_ROG_NKEY_KEYBOARD)) + hdev->quirks |= HID_QUIRK_HIDINPUT_FORCE; + ret = asus_worker_create(hdev, drvdata); if (ret) { hid_warn(hdev, "Failed to initialize worker: %d\n", ret); From bb52cc4a431c584b066433a4a8f502b4b8c4c0ae Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Sat, 22 Aug 2026 16:28:13 +0500 Subject: [PATCH 843/857] [FOR-UPSTREAM] HID: asus: add ROG Zephyrus Duo GX651AR keyboard over Bluetooth The GX651AR keyboard enumerates as 0b05:1ce6 over USB but pairs as 0b05:1ce7 in Bluetooth mode, where the keyboard, consumer and both ASUS vendor collections (reports 0x5a and 0x5d) sit on a single HID device. Add it with the same quirks as the USB entry. Bind to HID_GROUP_GENERIC so that hid-multitouch keeps the digitizer. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit b0dbc09e46a5d146c25081943f16fc0b1d0c0492) --- drivers/hid/hid-asus.c | 3 +++ drivers/hid/hid-ids.h | 1 + 2 files changed, 4 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 8cd6e1fc578f70..8beec04d9a6cf6 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1742,6 +1742,9 @@ static const struct hid_device_id asus_devices[] = { { HID_DEVICE(BUS_USB, HID_GROUP_GENERIC, USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_Z13_FOLIO), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, + { HID_DEVICE(BUS_BLUETOOTH, HID_GROUP_GENERIC, + USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3_BT), + QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, { HID_DEVICE(BUS_USB, HID_GROUP_GENERIC, USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_T101HA_KEYBOARD) }, { } diff --git a/drivers/hid/hid-ids.h b/drivers/hid/hid-ids.h index 969ea428a5bebe..c48791c352aa2e 100644 --- a/drivers/hid/hid-ids.h +++ b/drivers/hid/hid-ids.h @@ -228,6 +228,7 @@ #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD 0x1866 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2 0x19b6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3 0x1ce6 +#define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3_BT 0x1ce7 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_FOLIO 0x1a30 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_LIGHTBAR 0x18c6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY 0x1abe From e2c9f8aca03a0894d9644fe9bddfb4e548c6d82b Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Sat, 22 Aug 2026 16:35:21 +0500 Subject: [PATCH 844/857] [FOR-UPSTREAM] HID: asus: map the tent mode key on ROG Zephyrus Duo GX651AR Fn+F12 on the GX651AR keyboard emits ASUS vendor code 0x9c, which asus_input_mapping() does not know about, so asus_event() drops it as unmapped. Map it to KEY_F19. F13 to F18 are already used for ASUS toggles that have no generic keycode. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit 337a811e8f0536e8ff649c533d00bfa973e5618b) --- drivers/hid/hid-asus.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 8beec04d9a6cf6..4f578446004a8e 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1267,6 +1267,7 @@ static int asus_input_mapping(struct hid_device *hdev, case 0xa6: asus_map_key_clear(KEY_F16); break; /* ROG Ally QAM button */ case 0xa7: asus_map_key_clear(KEY_F17); break; /* ROG Ally ROG long-press */ case 0xa8: asus_map_key_clear(KEY_F18); break; /* ROG Ally ROG long-press-release */ + case 0x9c: asus_map_key_clear(KEY_F19); break; /* Zephyrus Duo tent mode */ default: /* ASUS lazily declares 256 usages, ignore the rest, From e9a66b6cd33aa57d1043b4995af0f77f7f975476 Mon Sep 17 00:00:00 2001 From: Mario Limonciello Date: Mon, 31 Aug 2026 00:51:07 -0500 Subject: [PATCH 845/857] [FROM-ML] iommu/amd: Add PerfOpt IOMMU performance optimization support Add support for the AMD IOMMU Performance Optimization (PerfOpt) feature as defined in the AMD I/O Virtualization Technology (IOMMU) Specification, Section 3.4.9 (MMIO Offset 016Ch). This feature allows privileged integrated I/O devices (GPUs) to bypass the IOMMU when directly accessing system memory. The IOMMU only enforces the IR/IW permission bits without GPA->SPA translations. amd_iommu_enable_perfopt() performs a detach/reattach cycle to rehome devices already on the identity domain with ATS/PRI/PASID/GCR3 disabled (skip_caps path). amd_iommu_disable_perfopt() restores those capabilities. The per-device dev_data->perfopt flag tracks state. PERF_OPT_EN is a single control bit per IOMMU, shared by every device behind that IOMMU, while enablement is requested per device. It is therefore reference counted (amd_iommu->perfopt_refcount): armed on the first requesting device and cleared on the last, so one device's teardown never clears the bit while a peer behind the same IOMMU still needs it. The per-device flag is cleared on every teardown path (blocked_domain_attach, release_device, and amd_iommu_disable_perfopt), dropping the reference with it, so a reused dev_data never carries stale PerfOpt state onto its next bind. On suspend/resume the hardware is reprogrammed from scratch: amd_iommu_perfopt_clear() forces the bit off without touching the reference count, and amd_iommu_perfopt_restore() re-asserts it from the count after early_enable_iommu(), so armed devices keep the optimization across resume without relying on each consumer driver to re-arm. The exported amd_iommu_enable_perfopt()/amd_iommu_disable_perfopt() run only from a consumer driver's bind/unbind path. group->mutex is not exposed to drivers, but a device bound to its native driver cannot have its IOMMU domain changed concurrently by the core, which serializes the detach/attach pair against core-driven attach. PerfOpt is opt-in -- only enabled when explicitly requested by a driver. Co-developed-by: Jatin Kataria Signed-off-by: Jatin Kataria Link: https://patch.msgid.link/20260831055108.1893285-2-mario.limonciello@amd.com Signed-off-by: Mario Limonciello (cherry picked from commit e946b6eaa5263685fbd8e80fb312fc96ac65dd17) --- drivers/iommu/amd/amd_iommu.h | 3 + drivers/iommu/amd/amd_iommu_types.h | 7 + drivers/iommu/amd/init.c | 43 ++++++ drivers/iommu/amd/iommu.c | 214 ++++++++++++++++++++++++++++ include/linux/amd-iommu.h | 11 ++ 5 files changed, 278 insertions(+) diff --git a/drivers/iommu/amd/amd_iommu.h b/drivers/iommu/amd/amd_iommu.h index a2fe804b038b64..1f8f9df8e6c240 100644 --- a/drivers/iommu/amd/amd_iommu.h +++ b/drivers/iommu/amd/amd_iommu.h @@ -48,6 +48,9 @@ extern u8 amd_iommu_hpt_vasize; extern unsigned long amd_iommu_pgsize_bitmap; extern bool amd_iommu_hatdis; +int amd_iommu_perfopt_clear(struct amd_iommu *iommu); +int amd_iommu_perfopt_restore(struct amd_iommu *iommu); + /* Protection domain ops */ void amd_iommu_init_identity_domain(void); struct protection_domain *protection_domain_alloc(void); diff --git a/drivers/iommu/amd/amd_iommu_types.h b/drivers/iommu/amd/amd_iommu_types.h index 3dbe20023456b4..755421e5cd7576 100644 --- a/drivers/iommu/amd/amd_iommu_types.h +++ b/drivers/iommu/amd/amd_iommu_types.h @@ -65,6 +65,7 @@ #define MMIO_MSI_ADDR_LO_OFFSET 0x015C #define MMIO_MSI_ADDR_HI_OFFSET 0x0160 #define MMIO_MSI_DATA_OFFSET 0x0164 +#define MMIO_PERF_OPT_OFFSET 0x016C #define MMIO_INTCAPXT_EVT_OFFSET 0x0170 #define MMIO_INTCAPXT_PPR_OFFSET 0x0178 #define MMIO_INTCAPXT_GALOG_OFFSET 0x0180 @@ -99,6 +100,8 @@ #define FEATURE_GLX GENMASK_ULL(15, 14) #define FEATURE_GAM_VAPIC BIT_ULL(21) #define FEATURE_PASMAX GENMASK_ULL(36, 32) +#define FEATURE_PERF_OPT BIT_ULL(45) +#define PERF_OPT_EN BIT(13) #define FEATURE_GIOSUP BIT_ULL(48) #define FEATURE_HASUP BIT_ULL(49) #define FEATURE_EPHSUP BIT_ULL(50) @@ -670,6 +673,9 @@ struct amd_iommu { /* Extended features 2 */ u64 features2; + /* Devices requesting PerfOpt; the shared PERF_OPT_EN bit is on while >0. Protected by @lock. */ + int perfopt_refcount; + /* PCI device id of the IOMMU device */ u16 devid; @@ -831,6 +837,7 @@ struct iommu_dev_data { u8 ppr :1; /* Enable device PPR support */ bool use_vapic; /* Enable device to use vapic mode */ bool defer_attach; + bool perfopt; struct ratelimit_state rs; /* Ratelimit IOPF messages */ }; diff --git a/drivers/iommu/amd/init.c b/drivers/iommu/amd/init.c index 40726dfef27336..ddcf56f1016751 100644 --- a/drivers/iommu/amd/init.c +++ b/drivers/iommu/amd/init.c @@ -1942,6 +1942,9 @@ static int __init init_iommu_one(struct amd_iommu *iommu, struct ivhd_header *h, if (!iommu->mmio_base) return -ENOMEM; + if (amd_iommu_perfopt_clear(iommu)) + pr_err("IOMMU%d: failed to clear PerfOpt\n", iommu->index); + return init_iommu_from_acpi(iommu, h); } @@ -3032,10 +3035,46 @@ static void enable_iommus_vapic(void) #endif } +static int clear_perfopt_all(void) +{ + struct amd_iommu *iommu; + int err, ret = 0; + + for_each_iommu(iommu) { + err = amd_iommu_perfopt_clear(iommu); + if (err) + ret = err; + } + + return ret; +} + +static int restore_perfopt_all(void) +{ + struct amd_iommu *iommu; + int err, ret = 0; + + for_each_iommu(iommu) { + err = amd_iommu_perfopt_restore(iommu); + if (err) + ret = err; + } + + return ret; +} + static void disable_iommus(void) { struct amd_iommu *iommu; + /* + * PerfOpt is an optional performance bit, so a failure to clear it must + * not skip the mandatory disable below. This also runs from the void + * amd_iommu_disable() shutdown/kexec path, which cannot report an error. + */ + if (clear_perfopt_all()) + pr_err("Failed to clear PerfOpt while disabling IOMMUs\n"); + for_each_iommu(iommu) iommu_disable(iommu); @@ -3061,6 +3100,10 @@ static void amd_iommu_resume(void *data) for_each_iommu(iommu) early_enable_iommu(iommu); + /* early_enable_iommu() cleared PERF_OPT_EN; re-assert it from the refcount. */ + if (restore_perfopt_all()) + pr_err("Failed to restore PerfOpt after IOMMU resume\n"); + iommu_enable_event_buffer(); amd_iommu_enable_interrupts(); } diff --git a/drivers/iommu/amd/iommu.c b/drivers/iommu/amd/iommu.c index 4dc306a4b5c620..fa60affdfc0350 100644 --- a/drivers/iommu/amd/iommu.c +++ b/drivers/iommu/amd/iommu.c @@ -2395,6 +2395,9 @@ static int attach_device(struct device *dev, if (ret) goto out; + if (dev_data->perfopt) + goto skip_caps; + /* Setup GCR3 table */ if (pdom_is_sva_capable(domain)) { ret = init_gcr3_table(dev_data, domain); @@ -2419,6 +2422,7 @@ static int attach_device(struct device *dev, pdev_enable_cap_ats(pdev); } +skip_caps: /* Update data structures */ dev_data->domain = domain; spin_lock_irqsave(&domain->lock, flags); @@ -2487,6 +2491,192 @@ static void detach_device(struct device *dev) mutex_unlock(&dev_data->mutex); } +/* Program the per-IOMMU PerfOpt enable bit. Caller must hold iommu->lock. */ +static int __perfopt_write(struct amd_iommu *iommu, bool enable) +{ + u32 old, val, readback; + + if (!(readq(iommu->mmio_base + MMIO_EXT_FEATURES) & FEATURE_PERF_OPT)) + return enable ? -ENODEV : 0; + + old = readl(iommu->mmio_base + MMIO_PERF_OPT_OFFSET); + if (old == U32_MAX) + return -EIO; + + val = enable ? old | PERF_OPT_EN : old & ~PERF_OPT_EN; + if (val != old) + writel(val, iommu->mmio_base + MMIO_PERF_OPT_OFFSET); + readback = readl(iommu->mmio_base + MMIO_PERF_OPT_OFFSET); + if (readback == U32_MAX || + (readback & PERF_OPT_EN) != (val & PERF_OPT_EN)) + return -EIO; + return 0; +} + +/* + * PERF_OPT_EN is a single bit shared by every device behind @iommu, so it is + * reference counted: armed on the first requesting device, cleared on the last. + */ +static int perfopt_get(struct amd_iommu *iommu) +{ + unsigned long flags; + int ret = 0; + + if (!iommu->mmio_base) + return 0; + + raw_spin_lock_irqsave(&iommu->lock, flags); + if (iommu->perfopt_refcount == 0) { + ret = __perfopt_write(iommu, true); + if (ret) + goto out; + } + iommu->perfopt_refcount++; +out: + raw_spin_unlock_irqrestore(&iommu->lock, flags); + return ret; +} + +static int perfopt_put(struct amd_iommu *iommu) +{ + unsigned long flags; + int ret = 0; + + if (!iommu->mmio_base) + return 0; + + raw_spin_lock_irqsave(&iommu->lock, flags); + if (iommu->perfopt_refcount > 0 && --iommu->perfopt_refcount == 0) + ret = __perfopt_write(iommu, false); + raw_spin_unlock_irqrestore(&iommu->lock, flags); + return ret; +} + +/* + * Force PERF_OPT_EN off without touching the refcount (used on init, shutdown, + * and suspend). The count is preserved so amd_iommu_perfopt_restore() can + * re-arm on resume. + */ +int amd_iommu_perfopt_clear(struct amd_iommu *iommu) +{ + unsigned long flags; + int ret; + + if (!iommu->mmio_base) + return 0; + + raw_spin_lock_irqsave(&iommu->lock, flags); + ret = __perfopt_write(iommu, false); + raw_spin_unlock_irqrestore(&iommu->lock, flags); + return ret; +} + +/* + * Re-assert PERF_OPT_EN from the refcount after the hardware was reprogrammed on + * resume, so devices armed before suspend keep the optimization without each + * consumer driver re-arming. + */ +int amd_iommu_perfopt_restore(struct amd_iommu *iommu) +{ + unsigned long flags; + int ret; + + if (!iommu->mmio_base) + return 0; + + raw_spin_lock_irqsave(&iommu->lock, flags); + ret = __perfopt_write(iommu, iommu->perfopt_refcount > 0); + raw_spin_unlock_irqrestore(&iommu->lock, flags); + return ret; +} + +int amd_iommu_enable_perfopt(struct pci_dev *pdev) +{ + struct iommu_dev_data *dev_data = dev_iommu_priv_get(&pdev->dev); + struct amd_iommu *iommu = rlookup_amd_iommu(&pdev->dev); + struct protection_domain *domain; + int ret; + + if (!iommu || !dev_data) + return -ENODEV; + + if (!(iommu->features & FEATURE_PERF_OPT)) + return -ENODEV; + + domain = dev_data->domain; + if (!domain) + return -ENODEV; + + /* Already armed for this device (e.g. re-entry on resume). */ + if (dev_data->perfopt) + return 0; + + /* + * The bit is only architecturally valid while the device is untranslated: + * identity domain with ATS/PRI/PASID off. The identity domain is + * SVA-capable so attach_device() enabled ATS/PRI/PASID and built a GCR3 + * table. Re-home the device onto the same identity domain with + * perfopt set, so the attach_device() skip_caps path leaves + * ATS/PRI/PASID off and no GCR3 table. This follows the detach/attach + * pattern used by amd_iommu_attach_device(). + * + * Locking: this and amd_iommu_disable_perfopt() run only from the + * consumer driver's bind/unbind path. group->mutex is not exposed to + * drivers, but a device bound to its native driver cannot have its domain + * changed concurrently by the core (VFIO ownership is mutually exclusive; + * sysfs domain changes require an unused group), so the detach/attach pair + * is serialized without it. + */ + dev_data->perfopt = true; + detach_device(&pdev->dev); + ret = attach_device(&pdev->dev, domain); + if (ret) + goto err_restore; + + ret = perfopt_get(iommu); + if (ret) + goto err_rearm; + + dev_info_once(&pdev->dev, "PerfOpt armed on IOMMU%d\n", iommu->index); + return 0; + +err_rearm: + detach_device(&pdev->dev); +err_restore: + dev_data->perfopt = false; + if (attach_device(&pdev->dev, domain)) + pci_err(pdev, "failed to restore state after PerfOpt setup; device left detached\n"); + dev_err_once(&pdev->dev, "PerfOpt failed to arm on IOMMU%d (%d)\n", + iommu->index, ret); + return ret; +} +EXPORT_SYMBOL_GPL(amd_iommu_enable_perfopt); + +void amd_iommu_disable_perfopt(struct pci_dev *pdev) +{ + struct iommu_dev_data *dev_data = dev_iommu_priv_get(&pdev->dev); + struct amd_iommu *iommu = rlookup_amd_iommu(&pdev->dev); + struct protection_domain *domain; + + if (!iommu || !dev_data || !dev_data->perfopt || !dev_data->domain) + return; + + if (WARN_ON(perfopt_put(iommu))) + pci_err(pdev, "failed to clear PerfOpt\n"); + + /* + * Restore ATS/PRI/PASID (and thus SVA) by re-homing the device onto its + * identity domain with the flag cleared, so a later bind without PerfOpt + * sees a normally-capable device. See the locking note in + * amd_iommu_enable_perfopt(). + */ + domain = dev_data->domain; + dev_data->perfopt = false; + detach_device(&pdev->dev); + if (attach_device(&pdev->dev, domain)) + pci_err(pdev, "failed to restore caps after PerfOpt disable\n"); +} +EXPORT_SYMBOL_GPL(amd_iommu_disable_perfopt); static struct iommu_device *amd_iommu_probe_device(struct device *dev) { struct iommu_device *iommu_dev; @@ -2554,6 +2744,14 @@ static struct iommu_device *amd_iommu_probe_device(struct device *dev) static void amd_iommu_release_device(struct device *dev) { struct iommu_dev_data *dev_data = dev_iommu_priv_get(dev); + struct amd_iommu *iommu = get_amd_iommu_from_dev_data(dev_data); + + if (dev_data->perfopt) { + if (WARN_ON(perfopt_put(iommu))) + dev_err(dev, "IOMMU%d: failed to clear PerfOpt on release\n", + iommu->index); + dev_data->perfopt = false; + } WARN_ON(dev_data->domain); @@ -2928,6 +3126,19 @@ static int blocked_domain_attach_device(struct iommu_domain *domain, struct iommu_domain *old) { struct iommu_dev_data *dev_data = dev_iommu_priv_get(dev); + struct amd_iommu *iommu = get_amd_iommu_from_dev_data(dev_data); + + /* + * blocked_domain is also the .release_domain, so this is the normal + * teardown path: drop the reference and clear the flag here too, and + * don't fail teardown if the WARN-guarded write doesn't stick. + */ + if (dev_data->perfopt) { + if (WARN_ON(perfopt_put(iommu))) + dev_err(dev, "IOMMU%d: failed to clear PerfOpt for blocked domain\n", + iommu->index); + dev_data->perfopt = false; + } if (dev_data->domain) detach_device(dev); @@ -2996,6 +3207,9 @@ static int amd_iommu_attach_device(struct iommu_domain *dom, struct device *dev, struct amd_iommu *iommu = get_amd_iommu_from_dev(dev); int ret; + if (dev_data->perfopt && !pdom_is_in_pt_mode(domain)) + return -EBUSY; + /* * Skip attach device to domain if new domain is same as * devices current domain diff --git a/include/linux/amd-iommu.h b/include/linux/amd-iommu.h index edcee9f5335a6f..e03575cbc08c6c 100644 --- a/include/linux/amd-iommu.h +++ b/include/linux/amd-iommu.h @@ -76,4 +76,15 @@ static inline int amd_iommu_snp_disable(void) { return 0; } static inline bool amd_iommu_sev_tio_supported(void) { return false; } #endif +#ifdef CONFIG_AMD_IOMMU +int amd_iommu_enable_perfopt(struct pci_dev *pdev); +void amd_iommu_disable_perfopt(struct pci_dev *pdev); +#else +static inline int amd_iommu_enable_perfopt(struct pci_dev *pdev) +{ + return 0; +} +static inline void amd_iommu_disable_perfopt(struct pci_dev *pdev) { } +#endif + #endif /* _ASM_X86_AMD_IOMMU_H */ From 235250e8fe8eb795aa23f4c629b41c24d25be9e1 Mon Sep 17 00:00:00 2001 From: Mario Limonciello Date: Mon, 31 Aug 2026 00:51:08 -0500 Subject: [PATCH 846/857] [FROM-ML] drm/amdgpu: Enable PerfOpt IOMMU perf optimization when GPU in identity domain Enable PerfOpt via amd_iommu_enable_perfopt() when the GPU's iommu_perfopt module parameter is enabled (default 1) and the GPU resides in the identity domain. The identity domain means the GPU is already performing direct DMA with the IOMMU only enforcing IR/IW permission bits -- no GPA->SPA translations. amd_iommu_enable_perfopt() clears ATS, PRI, PASID and SVA for the device. This is safe in identity domain because DTE[I]=0 means the IOMMU already returns target abort for ATS requests from this peripheral and the GPU manages its own TLB. PerfOpt is a soft, optional latency optimization: failing to arm it (for example on an IOMMU that does not implement the feature, which returns -ENODEV) must not be fatal, so probe and resume warn and continue rather than aborting. PERF_OPT_EN is a per-IOMMU control shared by all devices behind that IOMMU; the IOMMU driver reference counts it so that on systems where multiple devices share one IOMMU, one GPU's teardown does not clear the bit while a peer still requires it. Arming PerfOpt trades IOMMU DMA containment for lower DMA latency. This is enabled by default for GPUs in the identity domain as a deliberate, documented policy and can be disabled with iommu_perfopt=0. The AMD IOMMU spec indicates this is only supported on integrated GPUs so check explicitly for AMD_IS_APU (which is set by amdgpu_device_ip_early_init()). PerfOpt is disabled during GPU init teardown and restored on resume. Link: https://patch.msgid.link/20260831055108.1893285-3-mario.limonciello@amd.com Signed-off-by: Mario Limonciello (cherry picked from commit d100bc864a6bb23cf99dbc6282edf9d123445bb1) --- drivers/gpu/drm/amd/amdgpu/amdgpu.h | 1 + drivers/gpu/drm/amd/amdgpu/amdgpu_device.c | 41 ++++++++++++++++++++++ drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c | 12 +++++++ 3 files changed, 54 insertions(+) diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu.h b/drivers/gpu/drm/amd/amdgpu/amdgpu.h index 8a7c89afc88e18..6765e550685d92 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu.h +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu.h @@ -156,6 +156,7 @@ struct amdgpu_watchdog_timer { * Modules parameters. */ extern int amdgpu_modeset; +extern int amdgpu_iommu_perfopt; extern unsigned int amdgpu_vram_limit; extern int amdgpu_vis_vram_limit; extern int amdgpu_gart_size; diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c index ea25d528155965..8813e72fec5e0d 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c @@ -33,6 +33,7 @@ #include #include #include +#include #include #include #include @@ -3758,6 +3759,17 @@ amdgpu_device_should_register_switcheroo(struct amdgpu_device *adev, bool px) apple_gmux_detect(NULL, NULL))); } +static inline bool amdgpu_device_identity(struct amdgpu_device *adev) +{ + struct pci_dev *pdev = adev->pdev; + struct iommu_domain *domain = iommu_get_domain_for_dev(&pdev->dev); + + if (!domain) + return false; + + return domain->type == IOMMU_DOMAIN_IDENTITY; +} + /** * amdgpu_device_init - initialize the driver * @@ -3965,6 +3977,18 @@ int amdgpu_device_init(struct amdgpu_device *adev, if (r) return r; + if (amdgpu_iommu_perfopt != 0 && + amdgpu_device_identity(adev) && + adev->flags & AMD_IS_APU) { + int perfopt_ret = amd_iommu_enable_perfopt(pdev); + + /* Optional optimization; a failure to arm it must not abort probe. */ + if (perfopt_ret) + dev_warn(adev->dev, + "Failed to enable IOMMU PerfOpt (%d); continuing without it\n", + perfopt_ret); + } + /* * No need to remove conflicting FBs for non-display class devices. * This prevents the sysfb from being freed accidently. @@ -4334,6 +4358,9 @@ void amdgpu_device_fini_hw(struct amdgpu_device *adev) amdgpu_gart_dummy_page_fini(adev); + if (amdgpu_iommu_perfopt != 0) + amd_iommu_disable_perfopt(adev->pdev); + if (pci_dev_is_disconnected(adev->pdev)) amdgpu_device_unmap_mmio(adev); @@ -4699,6 +4726,20 @@ int amdgpu_device_resume(struct drm_device *dev, bool notify_clients) if (dev->switch_power_state == DRM_SWITCH_POWER_OFF) return 0; + if (amdgpu_iommu_perfopt != 0 && amdgpu_device_identity(adev)) { + int perfopt_ret = amd_iommu_enable_perfopt(adev->pdev); + + /* + * Must not return on failure: a bare return would leak the + * SR-IOV VF exclusive-mode acquisition taken above (released + * via the exit: path). + */ + if (perfopt_ret) + dev_warn(adev->dev, + "Failed to enable IOMMU PerfOpt (%d); continuing without it\n", + perfopt_ret); + } + if (adev->in_s0ix) amdgpu_dpm_gfx_state_change(adev, sGpuChangeState_D0Entry); diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c index 53738b40c97f6d..82c2b6518ab1be 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c @@ -187,6 +187,7 @@ char *amdgpu_disable_cu; char *amdgpu_virtual_display; int amdgpu_enforce_isolation = -1; int amdgpu_modeset = -1; +int amdgpu_iommu_perfopt = 1; /* Specifies the default granularity for SVM, used in buffer * migration and restoration of backing memory when handling @@ -394,6 +395,17 @@ module_param_named(fw_load_type, amdgpu_fw_load_type, int, 0444); MODULE_PARM_DESC(aspm, "ASPM support (1 = enable, 0 = disable, -1 = auto)"); module_param_named(aspm, amdgpu_aspm, int, 0444); +/** + * DOC: iommu_perfopt (int) + * Control the AMD IOMMU PerfOpt DMA-latency optimization + * (0 = disable; 1 = enable on APU devices in identity domain). + * This arms the IOMMU PerfOpt control (IOMMU spec, MMIO Offset 016Ch, EFR PerfOptSup / PerfOptEn) + * Arming it disables ATS, PRI, PASID and SVA for the GPU and removes IOMMU DMA containment for it, + * trading isolation for lower DMA latency. + */ +MODULE_PARM_DESC(iommu_perfopt, "Control IOMMU PerfOpt DMA-latency optimization (1 = enable on APU devices in identity domain, 0 = disable)"); +module_param_named(iommu_perfopt, amdgpu_iommu_perfopt, int, 0444); + /** * DOC: runpm (int) * Override for runtime power management control for dGPUs. The amdgpu driver can dynamically power down From 507c8e7e3593ab9edc941c862a6d03e033f2aeeb Mon Sep 17 00:00:00 2001 From: Panz Dev Date: Tue, 18 Aug 2026 17:14:31 +0200 Subject: [PATCH 847/857] [FROM-ML] HID: asus: do not send keyboard init reports to touchpads Commit 0919db9f3583 ("HID: asus: always fully initialize devices") added a loop during asus_probe() to send keyboard feature report initializations (asus_kbd_init) to all ASUS HID devices. On ASUS laptops with I2C/HID touchpads (such as the ASUS E200HA), sending keyboard feature reports (FEATURE_KBD_REPORT_ID) to touchpad endpoints sends invalid feature requests to touchpad hardware, corrupting probe state and causing the touchpad to become unresponsive. Wrap the asus_report_id_init loop in an `if (!drvdata->tp)` check so keyboard feature initialization only runs for actual keyboards. Tested on ASUS E200HA (where touchpad functionality is fully restored) and ASUS VivoBook Flip 14 TP401MA (confirming zero regressions). Fixes: 0919db9f3583 ("HID: asus: always fully initialize devices") Cc: stable@vger.kernel.org Signed-off-by: Panz Dev (cherry picked from commit b3c5876b375ed84ccbefea7849a1b4fb11486866) --- drivers/hid/hid-asus.c | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 4f578446004a8e..43f7aaa7d06927 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1502,12 +1502,14 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) return ret; } - for (int r = 0; r < ARRAY_SIZE(asus_report_id_init); r++) { - if (asus_has_report_id(hdev, asus_report_id_init[r])) { - ret = asus_kbd_init(hdev, asus_report_id_init[r]); - if (ret < 0) - hid_warn(hdev, "Failed to initialize 0x%x: %d.\n", - asus_report_id_init[r], ret); + if (!drvdata->tp) { + for (int r = 0; r < ARRAY_SIZE(asus_report_id_init); r++) { + if (asus_has_report_id(hdev, asus_report_id_init[r])) { + ret = asus_kbd_init(hdev, asus_report_id_init[r]); + if (ret < 0) + hid_warn(hdev, "Failed to initialize 0x%x: %d.\n", + asus_report_id_init[r], ret); + } } } From 4846cca30b81784f73b37b853aed12c9e9b77bb6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mati=CC=81as=20Marti=CC=81nez?= Date: Mon, 24 Aug 2026 18:31:03 -0400 Subject: [PATCH 848/857] [FROM-ML] HID: ayaneo: Add AYANEO 3 detachable controller driver MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The AYANEO 3 handheld has a detachable controller with swappable modules ("Magic Modules"). The controller exposes three USB HID interfaces behind 1c4f:0002 (a generic SigmaMicro VID/PID, hence the DMI gate): a gamepad, a keyboard for the extra buttons, and a vendor interface accepting 65-byte commands. Add a driver for the vendor interface providing module identification (module_left/module_right sysfs attributes), software eject of the modules (eject sysfs attribute, blocking until the firmware confirms the release handshake), and RGB control of the joystick rings as a multicolor LED class device (":rgb:joystick_rings"; userspace such as InputPlumber matches the function suffix). The firmware's fixed breathing pattern is exposed through the hw_pattern trigger ABI. This complements the ayaneo-ec platform driver, which exposes module attach state and controller power. A full physical eject is performed by writing to eject and then cutting power through ayaneo-ec's controller_power attribute; that orchestration is deliberately left to userspace. The protocol was reverse engineered in the Handheld Daemon project by Antheas Kapenekakis. Tested on an AYANEO 3: module identification, RGB solid and breathing, a full eject/reinsert/repower cycle, and repeated driver unbinds under a concurrent brightness-write load. Signed-off-by: Matías Martínez Reviewed-by: Denis Benato (cherry picked from commit 4e23d3d4f894f4b62a413e3d67b90b438ce2167b) --- .../testing/sysfs-class-led-driver-hid-ayaneo | 15 + .../ABI/testing/sysfs-driver-hid-ayaneo | 36 ++ MAINTAINERS | 8 + drivers/hid/Kconfig | 14 + drivers/hid/Makefile | 1 + drivers/hid/hid-ayaneo.c | 546 ++++++++++++++++++ 6 files changed, 620 insertions(+) create mode 100644 Documentation/ABI/testing/sysfs-class-led-driver-hid-ayaneo create mode 100644 Documentation/ABI/testing/sysfs-driver-hid-ayaneo create mode 100644 drivers/hid/hid-ayaneo.c diff --git a/Documentation/ABI/testing/sysfs-class-led-driver-hid-ayaneo b/Documentation/ABI/testing/sysfs-class-led-driver-hid-ayaneo new file mode 100644 index 00000000000000..00f100dba74e12 --- /dev/null +++ b/Documentation/ABI/testing/sysfs-class-led-driver-hid-ayaneo @@ -0,0 +1,15 @@ +What: /sys/class/leds//hw_pattern +Date: August 2026 +KernelVersion: 7.3 +Contact: Matías Martínez +Description: + Specify a hardware pattern for the AYANEO 3 joystick + rings LED. The firmware supports a single breathing + pattern, pulsing the current colour at a fixed, + firmware-controlled period: + + "0 " + + Both delta_t values are accepted but ignored, as the + period is not configurable. must be + non-zero. Any other pattern is rejected. diff --git a/Documentation/ABI/testing/sysfs-driver-hid-ayaneo b/Documentation/ABI/testing/sysfs-driver-hid-ayaneo new file mode 100644 index 00000000000000..807c4fc9d768b1 --- /dev/null +++ b/Documentation/ABI/testing/sysfs-driver-hid-ayaneo @@ -0,0 +1,36 @@ +What: /sys/bus/hid/drivers/hid-ayaneo//module_left +What: /sys/bus/hid/drivers/hid-ayaneo//module_right +Date: August 2026 +KernelVersion: 7.3 +Contact: Matías Martínez +Description: + Reports the type of the module currently inserted in the + left/right slot of the AYANEO 3 detachable controller, as + the raw identifier reported by the controller firmware in + hexadecimal (e.g. "0x04"). Bits 0-5 encode the module + type, bit 6 indicates the module is inserted rotated. + + Reading these attributes queries the controller and can + take up to a second. + +What: /sys/bus/hid/drivers/hid-ayaneo//eject +Date: August 2026 +KernelVersion: 7.3 +Contact: Matías Martínez +Description: + Write-only. Writing "left", "right" or "both" asks the + controller firmware to release the corresponding + module(s). The write blocks until the firmware confirms + the release handshake (typically a few seconds). The + module is physically released once controller power is + subsequently cut through the ayaneo-ec platform driver's + controller_power attribute; that final step is left to + userspace. + +What: /sys/bus/hid/drivers/hid-ayaneo//reset +Date: August 2026 +KernelVersion: 7.3 +Contact: Matías Martínez +Description: + Write-only. Writing "1" asks the controller firmware to + perform a quick reset of the controller configuration. diff --git a/MAINTAINERS b/MAINTAINERS index 38612eca298776..a5d869a534bc30 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -4518,6 +4518,14 @@ F: Documentation/devicetree/bindings/spi/axiado,ax3000-spi.yaml F: drivers/spi/spi-axiado.c F: drivers/spi/spi-axiado.h +AYANEO 3 CONTROLLER HID DRIVER +M: Matías Martínez +L: linux-input@vger.kernel.org +S: Maintained +F: Documentation/ABI/testing/sysfs-class-led-driver-hid-ayaneo +F: Documentation/ABI/testing/sysfs-driver-hid-ayaneo +F: drivers/hid/hid-ayaneo.c + AYANEO PLATFORM EC DRIVER M: Antheas Kapenekakis L: platform-driver-x86@vger.kernel.org diff --git a/drivers/hid/Kconfig b/drivers/hid/Kconfig index a81bf51cbcf103..22fa55eb17685f 100644 --- a/drivers/hid/Kconfig +++ b/drivers/hid/Kconfig @@ -205,6 +205,20 @@ config HID_AUREAL help Support for Aureal Cy se W-01RN Remote Controller and other Aureal derived remotes. +config HID_AYANEO + tristate "AYANEO 3 detachable controller support" + depends on USB_HID + depends on DMI + depends on LEDS_CLASS_MULTICOLOR + help + Provides support for the detachable controller ("Magic Modules") + of the AYANEO 3 handheld: module identification, software eject + and RGB control of the joystick rings. Complements the ayaneo-ec + platform driver, which handles module attach state and controller + power. + + Say Y or M here if you have an AYANEO 3. + config HID_BELKIN tristate "Belkin Flip KVM and Wireless keyboard" help diff --git a/drivers/hid/Makefile b/drivers/hid/Makefile index 48a863b245eed3..60add9348a8b16 100644 --- a/drivers/hid/Makefile +++ b/drivers/hid/Makefile @@ -35,6 +35,7 @@ obj-$(CONFIG_HID_APPLETB_KBD) += hid-appletb-kbd.o obj-$(CONFIG_HID_CREATIVE_SB0540) += hid-creative-sb0540.o obj-$(CONFIG_HID_ASUS) += hid-asus.o obj-$(CONFIG_HID_AUREAL) += hid-aureal.o +obj-$(CONFIG_HID_AYANEO) += hid-ayaneo.o obj-$(CONFIG_HID_BELKIN) += hid-belkin.o obj-$(CONFIG_HID_BETOP_FF) += hid-betopff.o obj-$(CONFIG_HID_BIGBEN_FF) += hid-bigbenff.o diff --git a/drivers/hid/hid-ayaneo.c b/drivers/hid/hid-ayaneo.c new file mode 100644 index 00000000000000..fc133abf57216f --- /dev/null +++ b/drivers/hid/hid-ayaneo.c @@ -0,0 +1,546 @@ +// SPDX-License-Identifier: GPL-2.0+ +/* + * HID driver for the AYANEO 3 detachable controller ("Magic Modules"). + * + * The AYANEO 3 controller exposes three USB HID interfaces behind + * VID 0x1c4f PID 0x0002 (a generic SigmaMicro ID, hence the DMI gate): + * a gamepad, a keyboard for the extra buttons, and a vendor interface + * (application usage 0xff000001) accepting 65-byte commands. + * + * This driver binds the vendor interface and provides: + * - module identification (which module type is inserted on each side) + * - software eject of the left/right modules + * - RGB control of the joystick rings as a multicolor LED class device + * + * It complements the ayaneo-ec platform driver, which exposes module + * attach state and controller power. A full eject is: write to this + * driver's "eject" attribute, then power the controller off through + * ayaneo-ec's controller_power once the eject completes. + * + * The protocol was reverse engineered in the Handheld Daemon project by + * Antheas Kapenekakis. + * + * Command format (65 bytes, unnumbered report): + * [0] report id (0) + * [1:3] little-endian sum of bytes 7..64 + * [3] command + * [4] subcommand + * [5:] payload + * The device replies with a 64-byte report echoing the subcommand at + * byte 3. + * + * Copyright (C) 2026 Matías Martínez + */ + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#define AYA3_REPORT_SIZE 65 +#define AYA3_RESP_SIZE 64 +#define AYA3_CMD_TIMEOUT_MS 300 +#define AYA3_CMD_ATTEMPTS 3 + +/* Subcommands (byte 4); byte 3 is 0x00 except for the config command */ +#define AYA3_SUBCMD_CHECK 0x08 +#define AYA3_CMD_CONFIG 0x21 +#define AYA3_SUBCMD_CONFIG 0x09 + +/* CHECK response fields */ +#define AYA3_RESP_CMD 3 +#define AYA3_RESP_EJECT_STATUS 19 +#define AYA3_RESP_MODULE_LEFT 32 +#define AYA3_RESP_MODULE_RIGHT 33 +/* Bits that stay set in the eject status byte after an eject completes */ +#define AYA3_EJECT_DONE_MASK 0x11 + +/* Config command eject/reset field */ +#define AYA3_EJECT_LEFT 0x07 +#define AYA3_EJECT_RIGHT 0x70 +#define AYA3_RESET 0x88 + +/* Config command RGB modes */ +#define AYA3_RGB_SOLID 0x01 +#define AYA3_RGB_PULSE 0x02 +#define AYA3_RGB_OFF 0xff + +#define AYA3_VIBRATION_DEFAULT 0x02 /* medium */ + +struct aya3 { + struct hid_device *hdev; + /* DMA-safe command buffer; guarded by lock */ + u8 *xfer; + /* Serializes commands and cached-config access */ + struct mutex lock; + struct completion resp_done; + u8 resp[AYA3_RESP_SIZE]; + u8 resp_expect; + bool resp_pending; + + u8 rgb[3]; + bool pulse; + u8 vibration; + + struct led_classdev_mc mcled; + struct mc_subled subleds[3]; +}; + +static int aya3_send(struct aya3 *aya) +{ + int ret; + + ret = hid_hw_output_report(aya->hdev, aya->xfer, AYA3_REPORT_SIZE); + if (ret == -ENOSYS) + ret = hid_hw_raw_request(aya->hdev, aya->xfer[0], aya->xfer, + AYA3_REPORT_SIZE, HID_OUTPUT_REPORT, + HID_REQ_SET_REPORT); + if (ret < 0) + return ret; + return 0; +} + +/** + * aya3_cmd() - send the command in aya->xfer and wait for the reply + * @aya: driver data; @aya->xfer holds the fully built 65-byte command + * @resp: destination for the AYA3_RESP_SIZE-byte reply, or NULL to + * discard it + * + * The device echoes the subcommand byte of the command it is answering, + * which aya3_raw_event() uses to match replies. Unanswered commands are + * retried up to AYA3_CMD_ATTEMPTS times. + * + * Context: process context; the caller must hold @aya->lock, which + * protects @aya->xfer and the reply state. + * Return: 0 on success, -ETIMEDOUT if every attempt went unanswered, or + * a negative errno if sending failed. + */ +static int aya3_cmd(struct aya3 *aya, u8 *resp) +{ + int attempt, ret; + + lockdep_assert_held(&aya->lock); + + for (attempt = 0; attempt < AYA3_CMD_ATTEMPTS; attempt++) { + reinit_completion(&aya->resp_done); + aya->resp_expect = aya->xfer[4]; + WRITE_ONCE(aya->resp_pending, true); + + ret = aya3_send(aya); + if (ret) { + WRITE_ONCE(aya->resp_pending, false); + return ret; + } + + if (wait_for_completion_timeout(&aya->resp_done, + msecs_to_jiffies(AYA3_CMD_TIMEOUT_MS))) { + if (resp) + memcpy(resp, aya->resp, AYA3_RESP_SIZE); + return 0; + } + } + WRITE_ONCE(aya->resp_pending, false); + return -ETIMEDOUT; +} + +static void aya3_checksum(u8 *buf) +{ + u16 sum = 0; + int i; + + for (i = 7; i < AYA3_REPORT_SIZE; i++) + sum += buf[i]; + put_unaligned_le16(sum, buf + 1); +} + +static int aya3_check(struct aya3 *aya, u8 *resp) +{ + memset(aya->xfer, 0, AYA3_REPORT_SIZE); + aya->xfer[4] = AYA3_SUBCMD_CHECK; + return aya3_cmd(aya, resp); +} + +/* + * The config command sets everything at once: RGB for both rings, + * vibration strength and the eject/reset field. The command can also + * carry joystick sensitivity; those bytes are left zero so the + * firmware setting is not clobbered on every RGB update. + */ +static int aya3_send_config(struct aya3 *aya, u8 eject) +{ + static const u8 template[AYA3_REPORT_SIZE] = { + [3] = AYA3_CMD_CONFIG, + [4] = AYA3_SUBCMD_CONFIG, + [32] = 0x01, + }; + u8 *buf = aya->xfer; + u8 mode = AYA3_RGB_OFF; + + if (aya->rgb[0] || aya->rgb[1] || aya->rgb[2]) + mode = aya->pulse ? AYA3_RGB_PULSE : AYA3_RGB_SOLID; + + memcpy(buf, template, AYA3_REPORT_SIZE); + /* Right ring, then left ring: mode, R, G, B */ + buf[8] = mode; + memcpy(buf + 9, aya->rgb, 3); + buf[12] = mode; + memcpy(buf + 13, aya->rgb, 3); + buf[20] = eject; + buf[24] = aya->vibration << 4; + aya3_checksum(buf); + + return aya3_cmd(aya, NULL); +} + +static int aya3_raw_event(struct hid_device *hdev, struct hid_report *report, + u8 *data, int size) +{ + struct aya3 *aya = hid_get_drvdata(hdev); + + if (!READ_ONCE(aya->resp_pending) || size < AYA3_RESP_SIZE) + return 0; + /* + * Replies carry no sequence number, only the subcommand echo. A + * late reply to a timed-out command can thus complete a newer + * command with the same subcommand; such replies are snapshots + * of the same query milliseconds apart, so this is harmless. + * Replies to a different subcommand are dropped here. + */ + if (data[AYA3_RESP_CMD] != aya->resp_expect) + return 0; + + memcpy(aya->resp, data, AYA3_RESP_SIZE); + WRITE_ONCE(aya->resp_pending, false); + complete(&aya->resp_done); + return 0; +} + +static ssize_t aya3_module_show(struct device *dev, char *buf, int offset) +{ + struct aya3 *aya = dev_get_drvdata(dev); + u8 resp[AYA3_RESP_SIZE]; + int ret; + + ret = mutex_lock_interruptible(&aya->lock); + if (ret) + return ret; + ret = aya3_check(aya, resp); + mutex_unlock(&aya->lock); + if (ret) + return ret; + + return sysfs_emit(buf, "0x%02x\n", resp[offset]); +} + +static ssize_t module_left_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + return aya3_module_show(dev, buf, AYA3_RESP_MODULE_LEFT); +} +static DEVICE_ATTR_RO(module_left); + +static ssize_t module_right_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + return aya3_module_show(dev, buf, AYA3_RESP_MODULE_RIGHT); +} +static DEVICE_ATTR_RO(module_right); + +static ssize_t eject_store(struct device *dev, struct device_attribute *attr, + const char *buf, size_t count) +{ + struct aya3 *aya = dev_get_drvdata(dev); + u8 resp[AYA3_RESP_SIZE]; + u8 eject; + int ret, err, i; + + if (sysfs_streq(buf, "left")) + eject = AYA3_EJECT_LEFT; + else if (sysfs_streq(buf, "right")) + eject = AYA3_EJECT_RIGHT; + else if (sysfs_streq(buf, "both")) + eject = AYA3_EJECT_LEFT | AYA3_EJECT_RIGHT; + else + return -EINVAL; + + ret = mutex_lock_interruptible(&aya->lock); + if (ret) + return ret; + + ret = aya3_send_config(aya, eject); + if (ret) + goto out; + + /* + * Wait for the firmware to report the eject as done. Userspace + * must then cut power through ayaneo-ec's controller_power for + * the module to be physically released. + */ + ret = -ETIMEDOUT; + for (i = 0; i < 20; i++) { + msleep(400); + err = aya3_check(aya, resp); + if (err == -ETIMEDOUT) + continue; /* busy mid-eject, keep polling */ + if (err) { + ret = err; + break; + } + if (!(resp[AYA3_RESP_EJECT_STATUS] & ~AYA3_EJECT_DONE_MASK)) { + ret = 0; + break; + } + } +out: + mutex_unlock(&aya->lock); + return ret ? ret : count; +} +static DEVICE_ATTR_WO(eject); + +static ssize_t reset_store(struct device *dev, struct device_attribute *attr, + const char *buf, size_t count) +{ + struct aya3 *aya = dev_get_drvdata(dev); + bool value; + int ret; + + ret = kstrtobool(buf, &value); + if (ret) + return ret; + if (!value) + return count; + + ret = mutex_lock_interruptible(&aya->lock); + if (ret) + return ret; + ret = aya3_send_config(aya, AYA3_RESET); + if (!ret) { + msleep(500); + ret = aya3_send_config(aya, 0); + } + mutex_unlock(&aya->lock); + return ret ? ret : count; +} +static DEVICE_ATTR_WO(reset); + +static struct attribute *aya3_attrs[] = { + &dev_attr_module_left.attr, + &dev_attr_module_right.attr, + &dev_attr_eject.attr, + &dev_attr_reset.attr, + NULL +}; +ATTRIBUTE_GROUPS(aya3); + +static int aya3_led_set(struct led_classdev *cdev, enum led_brightness value) +{ + struct led_classdev_mc *mc = lcdev_to_mccdev(cdev); + struct aya3 *aya = container_of(mc, struct aya3, mcled); + int ret, i; + + ret = mutex_lock_interruptible(&aya->lock); + if (ret) + return ret; + + led_mc_calc_color_components(mc, value); + for (i = 0; i < 3; i++) + aya->rgb[i] = min_t(unsigned int, aya->subleds[i].brightness, 255); + + ret = aya3_send_config(aya, 0); + if (ret) + hid_err(aya->hdev, "failed to update RGB config: %d\n", ret); + mutex_unlock(&aya->lock); + return ret; +} + +/* + * The firmware offers one fixed breathing pattern, pulsing the current + * colour at a period it controls. Expose it through the hw_pattern + * trigger ABI as the two-step pattern "0 "; the + * delta_t values and the repeat count are accepted but not tunable + * (the firmware always repeats indefinitely). + */ +static int aya3_pattern_set(struct led_classdev *cdev, + struct led_pattern *pattern, u32 len, int repeat) +{ + struct led_classdev_mc *mc = lcdev_to_mccdev(cdev); + struct aya3 *aya = container_of(mc, struct aya3, mcled); + int ret; + + if (len != 2 || pattern[0].brightness || !pattern[1].brightness) + return -EINVAL; + + ret = mutex_lock_interruptible(&aya->lock); + if (ret) + return ret; + aya->pulse = true; + ret = aya3_send_config(aya, 0); + mutex_unlock(&aya->lock); + return ret; +} + +static int aya3_pattern_clear(struct led_classdev *cdev) +{ + struct led_classdev_mc *mc = lcdev_to_mccdev(cdev); + struct aya3 *aya = container_of(mc, struct aya3, mcled); + int ret; + + ret = mutex_lock_interruptible(&aya->lock); + if (ret) + return ret; + aya->pulse = false; + ret = aya3_send_config(aya, 0); + mutex_unlock(&aya->lock); + return ret; +} + +static int aya3_register_led(struct aya3 *aya) +{ + struct led_classdev *cdev = &aya->mcled.led_cdev; + + aya->subleds[0].color_index = LED_COLOR_ID_RED; + aya->subleds[1].color_index = LED_COLOR_ID_GREEN; + aya->subleds[2].color_index = LED_COLOR_ID_BLUE; + aya->mcled.subled_info = aya->subleds; + aya->mcled.num_colors = 3; + + cdev->name = devm_kasprintf(&aya->hdev->dev, GFP_KERNEL, + "%s:rgb:joystick_rings", + dev_name(&aya->hdev->dev)); + if (!cdev->name) + return -ENOMEM; + cdev->brightness = 0; + cdev->max_brightness = 255; + cdev->brightness_set_blocking = aya3_led_set; + cdev->pattern_set = aya3_pattern_set; + cdev->pattern_clear = aya3_pattern_clear; + + /* + * Not devm: the LED must be unregistered before hid_hw_stop() in + * remove, or a concurrent brightness write could reach a torn + * down transport. + */ + return led_classdev_multicolor_register(&aya->hdev->dev, + &aya->mcled); +} + +static const struct dmi_system_id aya3_dmi_table[] = { + { + .matches = { + DMI_MATCH(DMI_BOARD_VENDOR, "AYANEO"), + DMI_MATCH(DMI_BOARD_NAME, "AYANEO 3"), + }, + }, + {} +}; + +static int aya3_probe(struct hid_device *hdev, const struct hid_device_id *id) +{ + struct aya3 *aya; + int ret; + + /* The VID/PID is a generic SigmaMicro ID; bind on AYANEO 3 only */ + if (!dmi_check_system(aya3_dmi_table)) + return -ENODEV; + + if (!hid_is_usb(hdev)) + return -ENODEV; + + ret = hid_parse(hdev); + if (ret) + return ret; + + /* Bind only the vendor interface, not the gamepad/keyboard ones */ + if (!hdev->maxcollection || + hdev->collection->usage != (HID_UP_MSVENDOR | 0x0001)) + return -ENODEV; + + aya = devm_kzalloc(&hdev->dev, sizeof(*aya), GFP_KERNEL); + if (!aya) + return -ENOMEM; + + aya->xfer = devm_kzalloc(&hdev->dev, AYA3_REPORT_SIZE, GFP_KERNEL); + if (!aya->xfer) + return -ENOMEM; + + aya->hdev = hdev; + aya->vibration = AYA3_VIBRATION_DEFAULT; + init_completion(&aya->resp_done); + ret = devm_mutex_init(&hdev->dev, &aya->lock); + if (ret) + return ret; + hid_set_drvdata(hdev, aya); + + ret = hid_hw_start(hdev, HID_CONNECT_HIDRAW); + if (ret) + return ret; + + ret = hid_hw_open(hdev); + if (ret) + goto err_stop; + + /* Input reports are not delivered during probe by default */ + hid_device_io_start(hdev); + + scoped_guard(mutex, &aya->lock) + ret = aya3_check(aya, NULL); + if (ret) + hid_warn(hdev, "controller did not answer status check: %d\n", + ret); + + ret = aya3_register_led(aya); + if (ret) + goto err_close; + + return 0; + +err_close: + hid_hw_close(hdev); +err_stop: + hid_hw_stop(hdev); + return ret; +} + +static void aya3_remove(struct hid_device *hdev) +{ + struct aya3 *aya = hid_get_drvdata(hdev); + + led_classdev_multicolor_unregister(&aya->mcled); + /* + * A brightness store racing with the unregister can requeue + * set_brightness_work after the flush inside + * led_classdev_unregister() runs but before the sysfs node is + * removed. Flush again now that nothing can requeue it, while + * the transport is still up. + */ + flush_work(&aya->mcled.led_cdev.set_brightness_work); + hid_hw_close(hdev); + hid_hw_stop(hdev); +} + +static const struct hid_device_id aya3_devices[] = { + { HID_USB_DEVICE(0x1c4f, 0x0002) }, + {} +}; +MODULE_DEVICE_TABLE(hid, aya3_devices); + +static struct hid_driver aya3_driver = { + .name = "hid-ayaneo", + .id_table = aya3_devices, + .probe = aya3_probe, + .remove = aya3_remove, + .raw_event = aya3_raw_event, + .driver = { + .dev_groups = aya3_groups, + }, +}; +module_hid_driver(aya3_driver); + +MODULE_AUTHOR("Matías Martínez "); +MODULE_DESCRIPTION("AYANEO 3 detachable controller driver"); +MODULE_LICENSE("GPL"); From c4b54e33756afc83a51bb09fbd987711ac7fdd65 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mati=CC=81as=20Marti=CC=81nez?= Date: Mon, 24 Aug 2026 15:20:21 -0400 Subject: [PATCH 849/857] [NOT-FOR-UPSTREAM] github: enable CONFIG_HID_AYANEO in the CI fragment MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Lets the build workflow compile the new driver. The real OGC config change is OpenGamingCollective/kernel-packages#35, which lands once the driver merges. Signed-off-by: Matías Martínez (cherry picked from commit 370d86f63b7f92593e9bec849e8b2e1ed51e5c37) --- .github/packaging/config.fragment | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/.github/packaging/config.fragment b/.github/packaging/config.fragment index 8d455d78cc2417..3611912b3b1663 100644 --- a/.github/packaging/config.fragment +++ b/.github/packaging/config.fragment @@ -121,4 +121,9 @@ CONFIG_LOCALVERSION="" # --- Staging drivers ---------------------------------------------------------- # All OGC handheld drivers live in mainline trees (drivers/hid, # drivers/platform/x86), not staging. -# CONFIG_STAGING is not set \ No newline at end of file +# CONFIG_STAGING is not set + +# --- Drivers under review in this repo --------------------------------------- +# hid-ayaneo (PR #3); mirrors OpenGamingCollective/kernel-packages#35, which +# lands the option in the real OGC config once the driver merges. +CONFIG_HID_AYANEO=m From 743e77cb2b2a261fd397f0ab446b4457d046ff1b Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Fri, 4 Sep 2026 13:04:33 +0000 Subject: [PATCH 850/857] [NOT-FOR-UPSTREAM] ogc: linux-unstable: fix sync (cherry picked from commit af0e02d989c8d96ad7d4b7711422523a57b7550d) --- .github/workflows/sync-linux-next.yml | 38 ++++++--------------------- 1 file changed, 8 insertions(+), 30 deletions(-) diff --git a/.github/workflows/sync-linux-next.yml b/.github/workflows/sync-linux-next.yml index a83fec1d5c146e..45b236c3f24cb6 100644 --- a/.github/workflows/sync-linux-next.yml +++ b/.github/workflows/sync-linux-next.yml @@ -44,38 +44,16 @@ jobs: # ------------------------------------------------------------------ # Find the upstream snapshot that master's fork commits sit on. # - # linux-next is rebuilt from scratch every day: commits merged into - # yesterday's tree are regularly dropped from today's history - # (subsystem trees get rebased), so: - # - "upstream/master..master" contains hundreds of unrelated - # upstream commits, not just our fork commits, and - # - a hardcoded marker hash gets orphaned by the very replay this - # workflow performs (run #2 then dies with "unknown revision", - # which is how this workflow broke in the first place). - # - # Anchor: walk first-parent history from master and stop at the - # second consecutive commit whose subject does not start with - # "ogc:" (the reserved prefix of every fork commit; cherry-pick - # preserves subjects, so this survives replays and force-pushes). - # The first of those two commits is the upstream snapshot the fork - # was built on. + # Anchor: the empty marker commit "ogc: linux-unstable: first + # commit" is always the first fork commit ever made. Its parent is + # therefore the upstream snapshot the fork was built on. # ------------------------------------------------------------------ - BASE="" - NONOGC="" - while read -r c; do - case "$(git show -s --format=%s "$c")" in - ogc:*) NONOGC="" ;; - *) - if [ -n "$NONOGC" ]; then BASE="$NONOGC"; break; fi - NONOGC="$c" - ;; - esac - done < <(git rev-list --first-parent master) - - if [ -z "$BASE" ]; then - echo "::error::Could not determine the upstream base of the fork; refusing to touch master." + MARKER="$(git log --format="%H" --grep="^ogc: linux-unstable: first commit$" master)" + if [ -z "$MARKER" ]; then + echo "::error::Could not find the 'ogc: linux-unstable: first commit' marker; refusing to touch master." exit 1 fi + BASE="$(git rev-parse "${MARKER}^")" echo "Upstream base: $(git show -s --oneline "$BASE")" # If master already sits on the current upstream tip there is @@ -134,4 +112,4 @@ jobs: done echo "Done. Pushing updated master." - git push --force-with-lease origin master \ No newline at end of file + git push --force-with-lease origin master From cc0de0fb6380e475682838052883e8566247681b Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Fri, 4 Sep 2026 14:16:53 +0000 Subject: [PATCH 851/857] [NOT-FOR-UPSTREAM] ogc: linux-unstable: test boot --- .github/packaging/boot-smoke-test.sh | 234 +++++++++++++++++++++++++++ .github/workflows/build-kernel.yml | 64 ++++++++ 2 files changed, 298 insertions(+) create mode 100755 .github/packaging/boot-smoke-test.sh diff --git a/.github/packaging/boot-smoke-test.sh b/.github/packaging/boot-smoke-test.sh new file mode 100755 index 00000000000000..021e8e596998f3 --- /dev/null +++ b/.github/packaging/boot-smoke-test.sh @@ -0,0 +1,234 @@ +#!/usr/bin/env bash +# SPDX-License-Identifier: GPL-2.0-only +# +# QEMU boot smoke test for the linux-unstable-ogc release builds. +# +# Boots a freshly built, *packaged* kernel image under QEMU with a tiny +# hand-rolled initramfs whose only job is to prove the kernel can +# decompress, run its initcalls and hand control to userspace: the +# embedded PID 1 prints a BOOT_OK marker on the serial console and then +# powers the VM off, which makes QEMU exit by itself (a successful boot +# never has to wait for the timeout). +# +# A boot matrix is exercised, because handhelds/gaming devices ship all +# sorts of firmware: +# +# bios-pc SeaBIOS (BIOS) + i440fx "pc" machine, kernel loaded by the +# firmware's Linux loader (fw_cfg). +# uefi-q35 OVMF (UEFI) + q35 machine, kernel booted through the EFI +# stub (this is what Steam Deck / ROG Ally / Legion Go / +# Ayaneo class devices actually boot from). +# +# Every boot also attaches a spread of boot-relevant PCI devices (ICH9 +# AHCI comes with q35, PIIX IDE with pc, plus NVMe, virtio-blk, +# virtio-scsi, e1000e and xHCI USB) so as many storage/USB/net boot +# drivers as the built kernel contains get enumerated and probed. +# +# Each distro job (Arch/Fedora/Debian) runs this against the exact image +# it is about to upload, before uploading it. The release job only runs +# when every distro job passes, so an unbootable kernel can never be +# published. +# +# Usage: boot-smoke-test.sh +# Env: KREL expected kernel release string; when set, the boot banner +# "Linux version " must appear on every console log. +# BOOT_SMOKE_TIMEOUT override the per-boot timeout in seconds +# (default: 300 under KVM, 1200 under TCG). +# Requires (installed by the calling CI job): qemu-system-x86_64, cpio, +# gzip, a C compiler (gcc or clang), an OVMF build for the UEFI +# leg (packages: ovmf / edk2-ovmf), coreutils (timeout, find). + +set -euo pipefail + +die() { echo "::error::$*" >&2; exit 1; } + +IMAGE="${1:?usage: boot-smoke-test.sh }" +[ -s "$IMAGE" ] || die "kernel image '$IMAGE' not found or empty" + +for tool in qemu-system-x86_64 cpio gzip timeout sha256sum find truncate; do + command -v "$tool" >/dev/null || die "required tool '$tool' not installed" +done +CC_BIN="$(command -v gcc || command -v clang || true)" +[ -n "$CC_BIN" ] || die "no C compiler (gcc or clang) found" + +WORK="$(mktemp -d /tmp/boot-smoke.XXXXXX)" +trap 'rm -rf "$WORK"' EXIT + +# --------------------------------------------------------------------------- +# PID 1 for the test initramfs: freestanding, no libc, raw x86_64 syscalls, +# so it compiles with gcc or clang on any distro without extra static-libc +# packages (the Fedora build env has no glibc-static, for example). +# --------------------------------------------------------------------------- +cat > "$WORK/init.c" <<'EOF' +/* + * Tiny freestanding PID 1 for the QEMU boot smoke test: print BOOT_OK on + * the serial console, then power the VM off so QEMU exits on its own. + */ +static long sys3(long nr, long a, long b, long c) +{ + long ret; + + __asm__ volatile ("syscall" + : "=a"(ret) + : "a"(nr), "D"(a), "S"(b), "d"(c) + : "rcx", "r11", "memory"); + return ret; +} + +void _start(void) +{ + static const char msg[] = + "BOOT_OK: linux-unstable-ogc booted to userspace init\n"; + long fd; + + /* CONFIG_DEVTMPFS_MOUNT does not apply to initramfs, so mount + * devtmpfs ourselves to get /dev/ttyS0. */ + sys3(83, (long)"/dev", 0755, 0); /* mkdir */ + sys3(165, (long)"devtmpfs", (long)"/dev", (long)"devtmpfs"); /* mount */ + + /* Announce success on the serial port and on stdio (fd 1 exists + * only when the kernel could open /dev/console from the initramfs). */ + fd = sys3(2, (long)"/dev/ttyS0", 1, 0); /* open */ + if (fd >= 0) + sys3(1, fd, (long)msg, sizeof(msg) - 1); /* write */ + sys3(1, 1, (long)msg, sizeof(msg) - 1); /* write */ + + /* reboot(LINUX_REBOOT_CMD_POWER_OFF) -> ACPI S5 -> QEMU exits. */ + sys3(169, 0xfee1deadL, 672274793L, 0x4321fedcL); + for (;;) + sys3(34, 0, 0, 0); /* pause */ +} +EOF + +mkdir -p "$WORK/initramfs/dev" +"$CC_BIN" -Os -g0 -static -no-pie -nostdlib -ffreestanding \ + -fno-stack-protector -fno-asynchronous-unwind-tables \ + -fno-unwind-tables -Wl,-z,noexecstack \ + -o "$WORK/initramfs/init" "$WORK/init.c" +# /dev/console lets the kernel wire init's stdio to the console; creating it +# needs mknod privileges and may fail for unprivileged callers. Harmless: the +# init above then opens /dev/ttyS0 itself after mounting devtmpfs. +mknod -m 600 "$WORK/initramfs/dev/console" c 5 1 2>/dev/null || true + +(cd "$WORK/initramfs" && find . -print0 | cpio --null -o -H newc --quiet) \ + | gzip -1 > "$WORK/initrd.img" + +# Use KVM when the runner exposes it, software emulation (TCG) otherwise. +# TCG boots a distro-config kernel in a couple of minutes; the generous +# timeout also covers hangs, which are exactly what this test must catch. +if [ -w /dev/kvm ]; then + ACCEL="kvm" + ACCEL_ARGS=(-accel kvm -cpu host) + TMO=300 +else + ACCEL="tcg" + ACCEL_ARGS=(-accel tcg -cpu max) + TMO=1200 +fi + +# --------------------------------------------------------------------------- +# Locate an OVMF build for the UEFI leg. Layouts, by distro: +# Debian: /usr/share/OVMF/OVMF_CODE(_4M).fd (package: ovmf) +# Fedora: /usr/share/edk2/ovmf/OVMF_CODE.fd (package: edk2-ovmf) +# Arch: /usr/share/edk2/x64/OVMF_CODE.4m.fd (package: edk2-ovmf) +# qemu: /usr/share/qemu/edk2-x86_64-code.fd (qemu-system-data) +# CODE and VARS must be the same flash size, hence the paired candidates. +# --------------------------------------------------------------------------- +OVMF_PAIR="" +for pair in \ + "/usr/share/OVMF/OVMF_CODE_4M.fd:/usr/share/OVMF/OVMF_VARS_4M.fd" \ + "/usr/share/OVMF/OVMF_CODE.fd:/usr/share/OVMF/OVMF_VARS.fd" \ + "/usr/share/edk2/x64/OVMF_CODE.4m.fd:/usr/share/edk2/x64/OVMF_VARS.4m.fd" \ + "/usr/share/edk2/x64/OVMF_CODE.fd:/usr/share/edk2/x64/OVMF_VARS.fd" \ + "/usr/share/edk2/ovmf/OVMF_CODE.fd:/usr/share/edk2/ovmf/OVMF_VARS.fd" \ + "/usr/share/ovmf/x64/OVMF_CODE.fd:/usr/share/ovmf/x64/OVMF_VARS.fd" \ + "/usr/share/qemu/edk2-x86_64-code.fd:/usr/share/qemu/edk2-x86_64-vars.fd" \ + ; do + if [ -r "${pair%%:*}" ] && [ -r "${pair##*:}" ]; then + OVMF_PAIR="$pair" + break + fi +done +[ -n "$OVMF_PAIR" ] || die "no OVMF (UEFI) firmware found; install the 'ovmf' (Debian) or 'edk2-ovmf' (Arch/Fedora) package" +OVMF_CODE="${OVMF_PAIR%%:*}" +cp "${OVMF_PAIR##*:}" "$WORK/OVMF_VARS.fd" # pflash needs a writable copy +echo "UEFI firmware: $OVMF_CODE" + +# Blank disks so the storage controllers have something to enumerate. +for d in nvme virtio scsi; do + truncate -s 64M "$WORK/disk-$d.img" +done + +# Device spread common to every boot: covers the storage/USB/net drivers a +# gaming kernel is most likely to boot from. Attached whether or not the +# kernel contains them; drivers that are modules are simply not probed, +# which costs nothing. +DRIVERS_ARGS=( + -drive if=none,id=dsk-nvme,format=raw,file="$WORK/disk-nvme.img" + -device nvme,drive=dsk-nvme,serial=ogcsmoke + -drive if=none,id=dsk-virtio,format=raw,file="$WORK/disk-virtio.img" + -device virtio-blk-pci,drive=dsk-virtio + -drive if=none,id=dsk-scsi,format=raw,file="$WORK/disk-scsi.img" + -device virtio-scsi-pci,id=scsi0 + -device scsi-hd,drive=dsk-scsi + -device e1000e,netdev=net0 -netdev user,id=net0,restrict=on + -device qemu-xhci -device usb-tablet +) + +# run_boot — boot, wait for QEMU to exit or the +# timeout to fire, then fail hard if the marker (or the expected release +# banner) is missing from the serial console log. +run_boot() { + local name="$1"; shift + local serial="$WORK/serial-$name.log" + : > "$serial" + + echo "[$name] Booting $(basename "$IMAGE") (sha256 $(sha256sum "$IMAGE" | cut -c1-16)...) with QEMU ($ACCEL, timeout ${TMO}s)" + timeout "$TMO" qemu-system-x86_64 "${ACCEL_ARGS[@]}" "$@" \ + "${DRIVERS_ARGS[@]}" \ + -m 2048 -smp 2 -nodefaults \ + -display none -monitor none -no-reboot \ + -serial "file:$serial" \ + -kernel "$IMAGE" -initrd "$WORK/initrd.img" \ + -append "console=ttyS0,115200n8 rdinit=/init panic=-1 nokaslr" \ + || true + # A panic with panic=-1 reboots instantly and -no-reboot makes QEMU + # exit, so both "qemu exited by itself" and "timeout killed it" end up + # in the marker check below. + + if ! grep -q "BOOT_OK" "$serial"; then + echo "::error::QEMU boot smoke test FAILED in the '$name' configuration: no BOOT_OK marker on the serial console (accel=$ACCEL, timeout=${TMO}s). The kernel is unbootable." + echo "----- last 250 lines of the $name guest serial console -----" + tail -n 250 "$serial" || true + echo "-------------------------------------------------------------" + exit 1 + fi + if [ -n "${KREL:-}" ] && ! grep -qF "Linux version ${KREL} " "$serial"; then + echo "::error::QEMU boot smoke test FAILED in the '$name' configuration: the booted kernel banner does not advertise release '${KREL}'." + grep -m1 "^Linux version" "$serial" || true + exit 1 + fi + echo "[$name] Boot smoke test PASSED: kernel booted to userspace init and powered off. Serial console tail:" + tail -n 10 "$serial" || true +} + +# 1. Classic BIOS boot: SeaBIOS firmware, i440fx (PIIX IDE built in). +run_boot bios-pc -machine pc + +# 2. UEFI boot: OVMF firmware, q35 (ICH9 AHCI + PCIe built in), kernel +# launched through the EFI stub — the path handhelds actually use. +run_boot uefi-q35 \ + -machine q35 \ + -drive if=pflash,format=raw,readonly=on,file="$OVMF_CODE" \ + -drive if=pflash,format=raw,file="$WORK/OVMF_VARS.fd" + +if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + { + echo '### QEMU boot smoke test' + echo + echo "- Image: $(basename "$IMAGE") (sha256 $(sha256sum "$IMAGE" | cut -c1-12)...)" + echo "- Accelerator: ${ACCEL}, timeout ${TMO}s per boot" + echo "- Matrix: bios-pc (SeaBIOS) and uefi-q35 (OVMF EFI stub), NVMe + virtio-blk + virtio-scsi + e1000e + xHCI attached" + echo '- Result: BOOT_OK in both configurations - kernel decompressed, booted and reached userspace init' + } >> "$GITHUB_STEP_SUMMARY" +fi diff --git a/.github/workflows/build-kernel.yml b/.github/workflows/build-kernel.yml index 623cd5e89c2561..3274de60d762f2 100644 --- a/.github/workflows/build-kernel.yml +++ b/.github/workflows/build-kernel.yml @@ -72,6 +72,7 @@ jobs: cp .github/packaging/merge-fragments.sh "${STAGE}/merge-fragments.sh" cp .github/packaging/config.fragment "${STAGE}/config.fragment" cp .github/packaging/fedora/kernel.spec "${STAGE}/kernel.spec" + cp .github/packaging/boot-smoke-test.sh "${STAGE}/boot-smoke-test.sh" # Source tarball: contents of the repo as ./linux/, without VCS data. tar --exclude-vcs -I 'gzip -1' -cf "${STAGE}/linux.tar.gz" \ @@ -197,6 +198,24 @@ jobs: # Ship the exact .config used for the build as well cp -v "src/linux/.config" "${DISTDIR}/config-arch-${KREL}" + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + pacman -S --needed --noconfirm qemu-system-x86 edk2-ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the package. This + # step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + PKG="$(find /tmp/dist -name 'linux-unstable-ogc-*.pkg.tar.zst' ! -name '*-headers-*' | head -n1)" + test -n "$PKG" + tar --zstd -xf "$PKG" -C "$WORK" "usr/lib/modules/${KREL}/vmlinuz" + test -s "$WORK/usr/lib/modules/${KREL}/vmlinuz" + bash /tmp/stage/boot-smoke-test.sh "$WORK/usr/lib/modules/${KREL}/vmlinuz" + - name: Clean build tree (free disk before cache save) run: rm -rf "${PKGDIR}/src" "${PKGDIR}/pkg" ; df -h / @@ -348,6 +367,27 @@ jobs: cp -v /tmp/stage/config "${DISTDIR}/config-fedora-${KREL}" ls -lh "${DISTDIR}" + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + dnf -y install qemu-system-x86 edk2-ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the kernel-core RPM. + # This step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + RPM="$(find /tmp/dist -name 'kernel-unstable-ogc-core-*.rpm' | head -n1)" + test -n "$RPM" + # rpm cpio entries are stored with a "./" prefix (same trick as + # the config extraction above). + rpm2cpio "$RPM" | cpio -idm --quiet -D "$WORK" "./lib/modules/${KREL}/vmlinuz" || true + IMG="$WORK/lib/modules/${KREL}/vmlinuz" + test -s "$IMG" || { echo "::error::vmlinuz not found inside ${RPM}"; exit 1; } + bash /tmp/stage/boot-smoke-test.sh "$IMG" + - name: Clean build tree (free disk before cache save) run: rm -rf /tmp/rpmbuild/BUILD /tmp/rpmbuild/BUILDROOT ; df -h / @@ -588,6 +628,28 @@ jobs: # replace the distribution's own linux-libc-dev package. ls -lh "${DISTDIR}" + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + apt-get update -qq + # qemu boots the kernel (OVMF provides the UEFI leg); cpio builds + # the test initramfs. + apt-get install -y --no-install-recommends qemu-system-x86 cpio ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the linux-image .deb. + # This step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + DEB="$(find /tmp/dist -name "linux-image-${KREL}_*.deb" | head -n1)" + test -n "$DEB" + dpkg-deb -x "$DEB" "$WORK" + IMG="$(find "$WORK" -name 'vmlinuz-*' -type f | head -n1)" + test -n "$IMG" + bash /tmp/stage/boot-smoke-test.sh "$IMG" + - name: Clean build tree (free disk before cache save) run: rm -rf /build ; df -h / @@ -653,6 +715,7 @@ jobs: echo "- Upstream version: ${KVER}${NEXT}" echo "- Base configs: Arch \`linux-headers\` + Fedora \`kernel-core\`, plus [OGC kernel-packages fragments](${KERNEL_PACKAGES_RAW}/config)" echo "- Compiler: clang / LLVM=1 (with ccache)" + echo "- Boot smoke test: every packaged kernel (Arch, Fedora, Debian) was booted in QEMU and reached userspace init before publishing" echo echo "### Artifacts" echo @@ -718,4 +781,5 @@ jobs: echo echo "- Kernel release: \`${KREL}\`" echo "- Release: https://github.com/${GITHUB_REPOSITORY}/releases/tag/${TAG}" + echo "- QEMU boot smoke test: passed (Arch, Fedora and Debian kernel images booted to userspace init)" } >> "$GITHUB_STEP_SUMMARY" \ No newline at end of file From 71e4bcf252279ac90d41281a98e4b806f0a2737a Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Fri, 4 Sep 2026 14:40:23 +0000 Subject: [PATCH 852/857] [NOT-FOR-UPSTREAM] ogc: linux-unstable: split build-kernel.yml into reusable workflows build-kernel.yml had grown to ~785 lines (four container jobs plus release plumbing). Split each job into its own reusable workflow and call them from a slim build-kernel.yml entry point via `uses:`: build-arch-packages.yml - Arch .pkg.tar.zst build build-fedora-packages.yml - Fedora RPM build build-debian-packages.yml - Debian .deb build publish-release.yml - collects artifacts, cuts the GitHub release Version strings computed by the prepare job are passed as workflow_call inputs (krel/kbase/kverdot/sha8/tag/kver/next), the KERNEL_PACKAGES_RAW env moves into the workflows that use it, and build workflows now declare contents: read (only the publisher needs contents: write). No behavior change: same containers, steps, caches, artifact names and QEMU boot smoke test gates. Called workflows share the caller run, so the release job still sees the *-packages artifacts and the needs: chain still blocks publishing on the boot tests. Also drops a stray uncommitted local edit in the Debian build step (duplicated LLVM=1 LLVM_IAS=1 WERROR=0 make flags) so the tree matches the committed pipeline. --- .github/workflows/build-arch-packages.yml | 163 +++++ .github/workflows/build-debian-packages.yml | 292 ++++++++ .github/workflows/build-fedora-packages.yml | 207 ++++++ .github/workflows/build-kernel.yml | 730 +------------------- .github/workflows/publish-release.yml | 157 +++++ 5 files changed, 856 insertions(+), 693 deletions(-) create mode 100644 .github/workflows/build-arch-packages.yml create mode 100644 .github/workflows/build-debian-packages.yml create mode 100644 .github/workflows/build-fedora-packages.yml create mode 100644 .github/workflows/publish-release.yml diff --git a/.github/workflows/build-arch-packages.yml b/.github/workflows/build-arch-packages.yml new file mode 100644 index 00000000000000..748a422d732aa4 --- /dev/null +++ b/.github/workflows/build-arch-packages.yml @@ -0,0 +1,163 @@ +# Arch Linux packaging, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml; runs inside the same workflow +# run, so the artifacts uploaded here are seen by publish-release.yml. + +name: Build Arch Linux packages + +on: + workflow_call: + inputs: + krel: + description: Full kernel release string (uname -r), e.g. 6.12.0-next-20250101-unstable-ogc-g12345678-1 + type: string + required: true + +permissions: + contents: read + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # The distro base config is extracted from the official Arch + # `linux-headers` package — same approach as the kernel-packages + # arch.yaml workflow. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + build: + name: Build Arch Linux packages + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/archlinux:base-devel + env: + PKGDIR: /tmp/pkgbuild + DISTDIR: /tmp/dist + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ inputs.krel }} + steps: + - name: Show disk space + run: df -h / + + - name: Bootstrap Arch Linux build environment + run: | + set -euxo pipefail + # Refresh keyring first to avoid signature failures on stale images + pacman -Sy --needed --noconfirm archlinux-keyring + pacman -Su --needed --noconfirm + # Everything makepkg needs (mirrors the makedepends of the PKGBUILD). + # rust + rust-bindgen are needed because the Arch config enables + # CONFIG_RUST=y. + pacman -S --needed --noconfirm \ + bc cpio gettext libelf pahole perl python tar xz zstd \ + gcc clang llvm lld ccache pigz file curl git \ + rust rust-bindgen + # makepkg refuses to run as root + useradd -m build + install -d -o build -g build -m 0777 /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-arch-${{ github.sha }} + restore-keys: | + ccache-arch- + + - name: Prepare ccache for the build user + run: chown -R build:build /ccache && su build -c 'ccache -s' || true + + - name: Assemble Arch kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: extract the official Arch Linux kernel .config from + # the distro's `linux-headers` package (same method as the OGC + # kernel-packages arch.yaml workflow). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + pacman -Sw --noconfirm --cachedir "$CACHE" linux-headers + PKG="$(find "$CACHE" -maxdepth 1 -name 'linux-headers-*.pkg.tar.zst' -type f)" + if [ -z "$PKG" ]; then + echo "::error::linux-headers package not found in cache"; exit 1 + fi + CFG="$(tar --zstd -tf "$PKG" | grep -E '^usr/lib/modules/[^/]+/build/\.config$' || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel .config not found inside $PKG"; exit 1 + fi + tar --zstd -xOf "$PKG" "$CFG" > config + test -s config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # Layer the OGC config fragments from kernel-packages on top, then + # the repo-local config.fragment (which wins). Order is critical: + # unsets apply first (removing things we explicitly don't want), + # then sets apply (enabling things we do want), so explicit enables + # can override explicit disables. + # ------------------------------------------------------------------ + for f in arch.config.set ogc.config.set arch.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + arch.config.unset ogc.config.unset \ + arch.config.set ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Stage PKGBUILD directory + run: | + set -euxo pipefail + mkdir -p "${PKGDIR}" "${DISTDIR}" + cd /tmp/stage + cp PKGBUILD config.fragment linux.tar.gz config "${PKGDIR}/" + chown -R build:build "${PKGDIR}" + + - name: Build packages with makepkg + run: | + set -euxo pipefail + cd "${PKGDIR}" + runuser -u build -- env HOME=/home/build CCACHE_DIR=/ccache \ + makepkg -f --noconfirm --noprogressbar + + ls -lh ./*.pkg.tar.zst + cp -v ./*.pkg.tar.zst "${DISTDIR}/" + # Ship the exact .config used for the build as well + cp -v "src/linux/.config" "${DISTDIR}/config-arch-${KREL}" + + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + pacman -S --needed --noconfirm qemu-system-x86 edk2-ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the package. This + # step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + PKG="$(find /tmp/dist -name 'linux-unstable-ogc-*.pkg.tar.zst' ! -name '*-headers-*' | head -n1)" + test -n "$PKG" + tar --zstd -xf "$PKG" -C "$WORK" "usr/lib/modules/${KREL}/vmlinuz" + test -s "$WORK/usr/lib/modules/${KREL}/vmlinuz" + bash /tmp/stage/boot-smoke-test.sh "$WORK/usr/lib/modules/${KREL}/vmlinuz" + + - name: Clean build tree (free disk before cache save) + run: rm -rf "${PKGDIR}/src" "${PKGDIR}/pkg" ; df -h / + + - name: Upload Arch packages + uses: actions/upload-artifact@v7 + with: + name: arch-packages + path: /tmp/dist + retention-days: 3 diff --git a/.github/workflows/build-debian-packages.yml b/.github/workflows/build-debian-packages.yml new file mode 100644 index 00000000000000..6eef922dc09eb4 --- /dev/null +++ b/.github/workflows/build-debian-packages.yml @@ -0,0 +1,292 @@ +# Debian .deb packaging, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml; runs inside the same workflow +# run, so the artifacts uploaded here are seen by publish-release.yml. + +name: Build Debian packages + +on: + workflow_call: + inputs: + krel: + description: Full kernel release string (uname -r), e.g. 6.12.0-next-20250101-unstable-ogc-g12345678-1 + type: string + required: true + kverdot: + description: Kernel version with hyphens replaced by dots, for KDEB_PKGVERSION + type: string + required: true + sha8: + description: Short (8 char) source commit the packages are built from + type: string + required: true + +permissions: + contents: read + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # The distro base config comes from the official Debian + # linux-config/linux-image packages. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + build: + name: Build Debian packages + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/library/debian:trixie + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ inputs.krel }} + KVERDOT: ${{ inputs.kverdot }} + SHA8: ${{ inputs.sha8 }} + steps: + - name: Show disk space + run: df -h / + + - name: Install build tools + run: | + set -euxo pipefail + apt-get update + # Satisfies the Build-Depends generated by scripts/package/mkdebian + # (debhelper-compat, bc, bison, flex, kmod, libdw/libelf/libssl-dev, + # python3, rsync) plus the clang toolchain used by all OGC builds. + apt-get install -y --no-install-recommends \ + build-essential debhelper rsync \ + bc bison flex python3 kmod \ + libelf-dev libdw-dev libssl-dev zlib1g-dev \ + dwarves zstd xz-utils \ + clang llvm lld ccache curl ca-certificates git \ + openssl + # dpkg-buildpackage refuses to run as root + useradd -m builder + install -d -o builder -g builder -m 0777 /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-debian-${{ github.sha }} + restore-keys: | + ccache-debian- + + - name: Assemble Debian kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: the official Debian kernel configuration. It ships in + # the small linux-config- package; fall back to /boot/config-* + # of the linux-image- package if that is unavailable. + # ------------------------------------------------------------------ + ABI="$(apt-cache depends linux-image-amd64 \ + | awk '/^ *Depends: *linux-image-[0-9]/ {print $2; exit}' \ + | sed 's/^linux-image-//')" + test -n "${ABI}" + echo "Debian kernel ABI: ${ABI}" + # linux-config is versioned by major.minor only (e.g. 6.12) and + # stores the per-flavour config xz-compressed, e.g. + # /usr/src/linux-config-6.12/config.amd64_none_amd64.xz + KMAJMIN="$(printf '%s' "${ABI}" | cut -d. -f1,2)" + mkdir -p /tmp/pkgcfg + if apt-get download "linux-config-${KMAJMIN}"; then + dpkg-deb -x linux-config-"${KMAJMIN}"_*.deb /tmp/pkgcfg + BASE="$(find /tmp/pkgcfg/usr/src -name 'config.amd64_none_amd64.xz' | head -n1)" + test -s "${BASE}" + xz -dc "${BASE}" > config + else + apt-get download "linux-image-${ABI}" + dpkg-deb -x linux-image-"${ABI}"_*.deb /tmp/pkgcfg + BASE="$(find /tmp/pkgcfg/boot -maxdepth 1 -name 'config-*' | head -n1)" + test -s "${BASE}" + cp "${BASE}" config + fi + test -s config + rm -rf /tmp/pkgcfg + + # ------------------------------------------------------------------ + # Layer the OGC config fragments and the repo-local config.fragment + # (which wins). kernel-packages ships no debian-specific fragments. + # Order is critical: unsets apply first, then sets, so explicit + # enables can override explicit disables. + # ------------------------------------------------------------------ + for f in ogc.config.set ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + ogc.config.unset \ + ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Ensure pahole >= 1.31 (sched_ext/BTF) + run: | + set -euxo pipefail + VER="$(pahole --version | tr -d 'v')" + MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" + echo "Installed pahole: ${VER}" + if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then + echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" + apt-get install -y --no-install-recommends cmake pkg-config + cd /tmp + curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz + echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c + tar -xf dwarves-1.31.tar.xz + cmake -B dwarves-1.31/build -S dwarves-1.31 \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr + make -C dwarves-1.31/build -j"$(nproc)" install + fi + pahole --version + + - name: Build .deb packages with the in-tree packaging + run: | + set -euxo pipefail + mkdir -p /build + tar -xzf /tmp/stage/linux.tar.gz -C /build + cd /build/linux + + # Config adjustments + kernel release suffix, identical scheme to + # the Arch and Fedora packages (uname -r == ${KREL}). + cp /tmp/stage/config .config + # Debian's config references distro-only certificate files that do + # not exist in this tree; use the ephemeral in-tree key instead. + scripts/config --set-str SYSTEM_TRUSTED_KEYS "" + scripts/config --set-str SYSTEM_REVOCATION_KEYS "" + # CONFIG_MODULE_SIG_KEY points at the Debian signing cert, which + # does not exist in this tree. Reset it to the kbuild default + # ("certs/signing_key.pem"): an ephemeral self-signed key generated + # during the build and trusted by the kernel itself. The Debian + # config sets CONFIG_MODULE_SIG_ALL=y, and with an empty key string + # scripts/Makefile.modinst resolves sig-key to "./", making + # sign-file read a directory as the private key (SSL DECODER error) + # on every module. + scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + scripts/config -u DEFAULT_HOSTNAME + scripts/config --set-str BUILD_SALT "${KREL}" + echo "-unstable-ogc-g${SHA8}" > localversion.10-pkgname + echo "-1" > localversion.20-pkgrel + # The staged tarball carries no VCS data, but mkdebian (via + # gen-diff-patch) runs "git diff HEAD" and aborts on failure. + # A throwaway repo with everything committed makes that a no-op. + git init -q + git config user.email "ci@opengamingcollective.org" + git config user.name "OpenGamingCollective CI" + git add -A + git commit -qm "linux-unstable-ogc ${KREL}" + chown -R builder:builder /build + + # scripts/setlocalversion appends a "+" whenever a git repo exists + # and LOCALVERSION is unset and the HEAD is not at an annotated + # version tag. The throwaway repo below would therefore corrupt the + # release string. Setting LOCALVERSION to the empty string (as + # documented in the script) suppresses that suffix everywhere. + runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ + LOCALVERSION= \ + make CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 \ + KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc \ + olddefconfig + + # Fail fast if the release string ever drifts from the other distros + REL="$(runuser -u builder -- env HOME=/home/builder LOCALVERSION= make -s kernelrelease)" + if [ "$REL" != "${KREL}" ]; then + echo "kernelrelease '$REL' does not match expected '${KREL}'" >&2 + exit 1 + fi + + # Generate the debian/ directory with the tree's own packaging. + # mkdebian is invoked directly as a script: a bare "make debian" is + # NOT a top-level make target in this tree (only *-pkg patterns are + # delegated to scripts/Makefile.package), and going through make + # without the exact same CC flags as the olddefconfig above made + # kbuild re-sync the config interactively. As a plain sh script it + # touches nothing kbuild-related. + # KDEB_PKGVERSION must not contain hyphens except the final revision + # separator (Debian policy), hence the dotted translation. + KDEBVER="${KVERDOT}.unstable.ogc.g${SHA8}-1" + runuser -u builder -- env HOME=/home/builder \ + srctree="$PWD" \ + ARCH=x86_64 SRCARCH=x86 UTS_MACHINE=x86_64 \ + KERNELRELEASE="${KREL}" \ + KCONFIG_CONFIG=.config \ + KDEB_SOURCENAME=linux-unstable-ogc \ + KDEB_PKGVERSION="${KDEBVER}" \ + KDEB_CHANGELOG_DIST=trixie \ + DEBFULLNAME="OpenGamingCollective CI" \ + DEBEMAIL="ci@opengamingcollective.org" \ + sh scripts/package/mkdebian + echo "debian arch: $(cat debian/arch)" + + # Kbuild only honours command-line variables, so inject the + # clang/ccache toolchain into the generated debian/rules (this is + # what "make bindeb-pkg" would otherwise lose). LOCALVERSION= (set, + # empty) keeps setlocalversion from appending "+" inside the build. + sed -i 's|^make-opts = |make-opts = CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 LOCALVERSION= KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc |' debian/rules + grep -n '^make-opts' debian/rules + + # Same invocation as "make bindeb-pkg" (scripts/Makefile.package), + # but with parallel jobs, --no-check-builddeps (the build deps are + # preinstalled above; the generated Build-Depends-Arch also names a + # distro-only cross-gcc package that need not exist), and without + # the multi-GB debug-symbol package (all other distro jobs ship no + # debug packages either). + runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ + LOCALVERSION= \ + DEB_BUILD_PROFILES="pkg.linux-unstable-ogc.nokerneldbg" \ + dpkg-buildpackage --build=binary --no-pre-clean --unsigned-changes \ + --no-check-builddeps \ + -R'make -f debian/rules' -j"$(nproc)" -a"$(cat debian/arch)" + + ls -lh /build/*.deb + + - name: Collect Debian artifacts + run: | + set -euxo pipefail + DISTDIR=/tmp/dist + mkdir -p "${DISTDIR}" + cp -v "/build/linux-image-${KREL}"_*.deb "${DISTDIR}/" + cp -v "/build/linux-headers-${KREL}"_*.deb "${DISTDIR}/" + cp -v /build/linux/.config "${DISTDIR}/config-debian-${KREL}" + # linux-libc-dev is intentionally NOT shipped: installing it would + # replace the distribution's own linux-libc-dev package. + ls -lh "${DISTDIR}" + + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + apt-get update -qq + # qemu boots the kernel (OVMF provides the UEFI leg); cpio builds + # the test initramfs. + apt-get install -y --no-install-recommends qemu-system-x86 cpio ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the linux-image .deb. + # This step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + DEB="$(find /tmp/dist -name "linux-image-${KREL}_*.deb" | head -n1)" + test -n "$DEB" + dpkg-deb -x "$DEB" "$WORK" + IMG="$(find "$WORK" -name 'vmlinuz-*' -type f | head -n1)" + test -n "$IMG" + bash /tmp/stage/boot-smoke-test.sh "$IMG" + + - name: Clean build tree (free disk before cache save) + run: rm -rf /build ; df -h / + + - name: Upload Debian packages + uses: actions/upload-artifact@v7 + with: + name: debian-packages + path: /tmp/dist + retention-days: 3 diff --git a/.github/workflows/build-fedora-packages.yml b/.github/workflows/build-fedora-packages.yml new file mode 100644 index 00000000000000..0a95d181f8ab77 --- /dev/null +++ b/.github/workflows/build-fedora-packages.yml @@ -0,0 +1,207 @@ +# Fedora RPM packaging, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml; runs inside the same workflow +# run, so the artifacts uploaded here are seen by publish-release.yml. + +name: Build Fedora RPM packages + +on: + workflow_call: + inputs: + krel: + description: Full kernel release string (uname -r), e.g. 6.12.0-next-20250101-unstable-ogc-g12345678-1 + type: string + required: true + kbase: + description: Base kernel version incl. prerelease suffix but no packaging suffix (e.g. 6.12.0-next-20250101) + type: string + required: true + kverdot: + description: Kernel version with hyphens replaced by dots, for the RPM Version field + type: string + required: true + sha8: + description: Short (8 char) source commit the packages are built from + type: string + required: true + +permissions: + contents: read + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # The distro base config is extracted from the official Fedora + # `kernel-core` package — same approach as the kernel-packages + # fedora.yaml workflow. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + build: + name: Build Fedora RPM packages + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/library/fedora:43 + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ inputs.krel }} + KBASE: ${{ inputs.kbase }} + KVERDOT: ${{ inputs.kverdot }} + SHA8: ${{ inputs.sha8 }} + steps: + - name: Show disk space + run: df -h / + + - name: Install build tools + run: | + set -euxo pipefail + dnf -y install dnf5-plugins rpm-build + dnf -y install cpio curl findutils tar gzip + mkdir -p /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-fedora-${{ github.sha }} + restore-keys: | + ccache-fedora- + + - name: Assemble Fedora kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: extract the official Fedora kernel .config from the + # distro's `kernel-core` package (same method as the OGC + # kernel-packages fedora.yaml workflow). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + dnf download --destdir "$CACHE" kernel-core + RPM="$(find "$CACHE" -maxdepth 1 -name 'kernel-core-*.rpm' -type f)" + if [ -z "$RPM" ]; then + echo "::error::kernel-core package not found in cache"; exit 1 + fi + CFG="$(rpm -qlp "$RPM" | grep -E '^/lib/modules/[^/]+/config$' | head -n1 || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel config not found inside $RPM"; exit 1 + fi + rpm2cpio "$RPM" | cpio -i --to-stdout ".${CFG}" > config + test -s config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # Layer the OGC config fragments on top, then the repo-local + # config.fragment (which wins). Order is critical: unsets apply + # first (removing things we explicitly don't want), then sets apply + # (enabling things we do want), so explicit enables can override + # explicit disables. + # ------------------------------------------------------------------ + for f in fedora.config.set ogc.config.set fedora.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + fedora.config.unset ogc.config.unset \ + fedora.config.set ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Finalize spec and install build dependencies + working-directory: /tmp/stage + run: | + set -euxo pipefail + # Substitute the CI placeholders (must happen before `dnf builddep` + # so RPM can parse Version:/Release:). + sed -i \ + -e "s/@@KBASEVER@@/${KBASE}/" \ + -e "s/@@KVERDOTTED@@/${KVERDOT}/" \ + -e "s/@@SHA8@@/${SHA8}/" \ + kernel.spec + grep -n '^Version:\|^Release:\|%define kbasever\|%define sha8' kernel.spec + ! grep -q '@@' kernel.spec + + # Installs the toolchain from the spec BuildRequires, including the + # distro dwarves/pahole. Run BEFORE the pahole check below so a + # source-built pahole is not overwritten by builddep. + dnf -y builddep kernel.spec + + - name: Ensure pahole >= 1.31 (sched_ext/BTF) + working-directory: /tmp + run: | + set -euxo pipefail + VER="$(pahole --version | tr -d 'v')" + MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" + echo "Installed pahole: ${VER}" + if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then + echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" + dnf -y install cmake make gcc elfutils-devel zlib-devel + curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz + echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c + tar -xf dwarves-1.31.tar.xz + cmake -B dwarves-1.31/build -S dwarves-1.31 \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -D__LIB=lib + make -C dwarves-1.31/build -j"$(nproc)" install + fi + pahole --version + + - name: Build RPMs with rpmbuild + working-directory: /tmp + run: | + set -euxo pipefail + TOPDIR=/tmp/rpmbuild + mkdir -p "${TOPDIR}"/{BUILD,BUILDROOT,RPMS,SOURCES,SPECS,SRPMS} + cp /tmp/stage/linux.tar.gz "${TOPDIR}/SOURCES/" + cp /tmp/stage/config "${TOPDIR}/SOURCES/config" + cp /tmp/stage/kernel.spec "${TOPDIR}/SPECS/kernel.spec" + + rpmbuild --define "_topdir ${TOPDIR}" -ba "${TOPDIR}/SPECS/kernel.spec" + + ls -lh "${TOPDIR}"/RPMS/x86_64/ + + - name: Collect Fedora artifacts + run: | + set -euxo pipefail + DISTDIR=/tmp/dist + mkdir -p "${DISTDIR}" + cp -v /tmp/rpmbuild/RPMS/x86_64/*.rpm "${DISTDIR}/" + cp -v /tmp/stage/config "${DISTDIR}/config-fedora-${KREL}" + ls -lh "${DISTDIR}" + + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + dnf -y install qemu-system-x86 edk2-ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the kernel-core RPM. + # This step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + RPM="$(find /tmp/dist -name 'kernel-unstable-ogc-core-*.rpm' | head -n1)" + test -n "$RPM" + # rpm cpio entries are stored with a "./" prefix (same trick as + # the config extraction above). + rpm2cpio "$RPM" | cpio -idm --quiet -D "$WORK" "./lib/modules/${KREL}/vmlinuz" || true + IMG="$WORK/lib/modules/${KREL}/vmlinuz" + test -s "$IMG" || { echo "::error::vmlinuz not found inside ${RPM}"; exit 1; } + bash /tmp/stage/boot-smoke-test.sh "$IMG" + + - name: Clean build tree (free disk before cache save) + run: rm -rf /tmp/rpmbuild/BUILD /tmp/rpmbuild/BUILDROOT ; df -h / + + - name: Upload Fedora packages + uses: actions/upload-artifact@v7 + with: + name: fedora-packages + path: /tmp/dist + retention-days: 3 diff --git a/.github/workflows/build-kernel.yml b/.github/workflows/build-kernel.yml index 3274de60d762f2..8ca98fa697d2d6 100644 --- a/.github/workflows/build-kernel.yml +++ b/.github/workflows/build-kernel.yml @@ -1,3 +1,16 @@ +# Entry point of the linux-unstable-ogc build pipeline. The heavy lifting +# lives in reusable workflows next to this file: +# +# build-arch-packages.yml - Arch Linux .pkg.tar.zst builds +# build-fedora-packages.yml - Fedora RPM builds +# build-debian-packages.yml - Debian .deb builds +# publish-release.yml - collects artifacts and cuts the GitHub release +# +# They are called with `uses:` so everything stays a single workflow run: +# artifacts uploaded by the build jobs are downloaded by publish-release.yml, +# and the needs: chain below keeps the release gated on the QEMU boot smoke +# tests that run inside each build workflow. + name: Build & release linux-unstable-ogc on: @@ -12,13 +25,6 @@ concurrency: group: kernel-build-${{ github.ref }} cancel-in-progress: true -env: - # Config fragments maintained by the OGC kernel-packages repository. - # Distro base configs are extracted from the official distro packages - # (Arch: linux-headers, Fedora: kernel-core) — same approach as the - # kernel-packages arch.yaml / fedora.yaml workflows. - KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main - jobs: prepare: name: Prepare sources and version @@ -87,699 +93,37 @@ jobs: retention-days: 3 arch: - name: Build Arch Linux packages + name: Arch packages needs: prepare - runs-on: ubuntu-latest - timeout-minutes: 330 - container: - image: docker.io/archlinux:base-devel - env: - PKGDIR: /tmp/pkgbuild - DISTDIR: /tmp/dist - CCACHE_DIR: /ccache - CCACHE_MAXSIZE: 10G - KREL: ${{ needs.prepare.outputs.krel }} - steps: - - name: Show disk space - run: df -h / - - - name: Bootstrap Arch Linux build environment - run: | - set -euxo pipefail - # Refresh keyring first to avoid signature failures on stale images - pacman -Sy --needed --noconfirm archlinux-keyring - pacman -Su --needed --noconfirm - # Everything makepkg needs (mirrors the makedepends of the PKGBUILD). - # rust + rust-bindgen are needed because the Arch config enables - # CONFIG_RUST=y. - pacman -S --needed --noconfirm \ - bc cpio gettext libelf pahole perl python tar xz zstd \ - gcc clang llvm lld ccache pigz file curl git \ - rust rust-bindgen - # makepkg refuses to run as root - useradd -m build - install -d -o build -g build -m 0777 /ccache - - - name: Download staged sources - uses: actions/download-artifact@v8 - with: - name: kernel-sources - path: /tmp/stage - - - name: Restore compiler cache - uses: actions/cache@v6 - with: - path: /ccache - key: ccache-arch-${{ github.sha }} - restore-keys: | - ccache-arch- - - - name: Prepare ccache for the build user - run: chown -R build:build /ccache && su build -c 'ccache -s' || true - - - name: Assemble Arch kernel config - working-directory: /tmp/stage - run: | - set -euxo pipefail - # ------------------------------------------------------------------ - # Base config: extract the official Arch Linux kernel .config from - # the distro's `linux-headers` package (same method as the OGC - # kernel-packages arch.yaml workflow). - # ------------------------------------------------------------------ - CACHE="/tmp/pkgcache" - mkdir -p "$CACHE" - pacman -Sw --noconfirm --cachedir "$CACHE" linux-headers - PKG="$(find "$CACHE" -maxdepth 1 -name 'linux-headers-*.pkg.tar.zst' -type f)" - if [ -z "$PKG" ]; then - echo "::error::linux-headers package not found in cache"; exit 1 - fi - CFG="$(tar --zstd -tf "$PKG" | grep -E '^usr/lib/modules/[^/]+/build/\.config$' || true)" - if [ -z "$CFG" ]; then - echo "::error::kernel .config not found inside $PKG"; exit 1 - fi - tar --zstd -xOf "$PKG" "$CFG" > config - test -s config - rm -rf "$CACHE" - - # ------------------------------------------------------------------ - # Layer the OGC config fragments from kernel-packages on top, then - # the repo-local config.fragment (which wins). Order is critical: - # unsets apply first (removing things we explicitly don't want), - # then sets apply (enabling things we do want), so explicit enables - # can override explicit disables. - # ------------------------------------------------------------------ - for f in arch.config.set ogc.config.set arch.config.unset ogc.config.unset; do - curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" - done - bash ./merge-fragments.sh config \ - arch.config.unset ogc.config.unset \ - arch.config.set ogc.config.set \ - config.fragment - echo "Config after fragment merge (head):" - head -n 3 config - - - name: Stage PKGBUILD directory - run: | - set -euxo pipefail - mkdir -p "${PKGDIR}" "${DISTDIR}" - cd /tmp/stage - cp PKGBUILD config.fragment linux.tar.gz config "${PKGDIR}/" - chown -R build:build "${PKGDIR}" - - - name: Build packages with makepkg - run: | - set -euxo pipefail - cd "${PKGDIR}" - runuser -u build -- env HOME=/home/build CCACHE_DIR=/ccache \ - makepkg -f --noconfirm --noprogressbar - - ls -lh ./*.pkg.tar.zst - cp -v ./*.pkg.tar.zst "${DISTDIR}/" - # Ship the exact .config used for the build as well - cp -v "src/linux/.config" "${DISTDIR}/config-arch-${KREL}" - - - name: Boot smoke test in QEMU (release gate) - run: | - set -euxo pipefail - pacman -S --needed --noconfirm qemu-system-x86 edk2-ovmf - # ---------------------------------------------------------------- - # Boot the exact kernel image that ships in the package. This - # step fails the job (and with it the release) if the kernel - # panics, hangs or never reaches userspace, so an unbootable - # kernel is never uploaded or published. - # ---------------------------------------------------------------- - WORK=/tmp/bootsmoke - rm -rf "$WORK"; mkdir -p "$WORK" - PKG="$(find /tmp/dist -name 'linux-unstable-ogc-*.pkg.tar.zst' ! -name '*-headers-*' | head -n1)" - test -n "$PKG" - tar --zstd -xf "$PKG" -C "$WORK" "usr/lib/modules/${KREL}/vmlinuz" - test -s "$WORK/usr/lib/modules/${KREL}/vmlinuz" - bash /tmp/stage/boot-smoke-test.sh "$WORK/usr/lib/modules/${KREL}/vmlinuz" - - - name: Clean build tree (free disk before cache save) - run: rm -rf "${PKGDIR}/src" "${PKGDIR}/pkg" ; df -h / - - - name: Upload Arch packages - uses: actions/upload-artifact@v7 - with: - name: arch-packages - path: /tmp/dist - retention-days: 3 + uses: ./.github/workflows/build-arch-packages.yml + with: + krel: ${{ needs.prepare.outputs.krel }} fedora: - name: Build Fedora RPM packages + name: Fedora packages needs: prepare - runs-on: ubuntu-latest - timeout-minutes: 330 - container: - image: docker.io/library/fedora:43 - env: - CCACHE_DIR: /ccache - CCACHE_MAXSIZE: 10G - KREL: ${{ needs.prepare.outputs.krel }} - KBASE: ${{ needs.prepare.outputs.kbase }} - KVERDOT: ${{ needs.prepare.outputs.kverdot }} - SHA8: ${{ needs.prepare.outputs.sha8 }} - steps: - - name: Show disk space - run: df -h / - - - name: Install build tools - run: | - set -euxo pipefail - dnf -y install dnf5-plugins rpm-build - dnf -y install cpio curl findutils tar gzip - mkdir -p /ccache - - - name: Download staged sources - uses: actions/download-artifact@v8 - with: - name: kernel-sources - path: /tmp/stage - - - name: Restore compiler cache - uses: actions/cache@v6 - with: - path: /ccache - key: ccache-fedora-${{ github.sha }} - restore-keys: | - ccache-fedora- - - - name: Assemble Fedora kernel config - working-directory: /tmp/stage - run: | - set -euxo pipefail - # ------------------------------------------------------------------ - # Base config: extract the official Fedora kernel .config from the - # distro's `kernel-core` package (same method as the OGC - # kernel-packages fedora.yaml workflow). - # ------------------------------------------------------------------ - CACHE="/tmp/pkgcache" - mkdir -p "$CACHE" - dnf download --destdir "$CACHE" kernel-core - RPM="$(find "$CACHE" -maxdepth 1 -name 'kernel-core-*.rpm' -type f)" - if [ -z "$RPM" ]; then - echo "::error::kernel-core package not found in cache"; exit 1 - fi - CFG="$(rpm -qlp "$RPM" | grep -E '^/lib/modules/[^/]+/config$' | head -n1 || true)" - if [ -z "$CFG" ]; then - echo "::error::kernel config not found inside $RPM"; exit 1 - fi - rpm2cpio "$RPM" | cpio -i --to-stdout ".${CFG}" > config - test -s config - rm -rf "$CACHE" - - # ------------------------------------------------------------------ - # Layer the OGC config fragments on top, then the repo-local - # config.fragment (which wins). Order is critical: unsets apply - # first (removing things we explicitly don't want), then sets apply - # (enabling things we do want), so explicit enables can override - # explicit disables. - # ------------------------------------------------------------------ - for f in fedora.config.set ogc.config.set fedora.config.unset ogc.config.unset; do - curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" - done - bash ./merge-fragments.sh config \ - fedora.config.unset ogc.config.unset \ - fedora.config.set ogc.config.set \ - config.fragment - echo "Config after fragment merge (head):" - head -n 3 config - - - name: Finalize spec and install build dependencies - working-directory: /tmp/stage - run: | - set -euxo pipefail - # Substitute the CI placeholders (must happen before `dnf builddep` - # so RPM can parse Version:/Release:). - sed -i \ - -e "s/@@KBASEVER@@/${KBASE}/" \ - -e "s/@@KVERDOTTED@@/${KVERDOT}/" \ - -e "s/@@SHA8@@/${SHA8}/" \ - kernel.spec - grep -n '^Version:\|^Release:\|%define kbasever\|%define sha8' kernel.spec - ! grep -q '@@' kernel.spec - - # Installs the toolchain from the spec BuildRequires, including the - # distro dwarves/pahole. Run BEFORE the pahole check below so a - # source-built pahole is not overwritten by builddep. - dnf -y builddep kernel.spec - - - name: Ensure pahole >= 1.31 (sched_ext/BTF) - working-directory: /tmp - run: | - set -euxo pipefail - VER="$(pahole --version | tr -d 'v')" - MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" - echo "Installed pahole: ${VER}" - if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then - echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" - dnf -y install cmake make gcc elfutils-devel zlib-devel - curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz - echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c - tar -xf dwarves-1.31.tar.xz - cmake -B dwarves-1.31/build -S dwarves-1.31 \ - -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -D__LIB=lib - make -C dwarves-1.31/build -j"$(nproc)" install - fi - pahole --version - - - name: Build RPMs with rpmbuild - working-directory: /tmp - run: | - set -euxo pipefail - TOPDIR=/tmp/rpmbuild - mkdir -p "${TOPDIR}"/{BUILD,BUILDROOT,RPMS,SOURCES,SPECS,SRPMS} - cp /tmp/stage/linux.tar.gz "${TOPDIR}/SOURCES/" - cp /tmp/stage/config "${TOPDIR}/SOURCES/config" - cp /tmp/stage/kernel.spec "${TOPDIR}/SPECS/kernel.spec" - - rpmbuild --define "_topdir ${TOPDIR}" -ba "${TOPDIR}/SPECS/kernel.spec" - - ls -lh "${TOPDIR}"/RPMS/x86_64/ - - - name: Collect Fedora artifacts - run: | - set -euxo pipefail - DISTDIR=/tmp/dist - mkdir -p "${DISTDIR}" - cp -v /tmp/rpmbuild/RPMS/x86_64/*.rpm "${DISTDIR}/" - cp -v /tmp/stage/config "${DISTDIR}/config-fedora-${KREL}" - ls -lh "${DISTDIR}" - - - name: Boot smoke test in QEMU (release gate) - run: | - set -euxo pipefail - dnf -y install qemu-system-x86 edk2-ovmf - # ---------------------------------------------------------------- - # Boot the exact kernel image that ships in the kernel-core RPM. - # This step fails the job (and with it the release) if the kernel - # panics, hangs or never reaches userspace, so an unbootable - # kernel is never uploaded or published. - # ---------------------------------------------------------------- - WORK=/tmp/bootsmoke - rm -rf "$WORK"; mkdir -p "$WORK" - RPM="$(find /tmp/dist -name 'kernel-unstable-ogc-core-*.rpm' | head -n1)" - test -n "$RPM" - # rpm cpio entries are stored with a "./" prefix (same trick as - # the config extraction above). - rpm2cpio "$RPM" | cpio -idm --quiet -D "$WORK" "./lib/modules/${KREL}/vmlinuz" || true - IMG="$WORK/lib/modules/${KREL}/vmlinuz" - test -s "$IMG" || { echo "::error::vmlinuz not found inside ${RPM}"; exit 1; } - bash /tmp/stage/boot-smoke-test.sh "$IMG" - - - name: Clean build tree (free disk before cache save) - run: rm -rf /tmp/rpmbuild/BUILD /tmp/rpmbuild/BUILDROOT ; df -h / - - - name: Upload Fedora packages - uses: actions/upload-artifact@v7 - with: - name: fedora-packages - path: /tmp/dist - retention-days: 3 + uses: ./.github/workflows/build-fedora-packages.yml + with: + krel: ${{ needs.prepare.outputs.krel }} + kbase: ${{ needs.prepare.outputs.kbase }} + kverdot: ${{ needs.prepare.outputs.kverdot }} + sha8: ${{ needs.prepare.outputs.sha8 }} debian: - name: Build Debian packages + name: Debian packages needs: prepare - runs-on: ubuntu-latest - timeout-minutes: 330 - container: - image: docker.io/library/debian:trixie - env: - CCACHE_DIR: /ccache - CCACHE_MAXSIZE: 10G - KREL: ${{ needs.prepare.outputs.krel }} - KVERDOT: ${{ needs.prepare.outputs.kverdot }} - SHA8: ${{ needs.prepare.outputs.sha8 }} - steps: - - name: Show disk space - run: df -h / - - - name: Install build tools - run: | - set -euxo pipefail - apt-get update - # Satisfies the Build-Depends generated by scripts/package/mkdebian - # (debhelper-compat, bc, bison, flex, kmod, libdw/libelf/libssl-dev, - # python3, rsync) plus the clang toolchain used by all OGC builds. - apt-get install -y --no-install-recommends \ - build-essential debhelper rsync \ - bc bison flex python3 kmod \ - libelf-dev libdw-dev libssl-dev zlib1g-dev \ - dwarves zstd xz-utils \ - clang llvm lld ccache curl ca-certificates git \ - openssl - # dpkg-buildpackage refuses to run as root - useradd -m builder - install -d -o builder -g builder -m 0777 /ccache - - - name: Download staged sources - uses: actions/download-artifact@v8 - with: - name: kernel-sources - path: /tmp/stage - - - name: Restore compiler cache - uses: actions/cache@v6 - with: - path: /ccache - key: ccache-debian-${{ github.sha }} - restore-keys: | - ccache-debian- - - - name: Assemble Debian kernel config - working-directory: /tmp/stage - run: | - set -euxo pipefail - # ------------------------------------------------------------------ - # Base config: the official Debian kernel configuration. It ships in - # the small linux-config- package; fall back to /boot/config-* - # of the linux-image- package if that is unavailable. - # ------------------------------------------------------------------ - ABI="$(apt-cache depends linux-image-amd64 \ - | awk '/^ *Depends: *linux-image-[0-9]/ {print $2; exit}' \ - | sed 's/^linux-image-//')" - test -n "${ABI}" - echo "Debian kernel ABI: ${ABI}" - # linux-config is versioned by major.minor only (e.g. 6.12) and - # stores the per-flavour config xz-compressed, e.g. - # /usr/src/linux-config-6.12/config.amd64_none_amd64.xz - KMAJMIN="$(printf '%s' "${ABI}" | cut -d. -f1,2)" - mkdir -p /tmp/pkgcfg - if apt-get download "linux-config-${KMAJMIN}"; then - dpkg-deb -x linux-config-"${KMAJMIN}"_*.deb /tmp/pkgcfg - BASE="$(find /tmp/pkgcfg/usr/src -name 'config.amd64_none_amd64.xz' | head -n1)" - test -s "${BASE}" - xz -dc "${BASE}" > config - else - apt-get download "linux-image-${ABI}" - dpkg-deb -x linux-image-"${ABI}"_*.deb /tmp/pkgcfg - BASE="$(find /tmp/pkgcfg/boot -maxdepth 1 -name 'config-*' | head -n1)" - test -s "${BASE}" - cp "${BASE}" config - fi - test -s config - rm -rf /tmp/pkgcfg - - # ------------------------------------------------------------------ - # Layer the OGC config fragments and the repo-local config.fragment - # (which wins). kernel-packages ships no debian-specific fragments. - # Order is critical: unsets apply first, then sets, so explicit - # enables can override explicit disables. - # ------------------------------------------------------------------ - for f in ogc.config.set ogc.config.unset; do - curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" - done - bash ./merge-fragments.sh config \ - ogc.config.unset \ - ogc.config.set \ - config.fragment - echo "Config after fragment merge (head):" - head -n 3 config - - - name: Ensure pahole >= 1.31 (sched_ext/BTF) - run: | - set -euxo pipefail - VER="$(pahole --version | tr -d 'v')" - MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" - echo "Installed pahole: ${VER}" - if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then - echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" - apt-get install -y --no-install-recommends cmake pkg-config - cd /tmp - curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz - echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c - tar -xf dwarves-1.31.tar.xz - cmake -B dwarves-1.31/build -S dwarves-1.31 \ - -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr - make -C dwarves-1.31/build -j"$(nproc)" install - fi - pahole --version - - - name: Build .deb packages with the in-tree packaging - run: | - set -euxo pipefail - mkdir -p /build - tar -xzf /tmp/stage/linux.tar.gz -C /build - cd /build/linux - - # Config adjustments + kernel release suffix, identical scheme to - # the Arch and Fedora packages (uname -r == ${KREL}). - cp /tmp/stage/config .config - # Debian's config references distro-only certificate files that do - # not exist in this tree; use the ephemeral in-tree key instead. - scripts/config --set-str SYSTEM_TRUSTED_KEYS "" - scripts/config --set-str SYSTEM_REVOCATION_KEYS "" - # CONFIG_MODULE_SIG_KEY points at the Debian signing cert, which - # does not exist in this tree. Reset it to the kbuild default - # ("certs/signing_key.pem"): an ephemeral self-signed key generated - # during the build and trusted by the kernel itself. The Debian - # config sets CONFIG_MODULE_SIG_ALL=y, and with an empty key string - # scripts/Makefile.modinst resolves sig-key to "./", making - # sign-file read a directory as the private key (SSL DECODER error) - # on every module. - scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" - scripts/config -u DEFAULT_HOSTNAME - scripts/config --set-str BUILD_SALT "${KREL}" - echo "-unstable-ogc-g${SHA8}" > localversion.10-pkgname - echo "-1" > localversion.20-pkgrel - # The staged tarball carries no VCS data, but mkdebian (via - # gen-diff-patch) runs "git diff HEAD" and aborts on failure. - # A throwaway repo with everything committed makes that a no-op. - git init -q - git config user.email "ci@opengamingcollective.org" - git config user.name "OpenGamingCollective CI" - git add -A - git commit -qm "linux-unstable-ogc ${KREL}" - chown -R builder:builder /build - - # scripts/setlocalversion appends a "+" whenever a git repo exists - # and LOCALVERSION is unset and the HEAD is not at an annotated - # version tag. The throwaway repo below would therefore corrupt the - # release string. Setting LOCALVERSION to the empty string (as - # documented in the script) suppresses that suffix everywhere. - runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ - LOCALVERSION= \ - make CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 \ - KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc \ - olddefconfig - - # Fail fast if the release string ever drifts from the other distros - REL="$(runuser -u builder -- env HOME=/home/builder LOCALVERSION= make -s kernelrelease)" - if [ "$REL" != "${KREL}" ]; then - echo "kernelrelease '$REL' does not match expected '${KREL}'" >&2 - exit 1 - fi - - # Generate the debian/ directory with the tree's own packaging. - # mkdebian is invoked directly as a script: a bare "make debian" is - # NOT a top-level make target in this tree (only *-pkg patterns are - # delegated to scripts/Makefile.package), and going through make - # without the exact same CC flags as the olddefconfig above made - # kbuild re-sync the config interactively. As a plain sh script it - # touches nothing kbuild-related. - # KDEB_PKGVERSION must not contain hyphens except the final revision - # separator (Debian policy), hence the dotted translation. - KDEBVER="${KVERDOT}.unstable.ogc.g${SHA8}-1" - runuser -u builder -- env HOME=/home/builder \ - srctree="$PWD" \ - ARCH=x86_64 SRCARCH=x86 UTS_MACHINE=x86_64 \ - KERNELRELEASE="${KREL}" \ - KCONFIG_CONFIG=.config \ - KDEB_SOURCENAME=linux-unstable-ogc \ - KDEB_PKGVERSION="${KDEBVER}" \ - KDEB_CHANGELOG_DIST=trixie \ - DEBFULLNAME="OpenGamingCollective CI" \ - DEBEMAIL="ci@opengamingcollective.org" \ - sh scripts/package/mkdebian - echo "debian arch: $(cat debian/arch)" - - # Kbuild only honours command-line variables, so inject the - # clang/ccache toolchain into the generated debian/rules (this is - # what "make bindeb-pkg" would otherwise lose). LOCALVERSION= (set, - # empty) keeps setlocalversion from appending "+" inside the build. - sed -i 's|^make-opts = |make-opts = CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 LOCALVERSION= KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc |' debian/rules - grep -n '^make-opts' debian/rules - - # Same invocation as "make bindeb-pkg" (scripts/Makefile.package), - # but with parallel jobs, --no-check-builddeps (the build deps are - # preinstalled above; the generated Build-Depends-Arch also names a - # distro-only cross-gcc package that need not exist), and without - # the multi-GB debug-symbol package (all other distro jobs ship no - # debug packages either). - runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ - LOCALVERSION= \ - DEB_BUILD_PROFILES="pkg.linux-unstable-ogc.nokerneldbg" \ - dpkg-buildpackage --build=binary --no-pre-clean --unsigned-changes \ - --no-check-builddeps \ - -R'make -f debian/rules' -j"$(nproc)" -a"$(cat debian/arch)" - - ls -lh /build/*.deb - - - name: Collect Debian artifacts - run: | - set -euxo pipefail - DISTDIR=/tmp/dist - mkdir -p "${DISTDIR}" - cp -v "/build/linux-image-${KREL}"_*.deb "${DISTDIR}/" - cp -v "/build/linux-headers-${KREL}"_*.deb "${DISTDIR}/" - cp -v /build/linux/.config "${DISTDIR}/config-debian-${KREL}" - # linux-libc-dev is intentionally NOT shipped: installing it would - # replace the distribution's own linux-libc-dev package. - ls -lh "${DISTDIR}" - - - name: Boot smoke test in QEMU (release gate) - run: | - set -euxo pipefail - apt-get update -qq - # qemu boots the kernel (OVMF provides the UEFI leg); cpio builds - # the test initramfs. - apt-get install -y --no-install-recommends qemu-system-x86 cpio ovmf - # ---------------------------------------------------------------- - # Boot the exact kernel image that ships in the linux-image .deb. - # This step fails the job (and with it the release) if the kernel - # panics, hangs or never reaches userspace, so an unbootable - # kernel is never uploaded or published. - # ---------------------------------------------------------------- - WORK=/tmp/bootsmoke - rm -rf "$WORK"; mkdir -p "$WORK" - DEB="$(find /tmp/dist -name "linux-image-${KREL}_*.deb" | head -n1)" - test -n "$DEB" - dpkg-deb -x "$DEB" "$WORK" - IMG="$(find "$WORK" -name 'vmlinuz-*' -type f | head -n1)" - test -n "$IMG" - bash /tmp/stage/boot-smoke-test.sh "$IMG" - - - name: Clean build tree (free disk before cache save) - run: rm -rf /build ; df -h / - - - name: Upload Debian packages - uses: actions/upload-artifact@v7 - with: - name: debian-packages - path: /tmp/dist - retention-days: 3 + uses: ./.github/workflows/build-debian-packages.yml + with: + krel: ${{ needs.prepare.outputs.krel }} + kverdot: ${{ needs.prepare.outputs.kverdot }} + sha8: ${{ needs.prepare.outputs.sha8 }} release: - name: Create GitHub release + name: Publish GitHub release needs: [prepare, arch, fedora, debian] - runs-on: ubuntu-latest - env: - TAG: ${{ needs.prepare.outputs.tag }} - KREL: ${{ needs.prepare.outputs.krel }} - KVER: ${{ needs.prepare.outputs.kver }} - NEXT: ${{ needs.prepare.outputs.next }} - steps: - - name: Download package artifacts - uses: actions/download-artifact@v8 - with: - path: /tmp/dist - pattern: "*-packages" - - - name: Flatten artifact directory - run: | - set -euxo pipefail - cd /tmp/dist - find . -mindepth 2 -maxdepth 2 -type f -exec mv -t . {} + - find . -mindepth 1 -type d -delete - ls -lh - - - name: Generate checksums - run: | - set -euxo pipefail - cd /tmp/dist - sha256sum * > SHA256SUMS - cat SHA256SUMS - - - name: Create GitHub release and upload packages - env: - GH_TOKEN: ${{ github.token }} - run: | - set -euxo pipefail - - # Idempotency: if a stale release exists for this tag (e.g. re-run - # of the same commit), remove it so this run can recreate it. - if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" 2>/dev/null; then - echo "Deleting existing release $TAG, will recreate it" - gh release delete "$TAG" --repo "$GITHUB_REPOSITORY" --yes --cleanup-tag - fi - # In case a tag without a release is left over, drop it too. - gh api -X DELETE "repos/${GITHUB_REPOSITORY}/git/refs/tags/${TAG}" >/dev/null 2>&1 || true - - NOTES="$(mktemp)" - { - echo "Automated build of linux-unstable-ogc." - echo - echo "- Source commit: https://github.com/${GITHUB_REPOSITORY}/commit/${GITHUB_SHA}" - echo "- Kernel release: ${KREL}" - echo "- Upstream version: ${KVER}${NEXT}" - echo "- Base configs: Arch \`linux-headers\` + Fedora \`kernel-core\`, plus [OGC kernel-packages fragments](${KERNEL_PACKAGES_RAW}/config)" - echo "- Compiler: clang / LLVM=1 (with ccache)" - echo "- Boot smoke test: every packaged kernel (Arch, Fedora, Debian) was booted in QEMU and reached userspace init before publishing" - echo - echo "### Artifacts" - echo - echo "#### Arch Linux" - echo - echo '- `linux-unstable-ogc` — kernel image and modules' - echo '- `linux-unstable-ogc-headers` — headers for building external modules' - echo "- \`config-arch-${KREL}\` — the exact .config used for this build" - echo - echo "#### Fedora" - echo - echo '- `kernel-unstable-ogc-core` — kernel image (vmlinuz) and core files' - echo '- `kernel-unstable-ogc-modules` — kernel modules' - echo '- `kernel-unstable-ogc-devel` — headers for building external modules' - echo "- \`config-fedora-${KREL}\` — the exact .config used for this build" - echo - echo "#### Debian (and derivatives)" - echo - echo "- \`linux-image-${KREL}\` — kernel image and modules" - echo "- \`linux-headers-${KREL}\` — headers for building external modules" - echo "- \`config-debian-${KREL}\` — the exact .config used for this build" - echo - echo '- `SHA256SUMS` — checksums of all artifacts' - echo - echo "### Install" - echo - echo "Arch Linux:" - echo - echo '```sh' - echo 'sudo pacman -U linux-unstable-ogc-headers-*.pkg.tar.zst linux-unstable-ogc-*.pkg.tar.zst' - echo '```' - echo - echo "Fedora:" - echo - echo '```sh' - echo 'sudo dnf install ./kernel-unstable-ogc-core-*.rpm ./kernel-unstable-ogc-modules-*.rpm' - echo '```' - echo - echo "Debian and derivatives:" - echo - echo '```sh' - echo 'sudo apt install ./linux-image-*.deb ./linux-headers-*.deb' - echo '```' - echo - echo "> The initramfs is generated automatically on install (mkinitcpio hooks" - echo "> on Arch, kernel-install/dracut on Fedora, initramfs-tools hooks on Debian)." - } > "$NOTES" - - cd /tmp/dist - # Upload every artifact produced by the distro jobs (package files - # plus the config-* files). SHA256SUMS itself is included by ./*. - gh release create "$TAG" \ - --repo "$GITHUB_REPOSITORY" \ - --target "$GITHUB_SHA" \ - --title "linux-unstable-ogc ${KREL}" \ - --notes-file "$NOTES" \ - ./* - - - name: Summary - run: | - { - echo "## linux-unstable-ogc build" - echo - echo "- Kernel release: \`${KREL}\`" - echo "- Release: https://github.com/${GITHUB_REPOSITORY}/releases/tag/${TAG}" - echo "- QEMU boot smoke test: passed (Arch, Fedora and Debian kernel images booted to userspace init)" - } >> "$GITHUB_STEP_SUMMARY" \ No newline at end of file + uses: ./.github/workflows/publish-release.yml + with: + tag: ${{ needs.prepare.outputs.tag }} + krel: ${{ needs.prepare.outputs.krel }} + kver: ${{ needs.prepare.outputs.kver }} + next: ${{ needs.prepare.outputs.next }} diff --git a/.github/workflows/publish-release.yml b/.github/workflows/publish-release.yml new file mode 100644 index 00000000000000..e9f641a6b36644 --- /dev/null +++ b/.github/workflows/publish-release.yml @@ -0,0 +1,157 @@ +# Release publisher, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml after all three distro builds +# (each gated on its QEMU boot smoke test) have succeeded. + +name: Publish linux-unstable-ogc release + +on: + workflow_call: + inputs: + tag: + description: Release tag, e.g. v6.12.0-next-20250101-g12345678 + type: string + required: true + krel: + description: Full kernel release string (uname -r) + type: string + required: true + kver: + description: Base kernel version from `make kernelversion` + type: string + required: true + next: + description: localversion-next suffix (may be empty) + type: string + required: true + +permissions: + contents: write + +env: + # Config fragments maintained by the OGC kernel-packages repository + # (referenced in the release notes). + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + release: + name: Create GitHub release + runs-on: ubuntu-latest + env: + TAG: ${{ inputs.tag }} + KREL: ${{ inputs.krel }} + KVER: ${{ inputs.kver }} + NEXT: ${{ inputs.next }} + steps: + - name: Download package artifacts + uses: actions/download-artifact@v8 + with: + path: /tmp/dist + pattern: "*-packages" + + - name: Flatten artifact directory + run: | + set -euxo pipefail + cd /tmp/dist + find . -mindepth 2 -maxdepth 2 -type f -exec mv -t . {} + + find . -mindepth 1 -type d -delete + ls -lh + + - name: Generate checksums + run: | + set -euxo pipefail + cd /tmp/dist + sha256sum * > SHA256SUMS + cat SHA256SUMS + + - name: Create GitHub release and upload packages + env: + GH_TOKEN: ${{ github.token }} + run: | + set -euxo pipefail + + # Idempotency: if a stale release exists for this tag (e.g. re-run + # of the same commit), remove it so this run can recreate it. + if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" 2>/dev/null; then + echo "Deleting existing release $TAG, will recreate it" + gh release delete "$TAG" --repo "$GITHUB_REPOSITORY" --yes --cleanup-tag + fi + # In case a tag without a release is left over, drop it too. + gh api -X DELETE "repos/${GITHUB_REPOSITORY}/git/refs/tags/${TAG}" >/dev/null 2>&1 || true + + NOTES="$(mktemp)" + { + echo "Automated build of linux-unstable-ogc." + echo + echo "- Source commit: https://github.com/${GITHUB_REPOSITORY}/commit/${GITHUB_SHA}" + echo "- Kernel release: ${KREL}" + echo "- Upstream version: ${KVER}${NEXT}" + echo "- Base configs: Arch \`linux-headers\` + Fedora \`kernel-core\`, plus [OGC kernel-packages fragments](${KERNEL_PACKAGES_RAW}/config)" + echo "- Compiler: clang / LLVM=1 (with ccache)" + echo "- Boot smoke test: every packaged kernel (Arch, Fedora, Debian) was booted in QEMU and reached userspace init before publishing" + echo + echo "### Artifacts" + echo + echo "#### Arch Linux" + echo + echo '- `linux-unstable-ogc` — kernel image and modules' + echo '- `linux-unstable-ogc-headers` — headers for building external modules' + echo "- \`config-arch-${KREL}\` — the exact .config used for this build" + echo + echo "#### Fedora" + echo + echo '- `kernel-unstable-ogc-core` — kernel image (vmlinuz) and core files' + echo '- `kernel-unstable-ogc-modules` — kernel modules' + echo '- `kernel-unstable-ogc-devel` — headers for building external modules' + echo "- \`config-fedora-${KREL}\` — the exact .config used for this build" + echo + echo "#### Debian (and derivatives)" + echo + echo "- \`linux-image-${KREL}\` — kernel image and modules" + echo "- \`linux-headers-${KREL}\` — headers for building external modules" + echo "- \`config-debian-${KREL}\` — the exact .config used for this build" + echo + echo '- `SHA256SUMS` — checksums of all artifacts' + echo + echo "### Install" + echo + echo "Arch Linux:" + echo + echo '```sh' + echo 'sudo pacman -U linux-unstable-ogc-headers-*.pkg.tar.zst linux-unstable-ogc-*.pkg.tar.zst' + echo '```' + echo + echo "Fedora:" + echo + echo '```sh' + echo 'sudo dnf install ./kernel-unstable-ogc-core-*.rpm ./kernel-unstable-ogc-modules-*.rpm' + echo '```' + echo + echo "Debian and derivatives:" + echo + echo '```sh' + echo 'sudo apt install ./linux-image-*.deb ./linux-headers-*.deb' + echo '```' + echo + echo "> The initramfs is generated automatically on install (mkinitcpio hooks" + echo "> on Arch, kernel-install/dracut on Fedora, initramfs-tools hooks on Debian)." + } > "$NOTES" + + cd /tmp/dist + # Upload every artifact produced by the distro jobs (package files + # plus the config-* files). SHA256SUMS itself is included by ./*. + gh release create "$TAG" \ + --repo "$GITHUB_REPOSITORY" \ + --target "$GITHUB_SHA" \ + --title "linux-unstable-ogc ${KREL}" \ + --notes-file "$NOTES" \ + ./* + + - name: Summary + run: | + { + echo "## linux-unstable-ogc build" + echo + echo "- Kernel release: \`${KREL}\`" + echo "- Release: https://github.com/${GITHUB_REPOSITORY}/releases/tag/${TAG}" + echo "- QEMU boot smoke test: passed (Arch, Fedora and Debian kernel images booted to userspace init)" + } >> "$GITHUB_STEP_SUMMARY" From 5be7813f340baa3b2cac9e8297fea101f2d8f35b Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Sat, 5 Sep 2026 13:39:38 +0000 Subject: [PATCH 853/857] ogc: linux-unstable: boot smoke test: pin loglevel=7 and dump serial log head on banner failure The smoke test greps the captured serial console log for the kernel banner ("Linux version "), but the banner is printed at KERN_NOTICE while some distro configs default the console to a quieter level: Arch ships CONFIG_CONSOLE_LOGLEVEL_DEFAULT=4, which suppresses notice (and info) messages entirely. The result was a confusing failure mode: the kernel booted to userspace just fine (BOOT_OK marker, printed by init directly on /dev/ttyS0), yet the banner check failed because the console log only carried a few high-priority lines. Pin loglevel=7 on the test kernel command line so every distro kernel logs verbosely enough for the banner to be captured, and replace the useless "^Linux version" grep in the failure path (banner lines are prefixed with a "[ 0.000000] " timestamp, so that grep could never match) with a dump of the first serial console lines. Verified locally against a kernel built with the exact Arch config pipeline: the run failed identically to CI before the change and passes both the bios-pc and uefi-q35 legs after it. --- .github/packaging/boot-smoke-test.sh | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/.github/packaging/boot-smoke-test.sh b/.github/packaging/boot-smoke-test.sh index 021e8e596998f3..f315029982fd72 100755 --- a/.github/packaging/boot-smoke-test.sh +++ b/.github/packaging/boot-smoke-test.sh @@ -34,6 +34,10 @@ # "Linux version " must appear on every console log. # BOOT_SMOKE_TIMEOUT override the per-boot timeout in seconds # (default: 300 under KVM, 1200 under TCG). +# The kernel command line pins loglevel=7: some distro configs default the +# console to a quieter level (Arch ships CONSOLE_LOGLEVEL_DEFAULT=4), which +# would suppress the KERN_NOTICE boot banner and make the banner check below +# fail on an otherwise perfectly bootable kernel. # Requires (installed by the calling CI job): qemu-system-x86_64, cpio, # gzip, a C compiler (gcc or clang), an OVMF build for the UEFI # leg (packages: ovmf / edk2-ovmf), coreutils (timeout, find). @@ -190,7 +194,7 @@ run_boot() { -display none -monitor none -no-reboot \ -serial "file:$serial" \ -kernel "$IMAGE" -initrd "$WORK/initrd.img" \ - -append "console=ttyS0,115200n8 rdinit=/init panic=-1 nokaslr" \ + -append "console=ttyS0,115200n8 rdinit=/init panic=-1 nokaslr loglevel=7" \ || true # A panic with panic=-1 reboots instantly and -no-reboot makes QEMU # exit, so both "qemu exited by itself" and "timeout killed it" end up @@ -205,7 +209,12 @@ run_boot() { fi if [ -n "${KREL:-}" ] && ! grep -qF "Linux version ${KREL} " "$serial"; then echo "::error::QEMU boot smoke test FAILED in the '$name' configuration: the booted kernel banner does not advertise release '${KREL}'." - grep -m1 "^Linux version" "$serial" || true + # Show what the console actually carried: log lines carry a + # "[ 0.000000] " timestamp prefix, so anchor-free context of + # the early console output is what makes this diagnosable. + echo "----- first 25 lines of the $name guest serial console -----" + head -n 25 "$serial" | sed 's/\r$//' + echo "-------------------------------------------------------------" exit 1 fi echo "[$name] Boot smoke test PASSED: kernel booted to userspace init and powered off. Serial console tail:" From 5fc295bbd217514201a15e14bf51ae9bc098d19a Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Fri, 4 Sep 2026 09:57:43 +0200 Subject: [PATCH 854/857] platform/x86: asus-wmi: Serialize WMI method evaluations with a mutex Concurrent evaluations of ASUS WMI management methods (from ACPI notify, HID, userspace daemons, and debugfs) enter the BIOS ACPI/SMM interface simultaneously, triggering re-entrant SMIs or EC mailbox buffer corruption. Fix this at the root by introducing a centralized evaluation helper (asus_wmi_evaluate_method_locked()) protected by a global mutex (asus_wmi_eval_lock) using guard(mutex). Route all evaluations of ASUS_WMI_MGMT_GUID (method3, method5, method_buf, and show_call) through this helper. A static mutex is required because asus_wmi_evaluate_method() is an exported symbol used by external modules (such as hid-asus and asus-armoury) that lack access to struct asus_wmi drvdata, and the underlying ASUS ACPI/EC management method is a single physical platform resource. The mutex is non-recursive: nested ACPI/WMI notify handlers must not call back into evaluate on the same task (defer via workqueue, as asus_rfkill_notify already does). Link: https://github.com/OpenGamingCollective/asusctl/issues/328 Fixes: ffb6ce7086ee ("platform/x86: asus-wmi: export function for evaluating WMI methods") Cc: stable@vger.kernel.org Signed-off-by: Marco Scardovi --- drivers/platform/x86/asus-wmi.c | 31 ++++++++++++++++++++++--------- 1 file changed, 22 insertions(+), 9 deletions(-) diff --git a/drivers/platform/x86/asus-wmi.c b/drivers/platform/x86/asus-wmi.c index a65090429ca703..065184176c0958 100644 --- a/drivers/platform/x86/asus-wmi.c +++ b/drivers/platform/x86/asus-wmi.c @@ -16,6 +16,7 @@ #include #include #include +#include #include #include #include @@ -28,6 +29,7 @@ #include #include #include +#include #include #include #include @@ -353,6 +355,21 @@ static void asus_wmi_show_deprecated(void) /* WMI ************************************************************************/ +/* + * Serializes all evaluations of ASUS_WMI_MGMT_GUID methods. + * Non-recursive: nested ACPI/WMI notify handlers must not call back into + * evaluate on the same task — defer via workqueue (see asus_rfkill_notify). + */ +static DEFINE_MUTEX(asus_wmi_eval_lock); + +static acpi_status asus_wmi_evaluate_method_locked(u32 method_id, + struct acpi_buffer *input, + struct acpi_buffer *output) +{ + guard(mutex)(&asus_wmi_eval_lock); + return wmi_evaluate_method(ASUS_WMI_MGMT_GUID, 0, method_id, input, output); +} + static int asus_wmi_evaluate_method3(u32 method_id, u32 arg0, u32 arg1, u32 arg2, u32 *retval) { @@ -367,8 +384,7 @@ static int asus_wmi_evaluate_method3(u32 method_id, union acpi_object *obj; u32 tmp = 0; - status = wmi_evaluate_method(ASUS_WMI_MGMT_GUID, 0, method_id, - &input, &output); + status = asus_wmi_evaluate_method_locked(method_id, &input, &output); pr_debug("%s called (0x%08x) with args: 0x%08x, 0x%08x, 0x%08x\n", __func__, method_id, arg0, arg1, arg2); @@ -419,8 +435,7 @@ static int asus_wmi_evaluate_method5(u32 method_id, union acpi_object *obj; u32 tmp = 0; - status = wmi_evaluate_method(ASUS_WMI_MGMT_GUID, 0, method_id, - &input, &output); + status = asus_wmi_evaluate_method_locked(method_id, &input, &output); pr_debug("%s called (0x%08x) with args: 0x%08x, 0x%08x, 0x%08x, 0x%08x, 0x%08x\n", __func__, method_id, arg0, arg1, arg2, arg3, arg4); @@ -467,8 +482,7 @@ static int asus_wmi_evaluate_method_buf(u32 method_id, union acpi_object *obj; int err = 0; - status = wmi_evaluate_method(ASUS_WMI_MGMT_GUID, 0, method_id, - &input, &output); + status = asus_wmi_evaluate_method_locked(method_id, &input, &output); pr_debug("%s called (0x%08x) with args: 0x%08x, 0x%08x\n", __func__, method_id, arg0, arg1); @@ -5026,9 +5040,8 @@ static int show_call(struct seq_file *m, void *data) union acpi_object *obj; acpi_status status; - status = wmi_evaluate_method(ASUS_WMI_MGMT_GUID, - 0, asus->debug.method_id, - &input, &output); + status = asus_wmi_evaluate_method_locked(asus->debug.method_id, + &input, &output); if (ACPI_FAILURE(status)) return -EIO; From eab03da630b2c15e5186951bcc76d6d05af311aa Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Fri, 4 Sep 2026 09:58:30 +0200 Subject: [PATCH 855/857] platform/x86: asus-wmi: Remove redundant per-device wmi_lock and duplicate rfkill ops With all WMI method evaluations serialized globally by asus_wmi_eval_lock in asus_wmi_evaluate_method_locked(), the per-device wmi_lock in struct asus_wmi is completely redundant. Remove wmi_lock from struct asus_wmi, its initialization in asus_wmi_rfkill_init(), and its manual locking in asus_rfkill_hotplug(). Consequently, asus_rfkill_wlan_set() becomes a simple pass-through to asus_rfkill_set(), rendering asus_rfkill_wlan_ops identical to asus_rfkill_ops. Drop asus_rfkill_wlan_set() and asus_rfkill_wlan_ops, allocating WLAN rfkill devices with &asus_rfkill_ops directly. Signed-off-by: Marco Scardovi --- drivers/platform/x86/asus-wmi.c | 37 ++------------------------------- 1 file changed, 2 insertions(+), 35 deletions(-) diff --git a/drivers/platform/x86/asus-wmi.c b/drivers/platform/x86/asus-wmi.c index 065184176c0958..f88f3a00a1fe8d 100644 --- a/drivers/platform/x86/asus-wmi.c +++ b/drivers/platform/x86/asus-wmi.c @@ -332,7 +332,6 @@ struct asus_wmi { struct hotplug_slot hotplug_slot; struct mutex hotplug_lock; - struct mutex wmi_lock; struct workqueue_struct *hotplug_workqueue; struct work_struct hotplug_work; @@ -2246,9 +2245,7 @@ static void asus_rfkill_hotplug(struct asus_wmi *asus) bool absent; u32 l; - mutex_lock(&asus->wmi_lock); blocked = asus_wlan_rfkill_blocked(asus); - mutex_unlock(&asus->wmi_lock); mutex_lock(&asus->hotplug_lock); pci_lock_rescan_remove(); @@ -2450,30 +2447,6 @@ static void asus_rfkill_query(struct rfkill *rfkill, void *data) rfkill_set_sw_state(priv->rfkill, !result); } -static int asus_rfkill_wlan_set(void *data, bool blocked) -{ - struct asus_rfkill *priv = data; - struct asus_wmi *asus = priv->asus; - int ret; - - /* - * This handler is enabled only if hotplug is enabled. - * In this case, the asus_wmi_set_devstate() will - * trigger a wmi notification and we need to wait - * this call to finish before being able to call - * any wmi method - */ - mutex_lock(&asus->wmi_lock); - ret = asus_rfkill_set(data, blocked); - mutex_unlock(&asus->wmi_lock); - return ret; -} - -static const struct rfkill_ops asus_rfkill_wlan_ops = { - .set_block = asus_rfkill_wlan_set, - .query = asus_rfkill_query, -}; - static const struct rfkill_ops asus_rfkill_ops = { .set_block = asus_rfkill_set, .query = asus_rfkill_query, @@ -2492,13 +2465,8 @@ static int asus_new_rfkill(struct asus_wmi *asus, arfkill->dev_id = dev_id; arfkill->asus = asus; - if (dev_id == ASUS_WMI_DEVID_WLAN && - asus->driver->quirks->hotplug_wireless) - *rfkill = rfkill_alloc(name, &asus->platform_device->dev, type, - &asus_rfkill_wlan_ops, arfkill); - else - *rfkill = rfkill_alloc(name, &asus->platform_device->dev, type, - &asus_rfkill_ops, arfkill); + *rfkill = rfkill_alloc(name, &asus->platform_device->dev, type, + &asus_rfkill_ops, arfkill); if (!*rfkill) return -EINVAL; @@ -2572,7 +2540,6 @@ static int asus_wmi_rfkill_init(struct asus_wmi *asus) int result = 0; mutex_init(&asus->hotplug_lock); - mutex_init(&asus->wmi_lock); result = asus_new_rfkill(asus, &asus->wlan, "asus-wlan", RFKILL_TYPE_WLAN, ASUS_WMI_DEVID_WLAN); From 550e21de4a50561a39fae4ab8e0a6cd7ba4aab92 Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Sun, 27 Sep 2026 10:09:17 +0200 Subject: [PATCH 856/857] [NOT-FOR-UPSTREAM] ogc: linux-unstable: skip commits already in linux-next The sync replay stops on the first cherry-pick that linux-next has since rewritten. Skip subjects listed in .github/sync-skip, drop patch-id equivalents and empty cherry-picks, and abort without pushing on the first conflict. Signed-off-by: Marco Scardovi --- .github/sync-skip | 11 ++++ .github/workflows/sync-linux-next.yml | 84 +++++++++++++++++++++++---- 2 files changed, 83 insertions(+), 12 deletions(-) create mode 100644 .github/sync-skip diff --git a/.github/sync-skip b/.github/sync-skip new file mode 100644 index 00000000000000..ab46510d337b35 --- /dev/null +++ b/.github/sync-skip @@ -0,0 +1,11 @@ +# Subjects of fork commits that must not be replayed. Upstream already has +# a newer version: a cherry-pick would conflict, or apply and duplicate code. +# One full subject per line. Empty lines and lines starting with # are ignored. +# After a successful sync those commits are gone, so unmatched lines do nothing. +[FROM-ML] drm/amd/display: Add 2.1 FreeSync support for AMD VSDB EDID Block +[FROM-ML] drm/edid: parse HDMI 2.1 gaming (ALLM/VRR) capabilities from HF-VSDB +[FROM-ML] drm/amd/display: Add HDMI 2.1 VRR support from HF-VSDB +[FROM-ML] drm/amd/display: Enable HDMI ALLM for Gaming-VRR +[FROM-ML] iommu/amd: Add PerfOpt IOMMU performance optimization support +[FROM-ML] drm/amdgpu: Enable PerfOpt IOMMU perf optimization when GPU in identity domain +[FROM-ML] HID: asus: do not send keyboard init reports to touchpads diff --git a/.github/workflows/sync-linux-next.yml b/.github/workflows/sync-linux-next.yml index 45b236c3f24cb6..46393a28e8c097 100644 --- a/.github/workflows/sync-linux-next.yml +++ b/.github/workflows/sync-linux-next.yml @@ -86,6 +86,22 @@ jobs: exit 1 fi + # Read before reset: the replay checks out upstream and the file + # is gone until its own commit is cherry-picked back. + SKIP_SUBJECTS="" + if [ -f .github/sync-skip ]; then + while IFS= read -r line || [ -n "$line" ]; do + case "$line" in + ''|\#*) continue ;; + esac + SKIP_SUBJECTS="${SKIP_SUBJECTS}"$'\n'"${line}" + done < .github/sync-skip + fi + + # Patch-ids equivalent to something already in linux-next. Limited + # to the fork range so the daily job does not hash all of history. + EQUIV_UPSTREAM="$(git cherry -v "$UP_BASE" master "$BASE" | awk '$1 == "-" { print $2 }')" + echo "Replaying ${N} commit(s):" for c in $MY_COMMITS; do echo " $c $(git show -s --format=%s "$c")" @@ -95,21 +111,65 @@ jobs: # Reset master to the new upstream tip, then replay the fork commits. git reset --hard "$UP_BASE" + subject_skipped() { + local subj="$1" line + while IFS= read -r line; do + [ -n "$line" ] || continue + [ "$line" = "$subj" ] && return 0 + done <<< "$SKIP_SUBJECTS" + return 1 + } + for c in $MY_COMMITS; do - echo "Cherry-picking: $c $(git show -s --format=%s "$c")" - - if ! git cherry-pick -x --allow-empty "$c"; then - git status || true - # If we're left mid-cherry-pick, abort to avoid a broken - # working state. Nothing was pushed, so master on origin is - # still intact. - if [ -f .git/CHERRY_PICK_HEAD ]; then - git cherry-pick --abort || true - fi - echo "::error::Cherry-pick failed for $c; nothing was pushed." - exit 1 + subj="$(git show -s --format=%s "$c")" + + if subject_skipped "$subj"; then + echo "Skipping $c (listed in .github/sync-skip): $subj" + continue fi + + case " $EQUIV_UPSTREAM " in + *" $c "*) + echo "Skipping $c (equivalent patch already in linux-next): $subj" + continue + ;; + esac + + echo "Cherry-picking: $c $subj" + + if git cherry-pick -x --allow-empty --empty=drop "$c"; then + continue + fi + + # --empty=drop did not consume it. An empty result is already upstream. + if [ -f .git/CHERRY_PICK_HEAD ] && + git diff --quiet && git diff --cached --quiet; then + echo "Cherry-pick of $c is empty; skipping." + git cherry-pick --skip + continue + fi + + echo "::group::Cherry-pick failed for $c" + echo "$subj" + git diff --name-only --diff-filter=U || true + git status || true + echo "::endgroup::" + if [ -f .git/CHERRY_PICK_HEAD ]; then + git cherry-pick --abort || true + fi + echo "::error::Cherry-pick failed for $c ($subj); nothing was pushed." + exit 1 done + NEW_MARKER="$(git log --format="%H" --grep="^ogc: linux-unstable: first commit$" HEAD)" + if [ -z "$NEW_MARKER" ]; then + echo "::error::Replay dropped the 'ogc: linux-unstable: first commit' marker; nothing was pushed." + exit 1 + fi + if ! git merge-base --is-ancestor "$UP_BASE" HEAD; then + echo "::error::Replay result does not contain upstream/master; nothing was pushed." + exit 1 + fi + echo "Done. Pushing updated master." git push --force-with-lease origin master From 904ababf3dcf29a193f9d53fe1c1a4200b627b01 Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Sun, 27 Sep 2026 09:59:40 +0200 Subject: [PATCH 857/857] [NOT-FOR-UPSTREAM] ogc: linux-unstable: pick up newer upstream MAINTAINERS Take the AXIADO TSADC entry and the ayaneo-ec ABI path from linux-next, and keep the AYANEO 3 controller entry. Signed-off-by: Marco Scardovi --- MAINTAINERS | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/MAINTAINERS b/MAINTAINERS index a5d869a534bc30..640d5b0dae85d1 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -4518,6 +4518,16 @@ F: Documentation/devicetree/bindings/spi/axiado,ax3000-spi.yaml F: drivers/spi/spi-axiado.c F: drivers/spi/spi-axiado.h +AXIADO TSADC DRIVER +M: Petar Stepanovic +M: Akhila Kavi +M: Prasad Bolisetty +L: linux-hwmon@vger.kernel.org +S: Supported +F: Documentation/devicetree/bindings/hwmon/axiado,ax3000-tsadc.yaml +F: Documentation/hwmon/axiado-tsadc.rst +F: drivers/hwmon/axiado-tsadc.c + AYANEO 3 CONTROLLER HID DRIVER M: Matías Martínez L: linux-input@vger.kernel.org @@ -4530,7 +4540,7 @@ AYANEO PLATFORM EC DRIVER M: Antheas Kapenekakis L: platform-driver-x86@vger.kernel.org S: Maintained -F: Documentation/ABI/testing/sysfs-platform-ayaneo +F: Documentation/ABI/testing/sysfs-platform-ayaneo-ec F: drivers/platform/x86/ayaneo-ec.c AZ6007 DVB DRIVER