diff options
500 files changed, 7430 insertions, 3369 deletions
diff --git a/.get_maintainer.ignore b/.get_maintainer.ignore index 5ad082b4dd03..d19b725fd803 100644 --- a/.get_maintainer.ignore +++ b/.get_maintainer.ignore @@ -4,6 +4,8 @@ Alyssa Rosenzweig <alyssa@rosenzweig.io> Askar Safin <safinaskar@gmail.com> Christoph Hellwig <hch@lst.de> Jeff Kirsher <jeffrey.t.kirsher@intel.com> +Johannes Berg <johannes.berg@intel.com> +Johannes Berg <johannes@sipsolutions.net> Marc Gonzalez <marc.w.gonzalez@free.fr> Nathan Chancellor <nathan@kernel.org> Ralf Baechle <ralf@linux-mips.org> @@ -211,7 +211,8 @@ Christophe Leroy <chleroy@kernel.org> <christophe.leroy@c-s.fr> Christophe Leroy <chleroy@kernel.org> <christophe.leroy@csgroup.eu> Christophe Leroy <chleroy@kernel.org> <christophe.leroy2@cs-soprasteria.com> Christophe Ricard <christophe.ricard@gmail.com> -Christopher Obbard <christopher.obbard@linaro.org> <chris.obbard@collabora.com> +Christopher Obbard <chris.obbard@oss.qualcomm.com> <chris.obbard@collabora.com> +Christopher Obbard <chris.obbard@oss.qualcomm.com> <christopher.obbard@linaro.org> Christoph Hellwig <hch@lst.de> Christoph Manszewski <c.manszewski@gmail.com> <christoph.manszewski@intel.com> Christoph Paasch <cpaasch@openai.com> <christoph.paasch@gmail.com> @@ -222,6 +223,7 @@ Chuck Lever <cel@kernel.org> <chuck.lever@oracle.com> Chuck Lever <cel@kernel.org> <cel@netapp.com> Chuck Lever <cel@kernel.org> <cel@citi.umich.edu> Claudiu Beznea <claudiu.beznea@tuxon.dev> <claudiu.beznea@microchip.com> +Coiby Xu <coiby.xu@gmail.com> <coxu@redhat.com> Colin Ian King <colin.i.king@gmail.com> <colin.king@canonical.com> Corey Minyard <minyard@acm.org> Damian Hobson-Garcia <dhobsong@igel.co.jp> diff --git a/Documentation/ABI/testing/sysfs-devices-platform-trackpoint b/Documentation/ABI/testing/sysfs-devices-platform-trackpoint index df11901a6b3d..7954434b0434 100644 --- a/Documentation/ABI/testing/sysfs-devices-platform-trackpoint +++ b/Documentation/ABI/testing/sysfs-devices-platform-trackpoint @@ -5,7 +5,7 @@ Contact: linux-input@vger.kernel.org Description: (RW) Trackpoint sensitivity. -What: /sys/devices/platform/i8042/.../intertia +What: /sys/devices/platform/i8042/.../inertia Date: Aug, 2005 KernelVersion: 2.6.14 Contact: linux-input@vger.kernel.org diff --git a/Documentation/arch/arm64/memory-tagging-extension.rst b/Documentation/arch/arm64/memory-tagging-extension.rst index e6fe428f0e2a..1d32fc5df183 100644 --- a/Documentation/arch/arm64/memory-tagging-extension.rst +++ b/Documentation/arch/arm64/memory-tagging-extension.rst @@ -208,11 +208,10 @@ will use the corresponding aligned address. tracer's space cannot be accessed or does not have valid tags. - ``-EPERM`` - the specified process cannot be traced. - ``-EIO`` - the tracee's address range cannot be accessed (e.g. invalid - address) and no tags copied. ``iov_len`` not updated. + address) or does not have valid tags (not mapped with the ``PROT_MTE`` + flag) and no tags copied. ``iov_len`` not updated. - ``-EFAULT`` - fault on accessing the tracer's memory (``struct iovec`` or ``iov_base`` buffer) and no tags copied. ``iov_len`` not updated. -- ``-EOPNOTSUPP`` - the tracee's address does not have valid tags (never - mapped with the ``PROT_MTE`` flag). ``iov_len`` not updated. **Note**: There are no transient errors for the requests above, so user programs should not retry in case of a non-zero system call return. diff --git a/Documentation/devicetree/bindings/display/msm/qcom,qcs8300-mdss.yaml b/Documentation/devicetree/bindings/display/msm/qcom,qcs8300-mdss.yaml index c41a86203e78..287cce4dbddc 100644 --- a/Documentation/devicetree/bindings/display/msm/qcom,qcs8300-mdss.yaml +++ b/Documentation/devicetree/bindings/display/msm/qcom,qcs8300-mdss.yaml @@ -150,7 +150,7 @@ examples: reg = <0>; dpu_intf0_out: endpoint { - remote-endpoint = <&mdss_dp0_in>; + remote-endpoint = <&mdss_dp0_in>; }; }; @@ -352,9 +352,9 @@ examples: }; port@1 { - reg = <1>; + reg = <1>; - mdss_dp_out: endpoint { }; + mdss_dp_out: endpoint { }; }; }; diff --git a/Documentation/devicetree/bindings/display/msm/qcom,sar2130p-mdss.yaml b/Documentation/devicetree/bindings/display/msm/qcom,sar2130p-mdss.yaml index 44c1bb9e4109..fb3d24bbac7d 100644 --- a/Documentation/devicetree/bindings/display/msm/qcom,sar2130p-mdss.yaml +++ b/Documentation/devicetree/bindings/display/msm/qcom,sar2130p-mdss.yaml @@ -248,9 +248,9 @@ examples: remote-endpoint = <&usb_dp_qmpphy_dp_in>; }; }; - }; + }; - dp_opp_table: opp-table { + dp_opp_table: opp-table { compatible = "operating-points-v2"; opp-162000000 { diff --git a/Documentation/devicetree/bindings/display/msm/qcom,sm6150-dpu.yaml b/Documentation/devicetree/bindings/display/msm/qcom,sm6150-dpu.yaml index b4f437172218..8cbf99f9fcba 100644 --- a/Documentation/devicetree/bindings/display/msm/qcom,sm6150-dpu.yaml +++ b/Documentation/devicetree/bindings/display/msm/qcom,sm6150-dpu.yaml @@ -81,7 +81,7 @@ examples: port@1 { reg = <1>; dpu_intf1_out: endpoint { - remote-endpoint = <&mdss_dsi0_in>; + remote-endpoint = <&mdss_dsi0_in>; }; }; }; @@ -90,18 +90,18 @@ examples: compatible = "operating-points-v2"; opp-19200000 { - opp-hz = /bits/ 64 <19200000>; - required-opps = <&rpmhpd_opp_low_svs>; + opp-hz = /bits/ 64 <19200000>; + required-opps = <&rpmhpd_opp_low_svs>; }; opp-25600000 { - opp-hz = /bits/ 64 <25600000>; - required-opps = <&rpmhpd_opp_svs>; + opp-hz = /bits/ 64 <25600000>; + required-opps = <&rpmhpd_opp_svs>; }; opp-307200000 { - opp-hz = /bits/ 64 <307200000>; - required-opps = <&rpmhpd_opp_nom>; + opp-hz = /bits/ 64 <307200000>; + required-opps = <&rpmhpd_opp_nom>; }; }; }; diff --git a/Documentation/devicetree/bindings/display/msm/qcom,sm8350-mdss.yaml b/Documentation/devicetree/bindings/display/msm/qcom,sm8350-mdss.yaml index 68176de854b3..f5acf528d0e4 100644 --- a/Documentation/devicetree/bindings/display/msm/qcom,sm8350-mdss.yaml +++ b/Documentation/devicetree/bindings/display/msm/qcom,sm8350-mdss.yaml @@ -223,7 +223,7 @@ examples: phys = <&mdss_dsi0_phy>; ports { - #address-cells = <1>; + #address-cells = <1>; #size-cells = <0>; port@0 { diff --git a/Documentation/devicetree/bindings/display/msm/qcom,x1e80100-mdss.yaml b/Documentation/devicetree/bindings/display/msm/qcom,x1e80100-mdss.yaml index 8d698a2e055a..8c65a80b46e4 100644 --- a/Documentation/devicetree/bindings/display/msm/qcom,x1e80100-mdss.yaml +++ b/Documentation/devicetree/bindings/display/msm/qcom,x1e80100-mdss.yaml @@ -208,47 +208,47 @@ examples: #sound-dai-cells = <0>; ports { - #address-cells = <1>; - #size-cells = <0>; + #address-cells = <1>; + #size-cells = <0>; - port@0 { - reg = <0>; + port@0 { + reg = <0>; - mdss_dp0_in: endpoint { - remote-endpoint = <&mdss_intf0_out>; - }; - }; + mdss_dp0_in: endpoint { + remote-endpoint = <&mdss_intf0_out>; + }; + }; - port@1 { - reg = <1>; + port@1 { + reg = <1>; - mdss_dp0_out: endpoint { - }; - }; + mdss_dp0_out: endpoint { + }; + }; }; mdss_dp0_opp_table: opp-table { - compatible = "operating-points-v2"; - - opp-160000000 { - opp-hz = /bits/ 64 <160000000>; - required-opps = <&rpmhpd_opp_low_svs>; - }; - - opp-270000000 { - opp-hz = /bits/ 64 <270000000>; - required-opps = <&rpmhpd_opp_svs>; - }; - - opp-540000000 { - opp-hz = /bits/ 64 <540000000>; - required-opps = <&rpmhpd_opp_svs_l1>; - }; - - opp-810000000 { - opp-hz = /bits/ 64 <810000000>; - required-opps = <&rpmhpd_opp_nom>; - }; + compatible = "operating-points-v2"; + + opp-160000000 { + opp-hz = /bits/ 64 <160000000>; + required-opps = <&rpmhpd_opp_low_svs>; + }; + + opp-270000000 { + opp-hz = /bits/ 64 <270000000>; + required-opps = <&rpmhpd_opp_svs>; + }; + + opp-540000000 { + opp-hz = /bits/ 64 <540000000>; + required-opps = <&rpmhpd_opp_svs_l1>; + }; + + opp-810000000 { + opp-hz = /bits/ 64 <810000000>; + required-opps = <&rpmhpd_opp_nom>; + }; }; }; }; diff --git a/Documentation/devicetree/bindings/input/mediatek,mt6779-keypad.yaml b/Documentation/devicetree/bindings/input/mediatek,mt6779-keypad.yaml index 914dd3283df3..20fa41f691ed 100644 --- a/Documentation/devicetree/bindings/input/mediatek,mt6779-keypad.yaml +++ b/Documentation/devicetree/bindings/input/mediatek,mt6779-keypad.yaml @@ -26,6 +26,7 @@ properties: - const: mediatek,mt6779-keypad - items: - enum: + - mediatek,mt6572-keypad - mediatek,mt6873-keypad - mediatek,mt8183-keypad - mediatek,mt8365-keypad diff --git a/Documentation/hwmon/cgbc-hwmon.rst b/Documentation/hwmon/cgbc-hwmon.rst index 3a5e6e6e8639..c6d09232392a 100644 --- a/Documentation/hwmon/cgbc-hwmon.rst +++ b/Documentation/hwmon/cgbc-hwmon.rst @@ -28,34 +28,44 @@ system. Name Description ============= ====================== temp1_input CPU temperature -temp2_input Box temperature +temp2_input Case temperature temp3_input Ambient temperature -temp4_input Board temperature -temp5_input Carrier temperature -temp6_input Chipset temperature -temp7_input Video temperature +temp4_input CPU Board temperature +temp5_input Carrier Board temperature +temp6_input System Chipset temperature +temp7_input Video Controller/Board temperature temp8_input Other temperature -temp9_input TOPDIM temperature -temp10_input BOTTOMDIM temperature -in0_input CPU voltage +temp9_input Top DIMM 0 temperature +temp10_input Bottom DIMM 0 temperature +temp11_input Alternate Board temperature +temp12_input Top DIMM 1 temperature +temp13_input Top DIMM 2 temperature +temp14_input Top DIMM 3 temperature +temp15_input Top DIMM 4 temperature +temp16_input Top DIMM 5 temperature +temp17_input Top DIMM 6 temperature +temp18_input Top DIMM 7 temperature +temp19_input Bottom DIMM 1 temperature +in0_input CPU Core voltage in1_input DC Runtime voltage in2_input DC Standby voltage in3_input CMOS Battery voltage -in4_input Battery voltage +in4_input Battery Supply voltage in5_input AC voltage in6_input Other voltage -in7_input 5V voltage +in7_input 5V Runtime voltage in8_input 5V Standby voltage -in9_input 3V3 voltage +in9_input 3V3 Runtime voltage in10_input 3V3 Standby voltage in11_input VCore A voltage in12_input VCore B voltage -in13_input 12V voltage -curr1_input DC current -curr2_input 5V current -curr3_input 12V current +in13_input 12V Runtime voltage +in14_input 12V Standby voltage +curr1_input DC Runtime current +curr2_input 5V Runtime current +curr3_input 12V Runtime current fan1_input CPU fan -fan2_input Box fan +fan2_input Case fan fan3_input Ambient fan fan4_input Chiptset fan fan5_input Video fan diff --git a/Documentation/netlink/specs/rt-neigh.yaml b/Documentation/netlink/specs/rt-neigh.yaml index 0f46ef313590..c8e55c98d564 100644 --- a/Documentation/netlink/specs/rt-neigh.yaml +++ b/Documentation/netlink/specs/rt-neigh.yaml @@ -341,6 +341,9 @@ attribute-sets: - name: interval-probe-time-ms type: u64 + checks: + min: 1 + max: 86400000 operations: enum-model: directional diff --git a/Documentation/networking/ip-sysctl.rst b/Documentation/networking/ip-sysctl.rst index 208f46967ee5..f7af0286341c 100644 --- a/Documentation/networking/ip-sysctl.rst +++ b/Documentation/networking/ip-sysctl.rst @@ -248,7 +248,7 @@ neigh/default/unres_qlen - INTEGER neigh/default/interval_probe_time_ms - INTEGER The probe interval for neighbor entries with NTF_MANAGED flag, - the min value is 1. + the min value is 1, and the max value is 86400000 (1 day). Default: 5000 @@ -874,6 +874,8 @@ tcp_rmem - vector of 3 INTEGERs: min, default, max case this value is ignored. Default: between 131072 and 32MB, depending on RAM size. + Each of the three values cannot be set below 4096. + tcp_sack - BOOLEAN Enable select acknowledgments (SACKS). diff --git a/MAINTAINERS b/MAINTAINERS index c2414447892c..cc3cae2e378b 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -9324,6 +9324,7 @@ F: Documentation/devicetree/bindings/interrupt-controller/econet,en751221-intc.y F: Documentation/devicetree/bindings/mips/econet.yaml F: Documentation/devicetree/bindings/timer/econet,en751221-timer.yaml F: arch/mips/boot/dts/econet/ +F: arch/mips/configs/econet_en751221_defconfig F: arch/mips/econet/ F: drivers/clocksource/timer-econet-en751221.c F: drivers/irqchip/irq-econet-en751221.c @@ -17031,8 +17032,8 @@ MEMBLOCK AND MEMORY MANAGEMENT INITIALIZATION M: Mike Rapoport <rppt@kernel.org> L: linux-mm@kvack.org S: Maintained -T: git git://git.kernel.org/pub/scm/linux/kernel/git/rppt/memblock.git for-next -T: git git://git.kernel.org/pub/scm/linux/kernel/git/rppt/memblock.git fixes +T: git git://git.kernel.org/pub/scm/linux/kernel/git/mm/memblock.git for-next +T: git git://git.kernel.org/pub/scm/linux/kernel/git/mm/memblock.git fixes F: Documentation/core-api/boot-time-mm.rst F: include/linux/kho/abi/memblock.h F: include/linux/memblock.h diff --git a/arch/arm/mach-socfpga/Kconfig b/arch/arm/mach-socfpga/Kconfig index eb72c240c248..528c5c1368c3 100644 --- a/arch/arm/mach-socfpga/Kconfig +++ b/arch/arm/mach-socfpga/Kconfig @@ -16,7 +16,7 @@ menuconfig ARCH_INTEL_SOCFPGA select ARM_ERRATA_775420 select PL310_ERRATA_588369 select PL310_ERRATA_727915 - select PL310_ERRATA_753970 if PL310 + select PL310_ERRATA_753970 select PL310_ERRATA_769419 select RESET_CONTROLLER diff --git a/arch/arm64/boot/dts/altera/socfpga_stratix10_socdk.dtsi b/arch/arm64/boot/dts/altera/socfpga_stratix10_socdk.dtsi index 1d50f7b21160..1d50f7b21160 100755..100644 --- a/arch/arm64/boot/dts/altera/socfpga_stratix10_socdk.dtsi +++ b/arch/arm64/boot/dts/altera/socfpga_stratix10_socdk.dtsi diff --git a/arch/arm64/boot/dts/altera/socfpga_stratix10_socdk_emmc.dts b/arch/arm64/boot/dts/altera/socfpga_stratix10_socdk_emmc.dts index b2a3449638dd..b2a3449638dd 100755..100644 --- a/arch/arm64/boot/dts/altera/socfpga_stratix10_socdk_emmc.dts +++ b/arch/arm64/boot/dts/altera/socfpga_stratix10_socdk_emmc.dts diff --git a/arch/arm64/boot/dts/amlogic/amlogic-t7-a311d2-an400.dts b/arch/arm64/boot/dts/amlogic/amlogic-t7-a311d2-an400.dts index cab2ee9ea0d3..ca7536f772ff 100644 --- a/arch/arm64/boot/dts/amlogic/amlogic-t7-a311d2-an400.dts +++ b/arch/arm64/boot/dts/amlogic/amlogic-t7-a311d2-an400.dts @@ -33,7 +33,7 @@ }; &uart_a { - clocks = <&xtal>, <&xtal>, <&xtal>; + clocks = <&xtal>, <&clkc_periphs CLKID_SYS_UART_A>, <&xtal>; clock-names = "xtal", "pclk", "baud"; status = "okay"; }; diff --git a/arch/arm64/boot/dts/amlogic/amlogic-t7-a311d2-khadas-vim4.dts b/arch/arm64/boot/dts/amlogic/amlogic-t7-a311d2-khadas-vim4.dts index c41525a34b72..77abfd555ec5 100644 --- a/arch/arm64/boot/dts/amlogic/amlogic-t7-a311d2-khadas-vim4.dts +++ b/arch/arm64/boot/dts/amlogic/amlogic-t7-a311d2-khadas-vim4.dts @@ -71,7 +71,6 @@ vin-supply = <&vddao_3v3>; gpio = <&gpio GPIOD_11 GPIO_ACTIVE_LOW>; regulator-boot-on; - regulator-always-on; }; sdio_pwrseq: sdio-pwrseq { @@ -105,6 +104,66 @@ enable-active-high; }; + vdd_ddr: regulator-vddddr { + /* + * SY8003ADFC Regulator. + */ + compatible = "pwm-regulator"; + regulator-name = "VDDDDR"; + regulator-min-microvolt = <690000>; + regulator-max-microvolt = <890000>; + pwm-supply = <&dc_in>; + pwms = <&pwm_ao_gh 0 1500 0>; + pwm-dutycycle-range = <100 0>; + regulator-boot-on; + regulator-always-on; + }; + + vdd_ee: regulator-vddee { + /* + * MP8756GD Regulator. + */ + compatible = "pwm-regulator"; + regulator-name = "VDDEE"; + regulator-min-microvolt = <700000>; + regulator-max-microvolt = <922000>; + pwm-supply = <&dc_in>; + pwms = <&pwm_ao_ab 0 1500 0>; + pwm-dutycycle-range = <100 0>; + regulator-boot-on; + regulator-always-on; + }; + + vdd_gpu: regulator-vddgpu { + /* + * SY8003ADFC Regulator. + */ + compatible = "pwm-regulator"; + regulator-name = "VDDGPU"; + regulator-min-microvolt = <700000>; + regulator-max-microvolt = <922000>; + pwm-supply = <&dc_in>; + pwms = <&pwm_ao_ef 0 1500 0>; + pwm-dutycycle-range = <100 0>; + regulator-boot-on; + regulator-always-on; + }; + + vdd_npu: regulator-vddnpu { + /* + * SY8003ADFC Regulator. + */ + compatible = "pwm-regulator"; + regulator-name = "VDDNPU"; + regulator-min-microvolt = <733000>; + regulator-max-microvolt = <933000>; + pwm-supply = <&dc_in>; + pwms = <&pwm_ao_ef 1 1500 0>; + pwm-dutycycle-range = <100 0>; + regulator-boot-on; + regulator-always-on; + }; + vddao_1v8: regulator-vddao-1v8 { compatible = "regulator-fixed"; regulator-name = "VDDAO_1V8"; @@ -123,6 +182,36 @@ regulator-always-on; }; + vddcpu_a: regulator-vddcpu-a { + /* + * MP8756GD Regulator. + */ + compatible = "pwm-regulator"; + regulator-name = "VDDCPU_A"; + regulator-min-microvolt = <689000>; + regulator-max-microvolt = <1049000>; + pwm-supply = <&dc_in>; + pwms = <&pwm_ao_cd 1 1500 0>; + pwm-dutycycle-range = <100 0>; + regulator-boot-on; + regulator-always-on; + }; + + vddcpu_b: regulator-vddcpu-b { + /* + * MP8756GD Regulator. + */ + compatible = "pwm-regulator"; + regulator-name = "VDDCPU_B"; + regulator-min-microvolt = <689000>; + regulator-max-microvolt = <1049000>; + pwm-supply = <&dc_in>; + pwms = <&pwm_ao_ab 1 1500 0>; + pwm-dutycycle-range = <100 0>; + regulator-boot-on; + regulator-always-on; + }; + vddio_1v8: regulator-vddio-1v8 { compatible = "regulator-fixed"; regulator-name = "VDDIO_1V8"; @@ -173,9 +262,27 @@ pinctrl-names = "default"; }; +&pwm_ao_ab { + status = "okay"; + pinctrl-0 = <&pwm_ao_a_pins>, <&pwm_ao_b_pins>; + pinctrl-names = "default"; +}; + &pwm_ao_cd { status = "okay"; - pinctrl-0 = <&pwm_ao_c_d_pins>; + pinctrl-0 = <&pwm_ao_c_d_pins>, <&pwm_ao_d_pins>; + pinctrl-names = "default"; +}; + +&pwm_ao_ef { + status = "okay"; + pinctrl-0 = <&pwm_ao_e_pins>, <&pwm_ao_f_pins>; + pinctrl-names = "default"; +}; + +&pwm_ao_gh { + status = "okay"; + pinctrl-0 = <&pwm_ao_g_e_pins>; pinctrl-names = "default"; }; @@ -266,6 +373,6 @@ &uart_a { status = "okay"; - clocks = <&xtal>, <&xtal>, <&xtal>; + clocks = <&xtal>, <&clkc_periphs CLKID_SYS_UART_A>, <&xtal>; clock-names = "xtal", "pclk", "baud"; }; diff --git a/arch/arm64/boot/dts/amlogic/amlogic-t7.dtsi b/arch/arm64/boot/dts/amlogic/amlogic-t7.dtsi index c3dc479b137d..719e111bc3dd 100644 --- a/arch/arm64/boot/dts/amlogic/amlogic-t7.dtsi +++ b/arch/arm64/boot/dts/amlogic/amlogic-t7.dtsi @@ -458,9 +458,25 @@ }; }; - pwm_ao_g_pins: pwm-ao-g { + pwm_ao_g_d11_pins: pwm-ao-g-d11 { mux { - groups = "pwm_ao_g"; + groups = "pwm_ao_g_d11"; + function = "pwm_ao_g"; + bias-disable; + }; + }; + + pwm_ao_g_d7_pins: pwm-ao-g-d7 { + mux { + groups = "pwm_ao_g_d7"; + function = "pwm_ao_g"; + bias-disable; + }; + }; + + pwm_ao_g_e_pins: pwm-ao-g-e { + mux { + groups = "pwm_ao_g_e"; function = "pwm_ao_g"; bias-disable; }; @@ -474,9 +490,17 @@ }; }; - pwm_ao_h_pins: pwm-ao-h { + pwm_ao_h_d5_pins: pwm-ao-h-d5 { mux { - groups = "pwm_ao_h"; + groups = "pwm_ao_h_d5"; + function = "pwm_ao_h"; + bias-disable; + }; + }; + + pwm_ao_h_d10_pins: pwm-ao-h-d10 { + mux { + groups = "pwm_ao_h_d10"; function = "pwm_ao_h"; bias-disable; }; @@ -522,9 +546,17 @@ }; }; - pwm_vs_pins: pwm-vs { + pwm_vs_y_pins: pwm-vs-y { + mux { + groups = "pwm_vs_y"; + function = "pwm_vs"; + bias-disable; + }; + }; + + pwm_vs_h_pins: pwm-vs-h { mux { - groups = "pwm_vs"; + groups = "pwm_vs_h"; function = "pwm_vs"; bias-disable; }; diff --git a/arch/arm64/boot/dts/renesas/r8a779f0.dtsi b/arch/arm64/boot/dts/renesas/r8a779f0.dtsi index cbb161c863ac..a5118c2f2742 100644 --- a/arch/arm64/boot/dts/renesas/r8a779f0.dtsi +++ b/arch/arm64/boot/dts/renesas/r8a779f0.dtsi @@ -901,6 +901,7 @@ clocks = <&cpg CPG_MOD 1514>, <&ufs30_clk>; clock-names = "fck", "ref_clk"; freq-table-hz = <200000000 200000000>, <38400000 38400000>; + lanes-per-direction = <1>; power-domains = <&sysc R8A779F0_PD_ALWAYS_ON>; resets = <&cpg 1514>; status = "disabled"; diff --git a/arch/arm64/boot/dts/renesas/r9a09g047.dtsi b/arch/arm64/boot/dts/renesas/r9a09g047.dtsi index 73757e8e2197..060405d4a2dd 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g047.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g047.dtsi @@ -1913,23 +1913,28 @@ mtl_tx_setup0: tx-queues-config { snps,tx-queues-to-use = <4>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; snps,priority = <0x1>; }; queue1 { + snps,weight = <0x12>; snps,dcb-algorithm; snps,priority = <0x2>; }; queue2 { + snps,weight = <0x14>; snps,dcb-algorithm; snps,priority = <0x4>; }; queue3 { + snps,weight = <0x18>; snps,dcb-algorithm; snps,priority = <0x8>; }; @@ -2013,23 +2018,28 @@ mtl_tx_setup1: tx-queues-config { snps,tx-queues-to-use = <4>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; snps,priority = <0x1>; }; queue1 { + snps,weight = <0x12>; snps,dcb-algorithm; snps,priority = <0x2>; }; queue2 { + snps,weight = <0x14>; snps,dcb-algorithm; snps,priority = <0x4>; }; queue3 { + snps,weight = <0x18>; snps,dcb-algorithm; snps,priority = <0x8>; }; diff --git a/arch/arm64/boot/dts/renesas/r9a09g056.dtsi b/arch/arm64/boot/dts/renesas/r9a09g056.dtsi index 76fa34ff3d07..77c2221a9e2a 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g056.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g056.dtsi @@ -1585,23 +1585,28 @@ mtl_tx_setup0: tx-queues-config { snps,tx-queues-to-use = <4>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; snps,priority = <0x1>; }; queue1 { + snps,weight = <0x12>; snps,dcb-algorithm; snps,priority = <0x2>; }; queue2 { + snps,weight = <0x14>; snps,dcb-algorithm; snps,priority = <0x4>; }; queue3 { + snps,weight = <0x18>; snps,dcb-algorithm; snps,priority = <0x8>; }; @@ -1686,23 +1691,28 @@ mtl_tx_setup1: tx-queues-config { snps,tx-queues-to-use = <4>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; snps,priority = <0x1>; }; queue1 { + snps,weight = <0x12>; snps,dcb-algorithm; snps,priority = <0x2>; }; queue2 { + snps,weight = <0x14>; snps,dcb-algorithm; snps,priority = <0x4>; }; queue3 { + snps,weight = <0x18>; snps,dcb-algorithm; snps,priority = <0x8>; }; diff --git a/arch/arm64/boot/dts/renesas/r9a09g057.dtsi b/arch/arm64/boot/dts/renesas/r9a09g057.dtsi index 639693d464a7..188ce9f9c7c2 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g057.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g057.dtsi @@ -1715,23 +1715,28 @@ mtl_tx_setup0: tx-queues-config { snps,tx-queues-to-use = <4>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; snps,priority = <0x1>; }; queue1 { + snps,weight = <0x12>; snps,dcb-algorithm; snps,priority = <0x2>; }; queue2 { + snps,weight = <0x14>; snps,dcb-algorithm; snps,priority = <0x4>; }; queue3 { + snps,weight = <0x18>; snps,dcb-algorithm; snps,priority = <0x8>; }; @@ -1816,23 +1821,28 @@ mtl_tx_setup1: tx-queues-config { snps,tx-queues-to-use = <4>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; snps,priority = <0x1>; }; queue1 { + snps,weight = <0x12>; snps,dcb-algorithm; snps,priority = <0x2>; }; queue2 { + snps,weight = <0x14>; snps,dcb-algorithm; snps,priority = <0x4>; }; queue3 { + snps,weight = <0x18>; snps,dcb-algorithm; snps,priority = <0x8>; }; diff --git a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi index 40494159831d..bac39390ead7 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g077.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g077.dtsi @@ -642,36 +642,45 @@ mtl_tx_setup0: tx-queues-config { snps,tx-queues-to-use = <8>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; }; queue1 { + snps,weight = <0x11>; snps,dcb-algorithm; }; queue2 { + snps,weight = <0x12>; snps,dcb-algorithm; }; queue3 { + snps,weight = <0x13>; snps,dcb-algorithm; }; queue4 { + snps,weight = <0x14>; snps,dcb-algorithm; }; queue5 { + snps,weight = <0x15>; snps,dcb-algorithm; }; queue6 { + snps,weight = <0x16>; snps,dcb-algorithm; }; queue7 { + snps,weight = <0x17>; snps,dcb-algorithm; }; }; @@ -788,36 +797,45 @@ mtl_tx_setup1: tx-queues-config { snps,tx-queues-to-use = <8>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; }; queue1 { + snps,weight = <0x11>; snps,dcb-algorithm; }; queue2 { + snps,weight = <0x12>; snps,dcb-algorithm; }; queue3 { + snps,weight = <0x13>; snps,dcb-algorithm; }; queue4 { + snps,weight = <0x14>; snps,dcb-algorithm; }; queue5 { + snps,weight = <0x15>; snps,dcb-algorithm; }; queue6 { + snps,weight = <0x16>; snps,dcb-algorithm; }; queue7 { + snps,weight = <0x17>; snps,dcb-algorithm; }; }; @@ -934,36 +952,45 @@ mtl_tx_setup2: tx-queues-config { snps,tx-queues-to-use = <8>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; }; queue1 { + snps,weight = <0x11>; snps,dcb-algorithm; }; queue2 { + snps,weight = <0x12>; snps,dcb-algorithm; }; queue3 { + snps,weight = <0x13>; snps,dcb-algorithm; }; queue4 { + snps,weight = <0x14>; snps,dcb-algorithm; }; queue5 { + snps,weight = <0x15>; snps,dcb-algorithm; }; queue6 { + snps,weight = <0x16>; snps,dcb-algorithm; }; queue7 { + snps,weight = <0x17>; snps,dcb-algorithm; }; }; diff --git a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi index e8d4f76949cc..03b976d93e10 100644 --- a/arch/arm64/boot/dts/renesas/r9a09g087.dtsi +++ b/arch/arm64/boot/dts/renesas/r9a09g087.dtsi @@ -643,36 +643,45 @@ mtl_tx_setup0: tx-queues-config { snps,tx-queues-to-use = <8>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; }; queue1 { + snps,weight = <0x11>; snps,dcb-algorithm; }; queue2 { + snps,weight = <0x12>; snps,dcb-algorithm; }; queue3 { + snps,weight = <0x13>; snps,dcb-algorithm; }; queue4 { + snps,weight = <0x14>; snps,dcb-algorithm; }; queue5 { + snps,weight = <0x15>; snps,dcb-algorithm; }; queue6 { + snps,weight = <0x16>; snps,dcb-algorithm; }; queue7 { + snps,weight = <0x17>; snps,dcb-algorithm; }; }; @@ -790,36 +799,45 @@ mtl_tx_setup1: tx-queues-config { snps,tx-queues-to-use = <8>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; }; queue1 { + snps,weight = <0x11>; snps,dcb-algorithm; }; queue2 { + snps,weight = <0x12>; snps,dcb-algorithm; }; queue3 { + snps,weight = <0x13>; snps,dcb-algorithm; }; queue4 { + snps,weight = <0x14>; snps,dcb-algorithm; }; queue5 { + snps,weight = <0x15>; snps,dcb-algorithm; }; queue6 { + snps,weight = <0x16>; snps,dcb-algorithm; }; queue7 { + snps,weight = <0x17>; snps,dcb-algorithm; }; }; @@ -937,36 +955,45 @@ mtl_tx_setup2: tx-queues-config { snps,tx-queues-to-use = <8>; + snps,tx-sched-wrr; queue0 { + snps,weight = <0x10>; snps,dcb-algorithm; }; queue1 { + snps,weight = <0x11>; snps,dcb-algorithm; }; queue2 { + snps,weight = <0x12>; snps,dcb-algorithm; }; queue3 { + snps,weight = <0x13>; snps,dcb-algorithm; }; queue4 { + snps,weight = <0x14>; snps,dcb-algorithm; }; queue5 { + snps,weight = <0x15>; snps,dcb-algorithm; }; queue6 { + snps,weight = <0x16>; snps,dcb-algorithm; }; queue7 { + snps,weight = <0x17>; snps,dcb-algorithm; }; }; diff --git a/arch/arm64/include/asm/percpu.h b/arch/arm64/include/asm/percpu.h index b57b2bb00967..8cf4068ce1b5 100644 --- a/arch/arm64/include/asm/percpu.h +++ b/arch/arm64/include/asm/percpu.h @@ -77,7 +77,7 @@ __percpu_##name##_case_##sz(void *ptr, unsigned long val) \ " stxr" #sfx "\t%w[loop], %" #w "[tmp], %[ptr]\n" \ " cbnz %w[loop], 1b", \ /* LSE atomics */ \ - #op_lse "\t%" #w "[val], %" #w "[tmp], %[ptr]\n" \ + #op_lse #sfx "\t%" #w "[val], %" #w "[tmp], %[ptr]\n" \ __nops(3)) \ : [loop] "=&r" (loop), [tmp] "=&r" (tmp), \ [ptr] "+Q"(*(u##sz *)ptr) \ @@ -98,7 +98,7 @@ __percpu_##name##_return_case_##sz(void *ptr, unsigned long val) \ " stxr" #sfx "\t%w[loop], %" #w "[ret], %[ptr]\n" \ " cbnz %w[loop], 1b", \ /* LSE atomics */ \ - #op_lse "\t%" #w "[val], %" #w "[ret], %[ptr]\n" \ + #op_lse #sfx "\t%" #w "[val], %" #w "[ret], %[ptr]\n" \ #op_llsc "\t%" #w "[ret], %" #w "[ret], %" #w "[val]\n" \ __nops(2)) \ : [loop] "=&r" (loop), [ret] "=&r" (ret), \ @@ -179,13 +179,13 @@ PERCPU_RET_OP(add, add, ldadd) _pcp_protect_return(__percpu_read_64, pcp) #define this_cpu_write_1(pcp, val) \ - _pcp_protect(__percpu_write_8, pcp, (unsigned long)val) + _pcp_protect(__percpu_write_8, pcp, (unsigned long)(val)) #define this_cpu_write_2(pcp, val) \ - _pcp_protect(__percpu_write_16, pcp, (unsigned long)val) + _pcp_protect(__percpu_write_16, pcp, (unsigned long)(val)) #define this_cpu_write_4(pcp, val) \ - _pcp_protect(__percpu_write_32, pcp, (unsigned long)val) + _pcp_protect(__percpu_write_32, pcp, (unsigned long)(val)) #define this_cpu_write_8(pcp, val) \ - _pcp_protect(__percpu_write_64, pcp, (unsigned long)val) + _pcp_protect(__percpu_write_64, pcp, (unsigned long)(val)) #define this_cpu_add_1(pcp, val) \ _pcp_protect(__percpu_add_case_8, pcp, val) @@ -206,13 +206,13 @@ PERCPU_RET_OP(add, add, ldadd) _pcp_protect_return(__percpu_add_return_case_64, pcp, val) #define this_cpu_and_1(pcp, val) \ - _pcp_protect(__percpu_andnot_case_8, pcp, ~val) + _pcp_protect(__percpu_andnot_case_8, pcp, ~(u8)(val)) #define this_cpu_and_2(pcp, val) \ - _pcp_protect(__percpu_andnot_case_16, pcp, ~val) + _pcp_protect(__percpu_andnot_case_16, pcp, ~(u16)(val)) #define this_cpu_and_4(pcp, val) \ - _pcp_protect(__percpu_andnot_case_32, pcp, ~val) + _pcp_protect(__percpu_andnot_case_32, pcp, ~(u32)(val)) #define this_cpu_and_8(pcp, val) \ - _pcp_protect(__percpu_andnot_case_64, pcp, ~val) + _pcp_protect(__percpu_andnot_case_64, pcp, ~(u64)(val)) #define this_cpu_or_1(pcp, val) \ _pcp_protect(__percpu_or_case_8, pcp, val) diff --git a/arch/arm64/kernel/hibernate-asm.S b/arch/arm64/kernel/hibernate-asm.S index 0e1d9c3c6a93..2baefe7a82d3 100644 --- a/arch/arm64/kernel/hibernate-asm.S +++ b/arch/arm64/kernel/hibernate-asm.S @@ -89,6 +89,8 @@ alternative_insn "dc cvau, x4", "dc civac, x4", ARM64_WORKAROUND_CLEAN_CACHE isb cbz x24, 3f /* Do we need to re-initialise EL2? */ + mov x1, x24 + mov x0, #HVC_SET_VECTORS hvc #0 3: ret SYM_CODE_END(swsusp_arch_suspend_exit) diff --git a/arch/arm64/kernel/hibernate.c b/arch/arm64/kernel/hibernate.c index 7bf117427777..424291c547f0 100644 --- a/arch/arm64/kernel/hibernate.c +++ b/arch/arm64/kernel/hibernate.c @@ -423,8 +423,8 @@ int __nocfi swsusp_arch_resume(void) * Create a second copy of just the linear map, and use this when * restoring. */ - rc = trans_pgd_create_copy(&trans_info, &tmp_pg_dir, PAGE_OFFSET, - PAGE_END); + rc = trans_pgd_create_copy(&trans_info, &tmp_pg_dir, + _PAGE_OFFSET(vabits_actual), PAGE_END); if (rc) return rc; diff --git a/arch/arm64/kernel/mte.c b/arch/arm64/kernel/mte.c index 1a9aad6ef22a..31f5c6b0510e 100644 --- a/arch/arm64/kernel/mte.c +++ b/arch/arm64/kernel/mte.c @@ -476,7 +476,7 @@ static int __access_remote_tags(struct mm_struct *mm, unsigned long addr, * was never mapped with PROT_MTE. */ if (!(vma->vm_flags & VM_MTE)) { - err = -EOPNOTSUPP; + err = -EIO; put_page(page); break; } diff --git a/arch/mips/Kconfig b/arch/mips/Kconfig index e2eb9627bd14..d7c67cebe065 100644 --- a/arch/mips/Kconfig +++ b/arch/mips/Kconfig @@ -396,10 +396,7 @@ config ECONET bool "EcoNet MIPS family" select BOOT_RAW select DEBUG_ZBOOT if DEBUG_KERNEL - select EARLY_PRINTK_8250 select ECONET_EN751221_TIMER - select SERIAL_8250 - select SERIAL_OF_PLATFORM select SYS_SUPPORTS_BIG_ENDIAN select SYS_HAS_CPU_MIPS32_R1 select SYS_HAS_CPU_MIPS32_R2 @@ -661,6 +658,7 @@ config EYEQ select USB_UHCI_BIG_ENDIAN_MMIO if CPU_BIG_ENDIAN select USE_OF select HOTPLUG_PARALLEL if HOTPLUG_CPU + select WEAK_REORDERING_BEYOND_LLSC help Select this to build a kernel supporting EyeQ SoC from Mobileye. diff --git a/arch/mips/cavium-octeon/octeon-platform.c b/arch/mips/cavium-octeon/octeon-platform.c index 47677b5d7ed0..f53c98372e0b 100644 --- a/arch/mips/cavium-octeon/octeon-platform.c +++ b/arch/mips/cavium-octeon/octeon-platform.c @@ -18,7 +18,7 @@ #include <asm/octeon/octeon.h> #include <asm/octeon/cvmx-helper-board.h> -#ifdef CONFIG_USB +#if IS_ENABLED(CONFIG_USB) #include <linux/usb/ehci_def.h> #include <linux/usb/ehci_pdriver.h> #include <linux/usb/ohci_pdriver.h> @@ -1080,7 +1080,7 @@ end_led: ; } -#ifdef CONFIG_USB +#if IS_ENABLED(CONFIG_USB) /* OHCI/UHCI USB */ alias_prop = fdt_getprop(initial_boot_params, aliases, "uctl", NULL); diff --git a/arch/mips/configs/econet_en751221_defconfig b/arch/mips/configs/econet_en751221_defconfig new file mode 100644 index 000000000000..b35d33fe4044 --- /dev/null +++ b/arch/mips/configs/econet_en751221_defconfig @@ -0,0 +1,264 @@ +# CONFIG_LOCALVERSION_AUTO is not set +CONFIG_KERNEL_XZ=y +CONFIG_SYSVIPC=y +CONFIG_POSIX_MQUEUE=y +# CONFIG_CROSS_MEMORY_ATTACH is not set +CONFIG_HIGH_RES_TIMERS=y +CONFIG_BPF_SYSCALL=y +CONFIG_BPF_JIT=y +CONFIG_PREEMPT=y +CONFIG_CGROUPS=y +CONFIG_MEMCG=y +CONFIG_BLK_CGROUP=y +CONFIG_CGROUP_SCHED=y +CONFIG_CFS_BANDWIDTH=y +CONFIG_RT_GROUP_SCHED=y +CONFIG_CGROUP_PIDS=y +CONFIG_CGROUP_RDMA=y +CONFIG_CPUSETS=y +CONFIG_CGROUP_CPUACCT=y +CONFIG_CGROUP_BPF=y +CONFIG_NAMESPACES=y +# CONFIG_TIME_NS is not set +CONFIG_USER_NS=y +CONFIG_BLK_DEV_INITRD=y +# CONFIG_RD_GZIP is not set +# CONFIG_RD_BZIP2 is not set +# CONFIG_RD_LZMA is not set +# CONFIG_RD_XZ is not set +# CONFIG_RD_LZO is not set +# CONFIG_RD_LZ4 is not set +# CONFIG_RD_ZSTD is not set +# CONFIG_INITRAMFS_PRESERVE_MTIME is not set +CONFIG_LD_DEAD_CODE_DATA_ELIMINATION=y +CONFIG_EXPERT=y +# CONFIG_SGETMASK_SYSCALL is not set +# CONFIG_IO_URING is not set +# CONFIG_RSEQ is not set +# CONFIG_CACHESTAT_SYSCALL is not set +CONFIG_ECONET=y +CONFIG_CPU_MIPS32_R2=y +CONFIG_ZBOOT_LOAD_ADDRESS=0x80020000 +CONFIG_ARCH_FORCE_MAX_ORDER=11 +# CONFIG_MIPS_FP_SUPPORT is not set +CONFIG_NR_CPUS=2 +CONFIG_HZ_100=y +CONFIG_MIPS_RAW_APPENDED_DTB=y +# CONFIG_SCHED_SMT is not set +CONFIG_JUMP_LABEL=y +# CONFIG_STACKPROTECTOR_STRONG is not set +# CONFIG_GCC_PLUGINS is not set +CONFIG_MODULES=y +CONFIG_MODULE_UNLOAD=y +# CONFIG_BLOCK_LEGACY_AUTOLOAD is not set +CONFIG_BLK_DEV_THROTTLING=y +# CONFIG_BLK_DEBUG_FS is not set +CONFIG_PARTITION_ADVANCED=y +CONFIG_OF_PARTITION=y +# CONFIG_MQ_IOSCHED_DEADLINE is not set +# CONFIG_MQ_IOSCHED_KYBER is not set +# CONFIG_CORE_DUMP_DEFAULT_ELF_HEADERS is not set +CONFIG_SLAB_FREELIST_RANDOM=y +CONFIG_SLAB_FREELIST_HARDENED=y +# CONFIG_COMPAT_BRK is not set +CONFIG_LRU_GEN=y +CONFIG_LRU_GEN_ENABLED=y +CONFIG_NET=y +CONFIG_PACKET=y +CONFIG_UNIX=y +CONFIG_INET=y +CONFIG_IP_MULTICAST=y +CONFIG_IP_ADVANCED_ROUTER=y +CONFIG_IP_MULTIPLE_TABLES=y +CONFIG_IP_ROUTE_MULTIPATH=y +CONFIG_IP_ROUTE_VERBOSE=y +CONFIG_IP_MROUTE=y +CONFIG_IP_MROUTE_MULTIPLE_TABLES=y +CONFIG_IP_PIMSM_V1=y +CONFIG_IP_PIMSM_V2=y +CONFIG_SYN_COOKIES=y +# CONFIG_INET_DIAG is not set +CONFIG_TCP_CONG_ADVANCED=y +# CONFIG_TCP_CONG_BIC is not set +# CONFIG_TCP_CONG_WESTWOOD is not set +# CONFIG_TCP_CONG_HTCP is not set +# CONFIG_IPV6_SIT is not set +CONFIG_IPV6_SUBTREES=y +CONFIG_IPV6_MROUTE=y +CONFIG_IPV6_MROUTE_MULTIPLE_TABLES=y +CONFIG_IPV6_PIMSM_V2=y +CONFIG_IPV6_SEG6_LWTUNNEL=y +CONFIG_MPTCP=y +CONFIG_NETFILTER=y +# CONFIG_NETFILTER_EGRESS is not set +CONFIG_NF_CONNTRACK=m +CONFIG_NF_CONNTRACK_MARK=y +CONFIG_NF_CONNTRACK_ZONES=y +CONFIG_NF_CONNTRACK_PROCFS=y +CONFIG_NF_CONNTRACK_TIMEOUT=y +# CONFIG_NF_CT_PROTO_SCTP is not set +CONFIG_NF_TABLES=m +CONFIG_NF_TABLES_INET=y +CONFIG_NF_TABLES_NETDEV=y +CONFIG_NFT_NUMGEN=m +CONFIG_NFT_CT=m +CONFIG_NFT_FLOW_OFFLOAD=m +CONFIG_NFT_LOG=m +CONFIG_NFT_LIMIT=m +CONFIG_NFT_MASQ=m +CONFIG_NFT_REDIR=m +CONFIG_NFT_NAT=m +CONFIG_NFT_QUOTA=m +CONFIG_NFT_REJECT=m +CONFIG_NFT_HASH=m +CONFIG_NFT_FIB_INET=m +CONFIG_NF_FLOW_TABLE_INET=m +CONFIG_NF_FLOW_TABLE=m +CONFIG_NFT_FIB_IPV4=m +CONFIG_NF_TABLES_ARP=y +CONFIG_NF_LOG_IPV4=m +CONFIG_NFT_FIB_IPV6=m +CONFIG_NF_LOG_IPV6=m +CONFIG_NF_TABLES_BRIDGE=m +CONFIG_BRIDGE=y +CONFIG_BRIDGE_VLAN_FILTERING=y +CONFIG_VLAN_8021Q=y +CONFIG_NET_SCHED=y +CONFIG_NET_SCH_FQ_CODEL=y +CONFIG_NET_SCH_DEFAULT=y +CONFIG_DEFAULT_FQ_CODEL=y +CONFIG_NET_SWITCHDEV=y +CONFIG_NET_L3_MASTER_DEV=y +CONFIG_RFKILL=y +# CONFIG_LWTUNNEL_BPF is not set +CONFIG_PCI=y +CONFIG_PCIEPORTBUS=y +CONFIG_PCIEAER=y +# CONFIG_PCIEASPM is not set +CONFIG_PCI_MSI=y +# CONFIG_VGA_ARB is not set +CONFIG_PCIE_MEDIATEK=y +CONFIG_MTD=y +CONFIG_MTD_BLOCK=y +CONFIG_MTD_CFI=y +CONFIG_MTD_CFI_INTELEXT=y +CONFIG_MTD_CFI_AMDSTD=y +CONFIG_MTD_COMPLEX_MAPPINGS=y +CONFIG_MTD_SPI_NAND=y +CONFIG_MTD_UBI=y +CONFIG_MTD_UBI_BEB_LIMIT=13 +CONFIG_MTD_UBI_BLOCK=y +CONFIG_NETDEVICES=y +# CONFIG_NET_VENDOR_ASIX is not set +# CONFIG_NET_VENDOR_ENGLEDER is not set +# CONFIG_NET_VENDOR_FUNGIBLE is not set +# CONFIG_NET_VENDOR_LITEX is not set +# CONFIG_NET_VENDOR_MICROSOFT is not set +# CONFIG_NET_VENDOR_VERTEXCOM is not set +# CONFIG_NET_VENDOR_WANGXUN is not set +CONFIG_PPP=m +CONFIG_PPP_FILTER=y +CONFIG_PPP_MULTILINK=y +CONFIG_PPPOE=m +CONFIG_PPP_ASYNC=m +# CONFIG_WLAN_VENDOR_ADMTEK is not set +# CONFIG_WLAN_VENDOR_ATH is not set +# CONFIG_WLAN_VENDOR_ATMEL is not set +# CONFIG_WLAN_VENDOR_BROADCOM is not set +# CONFIG_WLAN_VENDOR_INTEL is not set +# CONFIG_WLAN_VENDOR_INTERSIL is not set +# CONFIG_WLAN_VENDOR_MARVELL is not set +# CONFIG_WLAN_VENDOR_MEDIATEK is not set +# CONFIG_WLAN_VENDOR_MICROCHIP is not set +# CONFIG_WLAN_VENDOR_PURELIFI is not set +# CONFIG_WLAN_VENDOR_RALINK is not set +# CONFIG_WLAN_VENDOR_REALTEK is not set +# CONFIG_WLAN_VENDOR_RSI is not set +# CONFIG_WLAN_VENDOR_SILABS is not set +# CONFIG_WLAN_VENDOR_ST is not set +# CONFIG_WLAN_VENDOR_TI is not set +# CONFIG_WLAN_VENDOR_ZYDAS is not set +# CONFIG_WLAN_VENDOR_QUANTENNA is not set +# CONFIG_INPUT is not set +# CONFIG_SERIO is not set +# CONFIG_VT is not set +# CONFIG_LEGACY_PTYS is not set +# CONFIG_LEGACY_TIOCSTI is not set +CONFIG_SERIAL_8250=y +CONFIG_SERIAL_8250_CONSOLE=y +CONFIG_SERIAL_OF_PLATFORM=y +# CONFIG_HW_RANDOM is not set +# CONFIG_DEVMEM is not set +CONFIG_SPI=y +# CONFIG_PTP_1588_CLOCK is not set +CONFIG_GPIOLIB=y +# CONFIG_HWMON is not set +CONFIG_WATCHDOG=y +CONFIG_MFD_SYSCON=y +# CONFIG_USB_PCI is not set +CONFIG_NEW_LEDS=y +CONFIG_LEDS_CLASS=y +CONFIG_LEDS_CLASS_MULTICOLOR=y +CONFIG_LEDS_GPIO=m +CONFIG_LEDS_TRIGGERS=y +CONFIG_LEDS_TRIGGER_TIMER=y +CONFIG_LEDS_TRIGGER_HEARTBEAT=y +CONFIG_LEDS_TRIGGER_DEFAULT_ON=y +CONFIG_LEDS_TRIGGER_NETDEV=y +CONFIG_STAGING=y +# CONFIG_MIPS_PLATFORM_DEVICES is not set +CONFIG_COMMON_CLK_EN7523=y +# CONFIG_IOMMU_SUPPORT is not set +CONFIG_RESET_CONTROLLER=y +CONFIG_PHY_ECONET_PCIE=y +# CONFIG_DNOTIFY is not set +CONFIG_FANOTIFY=y +CONFIG_OVERLAY_FS=y +# CONFIG_OVERLAY_FS_REDIRECT_ALWAYS_FOLLOW is not set +# CONFIG_PROC_PAGE_MONITOR is not set +CONFIG_TMPFS=y +CONFIG_TMPFS_POSIX_ACL=y +CONFIG_JFFS2_FS=y +CONFIG_JFFS2_SUMMARY=y +CONFIG_JFFS2_FS_XATTR=y +CONFIG_JFFS2_COMPRESSION_OPTIONS=y +CONFIG_UBIFS_FS=y +CONFIG_SQUASHFS=y +CONFIG_SQUASHFS_FILE_DIRECT=y +CONFIG_SQUASHFS_COMPILE_DECOMP_MULTI_PERCPU=y +# CONFIG_SQUASHFS_ZLIB is not set +CONFIG_SQUASHFS_XZ=y +CONFIG_SQUASHFS_EMBEDDED=y +CONFIG_KEYS=y +CONFIG_INIT_STACK_NONE=y +CONFIG_FORTIFY_SOURCE=y +CONFIG_HARDENED_USERCOPY=y +CONFIG_LIST_HARDENED=y +CONFIG_CRYPTO_NULL=y +CONFIG_CRYPTO_AES=y +CONFIG_CRYPTO_CCM=y +CONFIG_CRYPTO_GCM=y +# CONFIG_CRYPTO_HW is not set +# CONFIG_XZ_DEC_X86 is not set +# CONFIG_XZ_DEC_POWERPC is not set +# CONFIG_XZ_DEC_ARM is not set +# CONFIG_XZ_DEC_ARMTHUMB is not set +# CONFIG_XZ_DEC_ARM64 is not set +# CONFIG_XZ_DEC_SPARC is not set +# CONFIG_XZ_DEC_RISCV is not set +CONFIG_PRINTK_TIME=y +# CONFIG_DEBUG_MISC is not set +CONFIG_DEBUG_INFO_DWARF_TOOLCHAIN_DEFAULT=y +CONFIG_DEBUG_INFO_REDUCED=y +CONFIG_FRAME_WARN=1024 +CONFIG_STRIP_ASM_SYMS=y +CONFIG_MAGIC_SYSRQ=y +# CONFIG_MAGIC_SYSRQ_SERIAL is not set +CONFIG_DEBUG_FS=y +# CONFIG_SLUB_DEBUG is not set +CONFIG_SCHED_STACK_END_CHECK=y +CONFIG_PANIC_ON_OOPS=y +CONFIG_PANIC_TIMEOUT=1 +CONFIG_RCU_CPU_STALL_TIMEOUT=60 +# CONFIG_RCU_TRACE is not set +# CONFIG_FTRACE is not set diff --git a/arch/powerpc/kernel/iommu.c b/arch/powerpc/kernel/iommu.c index ee1b5cb557c9..1ae8384637b5 100644 --- a/arch/powerpc/kernel/iommu.c +++ b/arch/powerpc/kernel/iommu.c @@ -1076,7 +1076,7 @@ int iommu_tce_check_ioba(unsigned long page_shift, if (ioba < offset) return -EINVAL; - if ((ioba + 1) > (offset + size)) + if ((ioba + npages < ioba) || (ioba - offset + npages > size)) return -EINVAL; return 0; diff --git a/arch/powerpc/kvm/book3s_hv_nested.c b/arch/powerpc/kvm/book3s_hv_nested.c index 22e616662255..a6ff42d7666c 100644 --- a/arch/powerpc/kvm/book3s_hv_nested.c +++ b/arch/powerpc/kvm/book3s_hv_nested.c @@ -1204,8 +1204,10 @@ static void kvmhv_emulate_tlbie_all_lpid(struct kvm_vcpu *vcpu, int ric) spin_lock(&kvm->mmu_lock); idr_for_each_entry(&kvm->arch.kvm_nested_guest_idr, gp, lpid) { + ++gp->refcnt; spin_unlock(&kvm->mmu_lock); kvmhv_emulate_tlbie_lpid(vcpu, gp, ric); + kvmhv_put_nested(gp); spin_lock(&kvm->mmu_lock); } spin_unlock(&kvm->mmu_lock); diff --git a/arch/powerpc/kvm/book3s_hv_uvmem.c b/arch/powerpc/kvm/book3s_hv_uvmem.c index 5fbb95d90e99..463aef870c4e 100644 --- a/arch/powerpc/kvm/book3s_hv_uvmem.c +++ b/arch/powerpc/kvm/book3s_hv_uvmem.c @@ -779,8 +779,11 @@ static int kvmppc_svm_page_in(struct vm_area_struct *vma, if (spage) { ret = uv_page_in(kvm->arch.lpid, pfn << page_shift, gpa, 0, page_shift); - if (ret) + if (ret) { + unlock_page(dpage); + put_page(dpage); goto out_finalize; + } } } diff --git a/arch/x86/Kconfig.cpu b/arch/x86/Kconfig.cpu index e4654388d794..6e7a366f0798 100644 --- a/arch/x86/Kconfig.cpu +++ b/arch/x86/Kconfig.cpu @@ -204,10 +204,21 @@ config CC_HAS_MARCH_NATIVE # usage warnings that only appear wth '-march=native'. depends on CC_IS_GCC || CLANG_VERSION >= 190100 +config RUSTC_HAS_APXF + # The kernel isn't ready for in-kernel APX instructions. Without + # explicit frontend gating of APX, the backend may emit those + # instructions in native builds. + # + # Rust 1.88 added the `apxf` feature option, but versions before 1.93 + # emit an `apxf` target attribute that only LLVM 23+ can interpret. + def_bool (RUSTC_VERSION >= 108800 && RUSTC_LLVM_MAJOR_VERSION >= 23) || \ + RUSTC_VERSION >= 109300 + config X86_NATIVE_CPU bool "Build and optimize for local/native CPU" depends on X86_64 depends on CC_HAS_MARCH_NATIVE + depends on !RUST || RUSTC_HAS_APXF help Optimize for the current CPU used to compile the kernel. Use this option if you intend to build the kernel for your diff --git a/arch/x86/Makefile b/arch/x86/Makefile index 598f178102ee..8af6b80cffdd 100644 --- a/arch/x86/Makefile +++ b/arch/x86/Makefile @@ -161,6 +161,11 @@ else ifdef CONFIG_X86_NATIVE_CPU KBUILD_CFLAGS += -march=native + # Prevent the compiler from generating EGPR use. The kernel is + # not yet prepared for general in-kernel use. + KBUILD_CFLAGS += $(call cc-option,-mno-apx-features=egpr) + + # generate_rust_target.rs handles Rust APX gating. KBUILD_RUSTFLAGS += -Ctarget-cpu=native else KBUILD_CFLAGS += -march=x86-64 -mtune=generic diff --git a/arch/x86/entry/entry_fred.c b/arch/x86/entry/entry_fred.c index fb3594ddf731..854899bfec5d 100644 --- a/arch/x86/entry/entry_fred.c +++ b/arch/x86/entry/entry_fred.c @@ -10,6 +10,7 @@ #include <asm/desc.h> #include <asm/fred.h> #include <asm/idtentry.h> +#include <asm/processor-flags.h> #include <asm/syscall.h> #include <asm/trapnr.h> #include <asm/traps.h> @@ -71,7 +72,15 @@ static noinstr void fred_intx(struct pt_regs *regs) #endif default: - return exc_general_protection(regs, 0); + /* + * Reconstruct the #GP fault state that IDT delivery would produce. + * Clear the software event flag so ERETU with TF set does not trap + * before the resumed instruction. See prevent_single_step_upon_eretu(). + */ + regs->ip -= regs->fred_ss.insnlen; + regs->flags |= X86_EFLAGS_RF; + regs->fred_ss.swevent = 0; + return exc_general_protection(regs, (regs->fred_ss.vector << 3) | 2); } } diff --git a/arch/x86/include/asm/div64.h b/arch/x86/include/asm/div64.h index 30fd06ede751..8a2d343f977e 100644 --- a/arch/x86/include/asm/div64.h +++ b/arch/x86/include/asm/div64.h @@ -111,7 +111,7 @@ static inline u64 mul_u64_add_u64_div_u64(u64 rax, u64 mul, u64 add, u64 div) if (!statically_true(!add)) asm ("addq %[add], %[lo]; adcq $0, %[hi]" : - [lo] "+r" (rax), [hi] "+r" (rdx) : [add] "irm" (add)); + [lo] "+r" (rax), [hi] "+r" (rdx) : [add] "erm" (add)); asm ("divq %[div]" : "+a" (rax), "+d" (rdx) : [div] "rm" (div)); diff --git a/arch/x86/include/asm/pgtable.h b/arch/x86/include/asm/pgtable.h index d5f4917c1edc..d551120a7c88 100644 --- a/arch/x86/include/asm/pgtable.h +++ b/arch/x86/include/asm/pgtable.h @@ -806,7 +806,7 @@ static inline pmd_t pmd_modify(pmd_t pmd, pgprot_t newprot) pmdval_t val = pmd_val(pmd), oldval = val; pmd_t pmd_result; - val &= (_HPAGE_CHG_MASK & ~_PAGE_DIRTY); + val &= _HPAGE_CHG_MASK; val |= check_pgprot(newprot) & ~_HPAGE_CHG_MASK; val = flip_protnone_guard(oldval, val, PHYSICAL_PMD_PAGE_MASK); diff --git a/arch/x86/include/asm/text-patching.h b/arch/x86/include/asm/text-patching.h index f2d142a0a862..ea09381070e8 100644 --- a/arch/x86/include/asm/text-patching.h +++ b/arch/x86/include/asm/text-patching.h @@ -164,9 +164,9 @@ unsigned long int3_emulate_pop(struct pt_regs *regs) } static __always_inline -void int3_emulate_call(struct pt_regs *regs, unsigned long func) +void int3_emulate_call(struct pt_regs *regs, unsigned long ip, unsigned long func) { - int3_emulate_push(regs, regs->ip - INT3_INSN_SIZE + CALL_INSN_SIZE); + int3_emulate_push(regs, ip); int3_emulate_jmp(regs, func); } diff --git a/arch/x86/kernel/alternative.c b/arch/x86/kernel/alternative.c index 91b1cdd16569..582c6d8307d1 100644 --- a/arch/x86/kernel/alternative.c +++ b/arch/x86/kernel/alternative.c @@ -6,6 +6,9 @@ #include <linux/vmalloc.h> #include <linux/memory.h> #include <linux/execmem.h> +#include <linux/cleanup.h> +#include <linux/kgdb.h> +#include <linux/mmap_lock.h> #include <asm/text-patching.h> #include <asm/insn.h> @@ -1198,6 +1201,41 @@ static bool cfi_debug __ro_after_init; bool cfi_bhi __ro_after_init = false; #endif +#ifdef CONFIG_FINEIBT +/* + * <fineibt_preamble_start>: + * 0: f3 0f 1e fa endbr64 + * 4: 2d 78 56 34 12 sub $0x12345678, %eax + * 9: 2e 0f 85 03 00 00 00 jne,pn 13 <fineibt_preamble_start+0x13> + * 10: 0f 1f 40 d6 nopl -0x2a(%rax) + * + * Note that the JNE target is the 0xD6 byte inside the NOPL, this decodes as + * UDB on x86_64 and raises #UD. + */ +asm( ".pushsection .rodata \n" + "fineibt_preamble_start: \n" + " endbr64 \n" + " subl $0x12345678, %eax \n" + "fineibt_preamble_bhi: \n" + " cs jne.d32 fineibt_preamble_start+0x13 \n" + "#fineibt_func: \n" + " nopl -42(%rax) \n" + "fineibt_preamble_end: \n" + ".popsection\n" +); + +extern u8 fineibt_preamble_start[]; +extern u8 fineibt_preamble_bhi[]; +extern u8 fineibt_preamble_end[]; + +#define fineibt_preamble_size (fineibt_preamble_end - fineibt_preamble_start) +#define fineibt_preamble_bhi (fineibt_preamble_bhi - fineibt_preamble_start) +#define fineibt_preamble_ud 0x13 +#define fineibt_preamble_hash 5 + +#define fineibt_prefix_size (fineibt_preamble_size - ENDBR_INSN_SIZE) +#endif /* CONFIG_FINEIBT */ + #ifdef CONFIG_CFI u32 cfi_get_func_hash(void *func) { @@ -1205,9 +1243,11 @@ u32 cfi_get_func_hash(void *func) func -= cfi_get_offset(); switch (cfi_mode) { +#ifdef CONFIG_FINEIBT case CFI_FINEIBT: - func += 7; + func += fineibt_preamble_hash; break; +#endif case CFI_KCFI: func += 1; break; @@ -1364,39 +1404,6 @@ early_param("cfi", cfi_parse_cmdline); */ /* - * <fineibt_preamble_start>: - * 0: f3 0f 1e fa endbr64 - * 4: 2d 78 56 34 12 sub $0x12345678, %eax - * 9: 2e 0f 85 03 00 00 00 jne,pn 13 <fineibt_preamble_start+0x13> - * 10: 0f 1f 40 d6 nopl -0x2a(%rax) - * - * Note that the JNE target is the 0xD6 byte inside the NOPL, this decodes as - * UDB on x86_64 and raises #UD. - */ -asm( ".pushsection .rodata \n" - "fineibt_preamble_start: \n" - " endbr64 \n" - " subl $0x12345678, %eax \n" - "fineibt_preamble_bhi: \n" - " cs jne.d32 fineibt_preamble_start+0x13 \n" - "#fineibt_func: \n" - " nopl -42(%rax) \n" - "fineibt_preamble_end: \n" - ".popsection\n" -); - -extern u8 fineibt_preamble_start[]; -extern u8 fineibt_preamble_bhi[]; -extern u8 fineibt_preamble_end[]; - -#define fineibt_preamble_size (fineibt_preamble_end - fineibt_preamble_start) -#define fineibt_preamble_bhi (fineibt_preamble_bhi - fineibt_preamble_start) -#define fineibt_preamble_ud 0x13 -#define fineibt_preamble_hash 5 - -#define fineibt_prefix_size (fineibt_preamble_size - ENDBR_INSN_SIZE) - -/* * <fineibt_caller_start>: * 0: b8 78 56 34 12 mov $0x12345678, %eax * 5: 4d 8d 5b f0 lea -0x10(%r11), %r11 @@ -2169,6 +2176,7 @@ int3_exception_notify(struct notifier_block *self, unsigned long val, void *data unsigned long selftest = (unsigned long)&int3_selftest_asm; struct die_args *args = data; struct pt_regs *regs = args->regs; + unsigned long ip; OPTIMIZER_HIDE_VAR(selftest); @@ -2181,7 +2189,8 @@ int3_exception_notify(struct notifier_block *self, unsigned long val, void *data if (regs->ip - INT3_INSN_SIZE != selftest) return NOTIFY_DONE; - int3_emulate_call(regs, (unsigned long)&int3_selftest_callee); + ip = regs->ip - INT3_INSN_SIZE + CALL_INSN_SIZE; + int3_emulate_call(regs, ip, (unsigned long)&int3_selftest_callee); return NOTIFY_STOP; } @@ -2372,6 +2381,38 @@ static void text_poke_memset(void *dst, const void *src, size_t len) typedef void text_poke_f(void *dst, const void *src, size_t len); +static void __poke_vmalloc_pages(struct page **pages, void *addr, + bool cross_page_boundary) +{ + pages[0] = vmalloc_to_page(addr); + if (cross_page_boundary) + pages[1] = vmalloc_to_page(addr + PAGE_SIZE); +} + +static void poke_vmalloc_pages(struct page **pages, void *addr, + bool cross_page_boundary) +{ + if (in_dbg_master()) { + /* + * If called from kgdb cannot sleep, but all other CPUs stopped + * anyway so safe to proceed without locks + */ + __poke_vmalloc_pages(pages, addr, cross_page_boundary); + } else { + /* + * execmem ROX ranges are shared between modules and can be + * collapsed to huge PMD entries, and this collapse can happen + * concurrently with a racing set_memory_rox(). + * + * Prevent vmalloc_to_page() from racing by acquiring an + * init_mm read lock which pairs with the init_mm write lock in + * cpa_collapse_large_pages(). + */ + guard(mmap_read_lock)(&init_mm); + __poke_vmalloc_pages(pages, addr, cross_page_boundary); + } +} + static void *__text_poke(text_poke_f func, void *addr, const void *src, size_t len) { bool cross_page_boundary = offset_in_page(addr) + len > PAGE_SIZE; @@ -2389,9 +2430,7 @@ static void *__text_poke(text_poke_f func, void *addr, const void *src, size_t l BUG_ON(!after_bootmem); if (!core_kernel_text((unsigned long)addr)) { - pages[0] = vmalloc_to_page(addr); - if (cross_page_boundary) - pages[1] = vmalloc_to_page(addr + PAGE_SIZE); + poke_vmalloc_pages(pages, addr, cross_page_boundary); } else { pages[0] = virt_to_page(addr); WARN_ON(!PageReserved(pages[0])); @@ -2721,7 +2760,7 @@ noinstr int smp_text_poke_int3_handler(struct pt_regs *regs) break; case CALL_INSN_OPCODE: - int3_emulate_call(regs, (long)ip + tpl->disp); + int3_emulate_call(regs, (long)ip, (long)ip + tpl->disp); break; case JMP32_INSN_OPCODE: diff --git a/arch/x86/kernel/amd_node.c b/arch/x86/kernel/amd_node.c index 762585775b5a..b7926ba3610a 100644 --- a/arch/x86/kernel/amd_node.c +++ b/arch/x86/kernel/amd_node.c @@ -251,7 +251,7 @@ __setup("amd_smn_debugfs_enable", amd_smn_enable_dfs); static int __init amd_smn_init(void) { u16 count, num_roots, roots_per_node, node, num_nodes; - struct pci_dev *root; + struct pci_dev *root __free(pci_dev_put) = NULL; if (!cpu_feature_enabled(X86_FEATURE_ZEN)) return 0; @@ -262,7 +262,6 @@ static int __init amd_smn_init(void) return 0; num_roots = 0; - root = NULL; while ((root = get_next_root(root))) { pci_dbg(root, "Reserving PCI config space\n"); @@ -299,14 +298,13 @@ static int __init amd_smn_init(void) count = 0; node = 0; - root = NULL; while (node < num_nodes && (root = get_next_root(root))) { /* Use one root for each node and skip the rest. */ if (count++ % roots_per_node) continue; pci_dbg(root, "is root for AMD node %u\n", node); - amd_roots[node++] = root; + amd_roots[node++] = pci_dev_get(root); } if (enable_dfs) { diff --git a/arch/x86/kernel/cpu/microcode/intel.c b/arch/x86/kernel/cpu/microcode/intel.c index 1142183c950c..9f09d388ed20 100644 --- a/arch/x86/kernel/cpu/microcode/intel.c +++ b/arch/x86/kernel/cpu/microcode/intel.c @@ -309,6 +309,26 @@ static void save_microcode_patch(struct microcode_intel *patch) pr_err("Unable to allocate microcode memory size: %u\n", size); } +static bool revision_is_safe(struct cpu_signature *sig, u32 rev) +{ + u32 vfm = IFM(x86_family(sig->sig), x86_model(sig->sig)); + + /* + * Erratum GNR98 can cause #MCs if "jumping over" revision 0x1000405. + * Avoid the jumps. + */ + if (vfm == INTEL_GRANITERAPIDS_X && + x86_stepping(sig->sig) == 1 && + sig->pf & 0x95 && + sig->rev < 0x1000405 && + rev > 0x1000405) { + pr_err_once("Erratum GNR98: skipping revision 0x%x.\n", rev); + return false; + } + + return true; +} + /* Scan blob for microcode matching the boot CPUs family, model, stepping */ static __init struct microcode_intel *scan_microcode(void *data, size_t size, struct ucode_cpu_info *uci, @@ -330,6 +350,9 @@ static __init struct microcode_intel *scan_microcode(void *data, size_t size, if (!intel_find_matching_signature(data, &uci->cpu_sig)) continue; + if (!revision_is_safe(&uci->cpu_sig, mc_header->rev)) + continue; + /* * For saving the early microcode, find the matching revision which * was loaded on the BSP. @@ -878,6 +901,9 @@ static enum ucode_state parse_microcode_blobs(int cpu, struct iov_iter *iter) if (!intel_find_matching_signature(mc, &uci->cpu_sig)) continue; + if (!revision_is_safe(&uci->cpu_sig, mc_header.rev)) + continue; + is_safe = ucode_validate_minrev(&mc_header); if (force_minrev && !is_safe) continue; diff --git a/arch/x86/kernel/kprobes/core.c b/arch/x86/kernel/kprobes/core.c index 4e5f8c1736ec..133ff20caccd 100644 --- a/arch/x86/kernel/kprobes/core.c +++ b/arch/x86/kernel/kprobes/core.c @@ -510,10 +510,9 @@ NOKPROBE_SYMBOL(kprobe_emulate_ret); static void kprobe_emulate_call(struct kprobe *p, struct pt_regs *regs) { - unsigned long func = regs->ip - INT3_INSN_SIZE + p->ainsn.size; + unsigned long ip = regs->ip - INT3_INSN_SIZE + p->ainsn.size; - func += p->ainsn.rel32; - int3_emulate_call(regs, func); + int3_emulate_call(regs, ip, ip + p->ainsn.rel32); } NOKPROBE_SYMBOL(kprobe_emulate_call); diff --git a/arch/x86/mm/mem_encrypt.c b/arch/x86/mm/mem_encrypt.c index 95bae74fdab2..3aefdef5bcb0 100644 --- a/arch/x86/mm/mem_encrypt.c +++ b/arch/x86/mm/mem_encrypt.c @@ -13,6 +13,7 @@ #include <linux/cc_platform.h> #include <linux/mem_encrypt.h> #include <linux/virtio_anchor.h> +#include <linux/iommu-dma.h> #include <asm/sev.h> @@ -30,7 +31,7 @@ bool force_dma_unencrypted(struct device *dev) * device does not support DMA to addresses that include the * encryption mask. */ - if (cc_platform_has(CC_ATTR_HOST_MEM_ENCRYPT)) { + if (cc_platform_has(CC_ATTR_HOST_MEM_ENCRYPT) && !use_dma_iommu(dev)) { u64 dma_enc_mask = DMA_BIT_MASK(__ffs64(sme_me_mask)); u64 dma_dev_mask = min_not_zero(dev->coherent_dma_mask, dev->bus_dma_limit); diff --git a/arch/x86/mm/pat/set_memory.c b/arch/x86/mm/pat/set_memory.c index c38faf39ce15..4652487b5572 100644 --- a/arch/x86/mm/pat/set_memory.c +++ b/arch/x86/mm/pat/set_memory.c @@ -22,6 +22,7 @@ #include <linux/cc_platform.h> #include <linux/set_memory.h> #include <linux/memregion.h> +#include <linux/cleanup.h> #include <asm/e820/api.h> #include <asm/processor.h> @@ -49,7 +50,8 @@ struct cpa_data { unsigned int flags; unsigned int force_split : 1, force_static_prot : 1, - force_flush_all : 1; + force_flush_all : 1, + init_mm_read_locked : 1; struct page **pages; }; @@ -409,7 +411,7 @@ static void __cpa_flush_tlb(void *data) static int collapse_large_pages(unsigned long addr, struct list_head *pgtables); -static void cpa_collapse_large_pages(struct cpa_data *cpa) +static void __cpa_collapse_large_pages(struct cpa_data *cpa) { unsigned long start, addr, end; struct ptdesc *ptdesc, *tmp; @@ -439,10 +441,30 @@ static void cpa_collapse_large_pages(struct cpa_data *cpa) list_for_each_entry_safe(ptdesc, tmp, &pgtables, pt_list) { list_del(&ptdesc->pt_list); - pagetable_free(ptdesc); + /* + * Only early alloc'd direct map should not be flagged PG_table + * here and those shouldn't be collapsed. However be abundantly + * cautious and handle the !PG_table case too. + */ + if (PageTable((ptdesc_page(ptdesc)))) + pagetable_dtor_free(ptdesc); + else + pagetable_free(ptdesc); } } +static void cpa_collapse_large_pages(struct cpa_data *cpa) +{ + /* + * Take the mmap write lock on init_mm to: + * - Avoid a use-after-free if raced by ptdump (which takes its own + * write lock on init_mm). + * - Serialise concurrent CPA walkers. + */ + scoped_guard(mmap_write_lock, &init_mm) + __cpa_collapse_large_pages(cpa); +} + static void cpa_flush(struct cpa_data *cpa, int cache) { unsigned int i; @@ -1120,11 +1142,10 @@ set: static int __split_large_page(struct cpa_data *cpa, pte_t *kpte, unsigned long address, - struct ptdesc *ptdesc) + pte_t *pbase) { unsigned long lpaddr, lpinc, ref_pfn, pfn, pfninc = 1; - struct page *base = ptdesc_page(ptdesc); - pte_t *pbase = (pte_t *)page_address(base); + struct page *base = virt_to_page(pbase); unsigned int i, level; pgprot_t ref_prot; bool nx, rw; @@ -1224,16 +1245,20 @@ __split_large_page(struct cpa_data *cpa, pte_t *kpte, unsigned long address, static int split_large_page(struct cpa_data *cpa, pte_t *kpte, unsigned long address) { - struct ptdesc *ptdesc; + pte_t *pte; spin_unlock(&cpa_lock); - ptdesc = pagetable_alloc(GFP_KERNEL, 0); + if (cpa->init_mm_read_locked) + mmap_read_unlock(&init_mm); + pte = pte_alloc_one_kernel(&init_mm); + if (cpa->init_mm_read_locked) + mmap_read_lock(&init_mm); spin_lock(&cpa_lock); - if (!ptdesc) + if (!pte) return -ENOMEM; - if (__split_large_page(cpa, kpte, address, ptdesc)) - pagetable_free(ptdesc); + if (__split_large_page(cpa, kpte, address, pte)) + pte_free_kernel(&init_mm, pte); return 0; } @@ -2121,7 +2146,11 @@ static int change_page_attr_set_clr(unsigned long *addr, int numpages, cpa.curpage = 0; cpa.force_split = force_split; - ret = __change_page_attr_set_clr(&cpa, 1); + /* Avoid race with concurrent CPA collapse. */ + cpa.init_mm_read_locked = true; + scoped_guard(mmap_read_lock, &init_mm) + ret = __change_page_attr_set_clr(&cpa, 1); + cpa.init_mm_read_locked = false; /* * Check whether we really changed something: diff --git a/drivers/ata/libahci.c b/drivers/ata/libahci.c index 6d72eb017b49..08ab56e95566 100644 --- a/drivers/ata/libahci.c +++ b/drivers/ata/libahci.c @@ -744,15 +744,28 @@ void ahci_start_fis_rx(struct ata_port *ap) struct ahci_port_priv *pp = ap->private_data; u32 tmp; - /* set FIS registers */ + /* + * On HBAs that only support 32-bit addressing PxCLBU is read only '0'. + * When applying the AHCI_HFLAG_32BIT_ONLY quirk, PxCLBU is RW, and the + * reset value is Implementation Specific, so we need to clear it to 0. + */ if (hpriv->cap & HOST_CAP_64) writel((pp->cmd_slot_dma >> 16) >> 16, port_mmio + PORT_LST_ADDR_HI); + else if (hpriv->flags & AHCI_HFLAG_32BIT_ONLY) + writel(0, port_mmio + PORT_LST_ADDR_HI); writel(pp->cmd_slot_dma & 0xffffffff, port_mmio + PORT_LST_ADDR); + /* + * On HBAs that only support 32-bit addressing PxFBU is read only '0'. + * When applying the AHCI_HFLAG_32BIT_ONLY quirk, PxFBU is RW, and the + * reset value is Implementation Specific, so we need to clear it to 0. + */ if (hpriv->cap & HOST_CAP_64) writel((pp->rx_fis_dma >> 16) >> 16, port_mmio + PORT_FIS_ADDR_HI); + else if (hpriv->flags & AHCI_HFLAG_32BIT_ONLY) + writel(0, port_mmio + PORT_FIS_ADDR_HI); writel(pp->rx_fis_dma & 0xffffffff, port_mmio + PORT_FIS_ADDR); /* enable FIS reception */ diff --git a/drivers/ata/libahci_platform.c b/drivers/ata/libahci_platform.c index 6e072d681341..f6524135ea11 100644 --- a/drivers/ata/libahci_platform.c +++ b/drivers/ata/libahci_platform.c @@ -620,10 +620,10 @@ struct ahci_host_priv *ahci_platform_get_resources(struct platform_device *pdev, of_platform_device_create(child, NULL, NULL); port_dev = of_find_device_by_node(child); - if (port_dev) { rc = ahci_platform_get_regulator(hpriv, port, &port_dev->dev); + put_device(&port_dev->dev); if (rc == -EPROBE_DEFER) goto err_out; } diff --git a/drivers/ata/libata-scsi.c b/drivers/ata/libata-scsi.c index b3666519b648..7e22bbc38238 100644 --- a/drivers/ata/libata-scsi.c +++ b/drivers/ata/libata-scsi.c @@ -2297,10 +2297,9 @@ static unsigned int ata_scsiop_inq_89(struct ata_device *dev, * logical block, so it can hold at most sector_size / 512 pages. * * Return: the maximum number of 512-byte pages a single translated WRITE SAME - * command may send to @dev (never less than one), that is the smaller of: - * - MAX PAGES PER DSM COMMAND (IDENTIFY DEVICE word 105), when the device - * reports a non-zero limit; and - * - the logical sector size expressed in 512-byte pages (see above). + * command may send to @dev, that is the smaller of MAX PAGES PER DSM COMMAND + * (IDENTIFY DEVICE word 105, when the device reports a non-zero limit) and + * the logical sector size expressed in 512-byte pages; never less than one. */ static unsigned int ata_dsm_trim_pages(struct ata_device *dev) { diff --git a/drivers/bluetooth/btintel_pcie.c b/drivers/bluetooth/btintel_pcie.c index 6d9649776ae7..6e6e2b19815c 100644 --- a/drivers/bluetooth/btintel_pcie.c +++ b/drivers/bluetooth/btintel_pcie.c @@ -404,6 +404,12 @@ static int btintel_pcie_send_sync(struct btintel_pcie_data *data, if (tfd_index > txq->count) return -ERANGE; + if (skb->len > BTINTEL_PCIE_BUFFER_SIZE - BTINTEL_PCIE_HCI_TYPE_LEN) { + bt_dev_err(hdev, "TX skb too large (%u > %u)", skb->len, + BTINTEL_PCIE_BUFFER_SIZE - BTINTEL_PCIE_HCI_TYPE_LEN); + return -EMSGSIZE; + } + /* Firmware raises alive interrupt on HCI_OP_RESET or * BTINTEL_HCI_OP_RESET */ @@ -502,7 +508,7 @@ static int btintel_pcie_submit_rx(struct btintel_pcie_data *data) frbd_index = data->ia.tr_hia[BTINTEL_PCIE_RXQ_NUM]; - if (frbd_index > rxq->count) + if (frbd_index >= rxq->count) return -ERANGE; /* Prepare for RX submit. It updates the FRBD with the address of DMA diff --git a/drivers/bluetooth/btmtk.c b/drivers/bluetooth/btmtk.c index 26d525acd659..7ea8bcd8a7ec 100644 --- a/drivers/bluetooth/btmtk.c +++ b/drivers/bluetooth/btmtk.c @@ -721,7 +721,12 @@ static int btmtk_usb_hci_wmt_sync(struct hci_dev *hdev, case BTMTK_WMT_FUNC_CTRL: if (!skb_pull_data(data->evt_skb, sizeof(wmt_evt_funcc->status))) { - status = BTMTK_WMT_ON_UNDONE; + /* A plain enable/disable request is acked with just + * the WMT header and no trailing status word; the + * result is carried in the header's own flag byte. + */ + status = wmt_evt->whdr.flag ? BTMTK_WMT_ON_UNDONE : + BTMTK_WMT_ON_DONE; break; } diff --git a/drivers/bluetooth/btmtksdio.c b/drivers/bluetooth/btmtksdio.c index 94aa60d9cc20..a15ae6598c66 100644 --- a/drivers/bluetooth/btmtksdio.c +++ b/drivers/bluetooth/btmtksdio.c @@ -217,7 +217,14 @@ static int mtk_hci_wmt_sync(struct hci_dev *hdev, } /* Parse and handle the return WMT event */ - wmt_evt = (struct btmtk_hci_wmt_evt *)bdev->evt_skb->data; + wmt_evt = skb_pull_data(bdev->evt_skb, sizeof(*wmt_evt)); + if (!wmt_evt) { + bt_dev_err(hdev, "WMT event too short (%u bytes)", + bdev->evt_skb->len); + err = -EINVAL; + goto err_free_skb; + } + if (wmt_evt->whdr.op != hdr->op) { bt_dev_err(hdev, "Wrong op received %d expected %d", wmt_evt->whdr.op, hdr->op); @@ -233,6 +240,17 @@ static int mtk_hci_wmt_sync(struct hci_dev *hdev, status = BTMTK_WMT_PATCH_DONE; break; case BTMTK_WMT_FUNC_CTRL: + if (!skb_pull_data(bdev->evt_skb, + sizeof(wmt_evt_funcc->status))) { + /* A plain enable/disable request is acked with just + * the WMT header and no trailing status word; the + * result is carried in the header's own flag byte. + */ + status = wmt_evt->whdr.flag ? BTMTK_WMT_ON_UNDONE : + BTMTK_WMT_ON_DONE; + break; + } + wmt_evt_funcc = (struct btmtk_hci_wmt_evt_funcc *)wmt_evt; if (be16_to_cpu(wmt_evt_funcc->status) == 0x404) status = BTMTK_WMT_ON_DONE; @@ -1244,10 +1262,8 @@ static int btmtksdio_shutdown(struct hci_dev *hdev) wmt_params.status = NULL; err = mtk_hci_wmt_sync(hdev, &wmt_params); - if (err < 0) { + if (err < 0) bt_dev_err(hdev, "Failed to send wmt func ctrl (%d)", err); - return err; - } ignore_wmt_cmd: pm_runtime_put_noidle(bdev->dev); diff --git a/drivers/bluetooth/btmtkuart.c b/drivers/bluetooth/btmtkuart.c index 27aa48ff3ac2..4af6fbbbd302 100644 --- a/drivers/bluetooth/btmtkuart.c +++ b/drivers/bluetooth/btmtkuart.c @@ -151,7 +151,14 @@ static int mtk_hci_wmt_sync(struct hci_dev *hdev, } /* Parse and handle the return WMT event */ - wmt_evt = (struct btmtk_hci_wmt_evt *)bdev->evt_skb->data; + wmt_evt = skb_pull_data(bdev->evt_skb, sizeof(*wmt_evt)); + if (!wmt_evt) { + bt_dev_err(hdev, "WMT event too short (%u bytes)", + bdev->evt_skb->len); + err = -EINVAL; + goto err_free_wc; + } + if (wmt_evt->whdr.op != hdr->op) { bt_dev_err(hdev, "Wrong op received %d expected %d", wmt_evt->whdr.op, hdr->op); @@ -167,6 +174,17 @@ static int mtk_hci_wmt_sync(struct hci_dev *hdev, status = BTMTK_WMT_PATCH_DONE; break; case BTMTK_WMT_FUNC_CTRL: + if (!skb_pull_data(bdev->evt_skb, + sizeof(wmt_evt_funcc->status))) { + /* A plain enable/disable request is acked with just + * the WMT header and no trailing status word; the + * result is carried in the header's own flag byte. + */ + status = wmt_evt->whdr.flag ? BTMTK_WMT_ON_UNDONE : + BTMTK_WMT_ON_DONE; + break; + } + wmt_evt_funcc = (struct btmtk_hci_wmt_evt_funcc *)wmt_evt; if (be16_to_cpu(wmt_evt_funcc->status) == 0x404) status = BTMTK_WMT_ON_DONE; diff --git a/drivers/bluetooth/btusb.c b/drivers/bluetooth/btusb.c index 002b9f975710..dc7191bf4234 100644 --- a/drivers/bluetooth/btusb.c +++ b/drivers/bluetooth/btusb.c @@ -71,6 +71,15 @@ static struct usb_driver btusb_driver; #define BTUSB_BROKEN_EXT_SCAN BIT(29) static const struct usb_device_id btusb_table[] = { + /* + * NXP IW610 (0471:0215): the composite device reports Bluetooth + * class at the whole-device level, so the generic entry below + * would also match this WiFi vendor interface. Ignore it here + * first so mwifiex-nxp can bind it instead. + */ + { USB_DEVICE_AND_INTERFACE_INFO(0x0471, 0x0215, 0xff, 0xff, 0xff), + .driver_info = BTUSB_IGNORE }, + /* Generic Bluetooth USB device */ { USB_DEVICE_INFO(0xe0, 0x01, 0x01) }, @@ -477,6 +486,14 @@ static const struct usb_device_id quirks_table[] = { { USB_DEVICE(0x1286, 0x2046), .driver_info = BTUSB_MARVELL }, { USB_DEVICE(0x1286, 0x204e), .driver_info = BTUSB_MARVELL }, + /* + * NXP IW610 BT interfaces (Marvell-lineage silicon, same quirk as + * the 0x1286 entries above). Scoped to the BT interface class, + * not just VID/PID -- see the btusb_table entry above. + */ + { USB_DEVICE_AND_INTERFACE_INFO(0x0471, 0x0215, 0xe0, 0x01, 0x01), + .driver_info = BTUSB_MARVELL }, + /* Intel Bluetooth devices */ { USB_DEVICE(0x8087, 0x0025), .driver_info = BTUSB_INTEL_COMBINED }, { USB_DEVICE(0x8087, 0x0026), .driver_info = BTUSB_INTEL_COMBINED }, diff --git a/drivers/bluetooth/hci_qca.c b/drivers/bluetooth/hci_qca.c index faa964735adb..7089e9b639b2 100644 --- a/drivers/bluetooth/hci_qca.c +++ b/drivers/bluetooth/hci_qca.c @@ -2228,8 +2228,8 @@ static void qca_power_off(struct hci_uart *hu) bool sw_ctrl_state; struct qca_power *power; - /* From this point we go into power off state. But serial port is - * still open, stop queueing the IBS data and flush all the buffered + /* From this point we go into power off state. But serial port may + * still be open, stop queueing the IBS data and flush all the buffered * data in skb's. */ spin_lock_irqsave(&qca->hci_ibs_lock, flags); @@ -2251,8 +2251,14 @@ static void qca_power_off(struct hci_uart *hu) case QCA_WCN3990: case QCA_WCN3991: case QCA_WCN3998: - host_set_baudrate(hu, 2400); - qca_send_power_pulse(hu, false); + /* Both of these write to the serial port which may have + * already been closed by hci_uart_close(), which closes + * the port if HCI_QUIRK_NON_PERSISTENT_SETUP is set. + */ + if (test_bit(HCI_UART_PROTO_READY, &hu->flags)) { + host_set_baudrate(hu, 2400); + qca_send_power_pulse(hu, false); + } break; default: break; diff --git a/drivers/clk/clk-scpi.c b/drivers/clk/clk-scpi.c index 2328d2abf6d8..b2d412ca1d22 100644 --- a/drivers/clk/clk-scpi.c +++ b/drivers/clk/clk-scpi.c @@ -73,7 +73,7 @@ static unsigned long scpi_dvfs_recalc_rate(struct clk_hw *hw, int idx = clk->scpi_ops->dvfs_get_idx(clk->id); const struct scpi_opp *opp; - if (idx < 0) + if (idx < 0 || idx >= clk->info->count) return 0; opp = clk->info->opps + idx; @@ -272,10 +272,15 @@ static int scpi_clocks_probe(struct platform_device *pdev) if (match->data != &scpi_dvfs_ops) continue; /* Add the virtual cpufreq device if it's DVFS clock provider */ + if (cpufreq_dev) + continue; cpufreq_dev = platform_device_register_simple("scpi-cpufreq", - -1, NULL, 0); - if (IS_ERR(cpufreq_dev)) + PLATFORM_DEVID_NONE, + NULL, 0); + if (IS_ERR(cpufreq_dev)) { pr_warn("unable to register cpufreq device"); + cpufreq_dev = NULL; + } } return 0; } diff --git a/drivers/crypto/caam/jr.c b/drivers/crypto/caam/jr.c index 239469f6882e..cf725a6ac0fb 100644 --- a/drivers/crypto/caam/jr.c +++ b/drivers/crypto/caam/jr.c @@ -583,12 +583,29 @@ static int caam_jr_probe(struct platform_device *pdev) struct caam_drv_private_jr *jrpriv; static int total_jobrs; void __iomem *ctrl; + struct resource *r; int error; int irq; - ctrl = devm_platform_ioremap_resource(pdev, 0); - if (IS_ERR(ctrl)) - return PTR_ERR(ctrl); + /* + * The job rings live inside the register window of their parent + * fsl,sec-v4.0 node, which caam_probe() already reserves (and maps) + * via devm_of_iomap(). A requested region that overlaps that + * reservation, e.g. from devm_platform_ioremap_resource(), would + * therefore fail with -EBUSY, so map the registers without claiming + * the region here. + */ + r = platform_get_resource(pdev, IORESOURCE_MEM, 0); + if (!r) { + dev_err(&pdev->dev, "platform_get_resource() failed\n"); + return -EINVAL; + } + + ctrl = devm_ioremap(&pdev->dev, r->start, resource_size(r)); + if (!ctrl) { + dev_err(&pdev->dev, "devm_ioremap() failed\n"); + return -ENOMEM; + } irq = platform_get_irq(pdev, 0); if (irq < 0) diff --git a/drivers/dma-buf/Kconfig b/drivers/dma-buf/Kconfig index 7efc0f0d0712..e4f078a326a4 100644 --- a/drivers/dma-buf/Kconfig +++ b/drivers/dma-buf/Kconfig @@ -43,7 +43,7 @@ config UDMABUF config DMABUF_DEBUG bool "DMA-BUF debug checks" depends on DMA_SHARED_BUFFER - default y if DEBUG + default y if DEBUG_KERNEL help This option enables additional checks for DMA-BUF importers and exporters. Specifically it validates that importers do not peek at the diff --git a/drivers/dma-buf/dma-buf-mapping.c b/drivers/dma-buf/dma-buf-mapping.c index 794acff2546a..833be519e1e6 100644 --- a/drivers/dma-buf/dma-buf-mapping.c +++ b/drivers/dma-buf/dma-buf-mapping.c @@ -5,16 +5,18 @@ */ #include <linux/dma-buf-mapping.h> #include <linux/dma-resv.h> +#include <linux/overflow.h> +#include <linux/align.h> + +#define MAX_SG_ENT_SZ ALIGN_DOWN(UINT_MAX, PAGE_SIZE) static struct scatterlist *fill_sg_entry(struct scatterlist *sgl, size_t length, dma_addr_t addr) { - unsigned int len, nents; - int i; + size_t len; - nents = DIV_ROUND_UP(length, UINT_MAX); - for (i = 0; i < nents; i++) { - len = min_t(size_t, length, UINT_MAX); + while (length) { + len = min(length, MAX_SG_ENT_SZ); length -= len; /* * DMABUF abuses scatterlist to create a scatterlist @@ -24,8 +26,10 @@ static struct scatterlist *fill_sg_entry(struct scatterlist *sgl, size_t length, * does not require the CPU list for mapping or unmapping. */ sg_set_page(sgl, NULL, 0, 0); - sg_dma_address(sgl) = addr + (dma_addr_t)i * UINT_MAX; + sg_dma_address(sgl) = addr; sg_dma_len(sgl) = len; + addr += len; + /* Unconditionally advance. On last segment, this becomes NULL */ sgl = sg_next(sgl); } @@ -40,15 +44,19 @@ static unsigned int calc_sg_nents(struct dma_iova_state *state, size_t i; if (!state || !dma_use_iova(state)) { - for (i = 0; i < nr_ranges; i++) - nents += DIV_ROUND_UP(phys_vec[i].len, UINT_MAX); + for (i = 0; i < nr_ranges; i++) { + unsigned int added = DIV_ROUND_UP(phys_vec[i].len, MAX_SG_ENT_SZ); + + if (check_add_overflow(nents, added, &nents)) + return 0; + } } else { /* * In IOVA case, there is only one SG entry which spans * for whole IOVA address space, but we need to make sure * that it fits sg->length, maybe we need more. */ - nents = DIV_ROUND_UP(size, UINT_MAX); + nents = DIV_ROUND_UP(size, MAX_SG_ENT_SZ); } return nents; @@ -95,9 +103,10 @@ struct sg_table *dma_buf_phys_vec_to_sgt(struct dma_buf_attachment *attach, size_t nr_ranges, size_t size, enum dma_data_direction dir) { - unsigned int nents, mapped_len = 0; struct dma_buf_dma *dma; struct scatterlist *sgl; + size_t mapped_len = 0; + unsigned int nents; dma_addr_t addr; size_t i; int ret; @@ -133,6 +142,8 @@ struct sg_table *dma_buf_phys_vec_to_sgt(struct dma_buf_attachment *attach, } nents = calc_sg_nents(dma->state, phys_vec, nr_ranges, size); + + /* sg_alloc_table will cleanly fail and return -EINVAL if nents == 0 */ ret = sg_alloc_table(&dma->sgt, nents, GFP_KERNEL | __GFP_ZERO); if (ret) goto err_free_state; diff --git a/drivers/dma-buf/dma-fence.c b/drivers/dma-buf/dma-fence.c index 05090fb0fd5a..bd58688b81a7 100644 --- a/drivers/dma-buf/dma-fence.c +++ b/drivers/dma-buf/dma-fence.c @@ -1170,7 +1170,12 @@ const char __rcu *dma_fence_driver_name(struct dma_fence *fence) /* RCU protection is required for safe access to returned string */ ops = rcu_dereference(fence->ops); - if (ops) + + /* + * Make load ordering irrelevant by checking both signaled state and ops + * pointer and ops pointer is only set to NULL on newer implementations. + */ + if (!dma_fence_test_signaled_flag(fence) && ops) return (const char __rcu *)ops->get_driver_name(fence); else return (const char __rcu *)"detached-driver"; @@ -1203,7 +1208,12 @@ const char __rcu *dma_fence_timeline_name(struct dma_fence *fence) /* RCU protection is required for safe access to returned string */ ops = rcu_dereference(fence->ops); - if (ops) + + /* + * Make load ordering irrelevant by checking both signaled state and ops + * pointer and ops pointer is only set to NULL on newer implementations. + */ + if (!dma_fence_test_signaled_flag(fence) && ops) return (const char __rcu *)ops->get_timeline_name(fence); else return (const char __rcu *)"signaled-timeline"; diff --git a/drivers/dma/dmaengine.c b/drivers/dma/dmaengine.c index 6ffd8bd82154..c71763047126 100644 --- a/drivers/dma/dmaengine.c +++ b/drivers/dma/dmaengine.c @@ -428,11 +428,18 @@ static void dma_device_release(struct kref *ref) list_del_rcu(&device->global_node); dma_channel_rebalance(); + synchronize_rcu(); if (device->device_release) device->device_release(device); } +static int __must_check dma_device_get(struct dma_device *device) +{ + lockdep_assert_held(&dma_list_mutex); + return kref_get_unless_zero(&device->ref); +} + static void dma_device_put(struct dma_device *device) { lockdep_assert_held(&dma_list_mutex); @@ -460,8 +467,7 @@ static int dma_chan_get(struct dma_chan *chan) if (!try_module_get(owner)) return -ENODEV; - ret = kref_get_unless_zero(&chan->device->ref); - if (!ret) { + if (!dma_device_get(chan->device)) { ret = -ENODEV; goto module_put_out; } @@ -495,10 +501,13 @@ module_put_out: */ static void dma_chan_put(struct dma_chan *chan) { + struct module *owner; + /* This channel is not in use, bail out */ if (!chan->client_count) return; + owner = dma_chan_to_owner(chan); chan->client_count--; /* This channel is not in use anymore, free it */ @@ -515,8 +524,10 @@ static void dma_chan_put(struct dma_chan *chan) chan->route_data = NULL; } - dma_device_put(chan->device); - module_put(dma_chan_to_owner(chan)); + /* This channel is not in use anymore, drop the device ref */ + if (!chan->client_count) + dma_device_put(chan->device); + module_put(owner); } enum dma_status dma_sync_wait(struct dma_chan *chan, dma_cookie_t cookie) diff --git a/drivers/dma/mmp_pdma.c b/drivers/dma/mmp_pdma.c index 386e85cd4882..ed520737882b 100644 --- a/drivers/dma/mmp_pdma.c +++ b/drivers/dma/mmp_pdma.c @@ -52,7 +52,6 @@ #define DCSR_EORINTR BIT(9) /* The end of Receive */ #define DRCMR_BASE 0x0100 -#define DRCMR_EXT_BASE_K3 0x1000 #define DRCMR_EXT_BASE_DEFAULT 0x1100 #define DRCMR_REQ_LIMIT 64 #define DRCMR_MAPVLD BIT(7) /* Map Valid (read / write) */ @@ -713,7 +712,7 @@ mmp_pdma_prep_slave_sg(struct dma_chan *dchan, struct scatterlist *sgl, for_each_sg(sgl, sg, sg_len, i) { addr = sg_dma_address(sg); - avail = sg_dma_len(sgl); + avail = sg_dma_len(sg); do { len = min_t(size_t, avail, PDMA_MAX_DESC_BYTES); @@ -1219,7 +1218,7 @@ static const struct mmp_pdma_ops spacemit_k3_pdma_ops = { .get_desc_dst_addr = get_desc_dst_addr_64, .run_bits = (DCSR_RUN | DCSR_LPAEEN | DCSR_EORIRQEN | DCSR_EORSTOPEN), .dma_width = 64, - .drcmr_ext_base = DRCMR_EXT_BASE_K3, + .drcmr_ext_base = DRCMR_EXT_BASE_DEFAULT, }; static const struct of_device_id mmp_pdma_dt_ids[] = { diff --git a/drivers/dma/pxa_dma.c b/drivers/dma/pxa_dma.c index fa2ee0b3e09f..fc43124fefa8 100644 --- a/drivers/dma/pxa_dma.c +++ b/drivers/dma/pxa_dma.c @@ -744,6 +744,7 @@ pxad_alloc_desc(struct pxad_chan *chan, unsigned int nb_hw_desc) sw_desc = kzalloc_flex(*sw_desc, hw_desc, nb_hw_desc, GFP_NOWAIT); if (!sw_desc) return NULL; + sw_desc->nb_desc = nb_hw_desc; sw_desc->desc_pool = chan->desc_pool; for (i = 0; i < nb_hw_desc; i++) { @@ -752,10 +753,10 @@ pxad_alloc_desc(struct pxad_chan *chan, unsigned int nb_hw_desc) dev_err(&chan->vc.chan.dev->device, "%s(): Couldn't allocate the %dth hw_desc from dma_pool %p\n", __func__, i, sw_desc->desc_pool); + sw_desc->nb_desc = i; goto err; } - sw_desc->nb_desc++; sw_desc->hw_desc[i] = desc; if (i == 0) diff --git a/drivers/dma/sprd-dma.c b/drivers/dma/sprd-dma.c index 087fea3af2e4..19b32a23c882 100644 --- a/drivers/dma/sprd-dma.c +++ b/drivers/dma/sprd-dma.c @@ -1212,7 +1212,7 @@ static int sprd_dma_probe(struct platform_device *pdev) ret = pm_runtime_get_sync(&pdev->dev); if (ret < 0) - goto err_rpm; + goto err_register; ret = dma_async_device_register(&sdev->dma_dev); if (ret < 0) { @@ -1234,7 +1234,6 @@ err_of_register: err_register: pm_runtime_put_noidle(&pdev->dev); pm_runtime_disable(&pdev->dev); -err_rpm: sprd_dma_disable(sdev); return ret; } diff --git a/drivers/dma/sun6i-dma.c b/drivers/dma/sun6i-dma.c index f47a326dd7ff..7704b016aed8 100644 --- a/drivers/dma/sun6i-dma.c +++ b/drivers/dma/sun6i-dma.c @@ -354,8 +354,10 @@ static size_t sun6i_get_chan_size(struct sun6i_pchan *pchan) size_t bytes; dma_addr_t pos; - pos = readl(pchan->base + DMA_CHAN_LLI_ADDR); - bytes = readl(pchan->base + DMA_CHAN_CUR_CNT); + do { + pos = readl(pchan->base + DMA_CHAN_LLI_ADDR); + bytes = readl(pchan->base + DMA_CHAN_CUR_CNT); + } while (pos != readl(pchan->base + DMA_CHAN_LLI_ADDR)); if (pos == LLI_LAST_ITEM) return bytes; @@ -979,7 +981,6 @@ static enum dma_status sun6i_dma_tx_status(struct dma_chan *chan, struct sun6i_pchan *pchan = vchan->phy; struct sun6i_dma_lli *lli; struct virt_dma_desc *vd; - struct sun6i_desc *txd; enum dma_status ret; unsigned long flags; size_t bytes = 0; @@ -991,9 +992,9 @@ static enum dma_status sun6i_dma_tx_status(struct dma_chan *chan, spin_lock_irqsave(&vchan->vc.lock, flags); vd = vchan_find_desc(&vchan->vc, cookie); - txd = to_sun6i_desc(&vd->tx); if (vd) { + struct sun6i_desc *txd = to_sun6i_desc(&vd->tx); for (lli = txd->v_lli; lli != NULL; lli = lli->v_lli_next) bytes += lli->len; } else if (!pchan || !pchan->desc) { diff --git a/drivers/dma/ti/k3-udma-glue.c b/drivers/dma/ti/k3-udma-glue.c index 686dc140293e..70eaf7ee57e6 100644 --- a/drivers/dma/ti/k3-udma-glue.c +++ b/drivers/dma/ti/k3-udma-glue.c @@ -1243,8 +1243,9 @@ void k3_udma_glue_release_rx_chn(struct k3_udma_glue_rx_channel *rx_chn) rx_chn->psil_paired = false; } - for (i = 0; i < rx_chn->flow_num; i++) - k3_udma_glue_release_rx_flow(rx_chn, i); + if (rx_chn->flows) + for (i = 0; i < rx_chn->flow_num; i++) + k3_udma_glue_release_rx_flow(rx_chn, i); if (xudma_rflow_is_gp(rx_chn->common.udmax, rx_chn->flow_id_base)) xudma_free_gp_rflow_range(rx_chn->common.udmax, diff --git a/drivers/dma/xilinx/xilinx_dma.c b/drivers/dma/xilinx/xilinx_dma.c index bef2b031dba1..cffe7c6fa640 100644 --- a/drivers/dma/xilinx/xilinx_dma.c +++ b/drivers/dma/xilinx/xilinx_dma.c @@ -756,15 +756,25 @@ xilinx_aximcdma_alloc_tx_segment(struct xilinx_dma_chan *chan) return segment; } -static void xilinx_dma_clean_hw_desc(struct xilinx_axidma_desc_hw *hw) +static void xilinx_dma_clean_hw_desc(struct xilinx_dma_chan *chan, + struct xilinx_axidma_tx_segment *segment) { - u32 next_desc = hw->next_desc; - u32 next_desc_msb = hw->next_desc_msb; + dma_addr_t next; + u32 i; - memset(hw, 0, sizeof(struct xilinx_axidma_desc_hw)); + /* + * Restore the buffer descriptor's next descriptor pointer to the value + * set up in xilinx_dma_alloc_chan_resources(). Otherwise using the DMA + * in cyclic mode leaves the next descriptor pointer altered and + * prevents subsequent non-cyclic transfers. + */ + i = segment - chan->seg_v; + next = chan->seg_p + + sizeof(*chan->seg_v) * ((i + 1) % XILINX_DMA_NUM_DESCS); - hw->next_desc = next_desc; - hw->next_desc_msb = next_desc_msb; + memset(&segment->hw, 0, sizeof(segment->hw)); + segment->hw.next_desc = lower_32_bits(next); + segment->hw.next_desc_msb = upper_32_bits(next); } static void xilinx_mcdma_clean_hw_desc(struct xilinx_aximcdma_desc_hw *hw) @@ -786,7 +796,7 @@ static void xilinx_mcdma_clean_hw_desc(struct xilinx_aximcdma_desc_hw *hw) static void xilinx_dma_free_tx_segment(struct xilinx_dma_chan *chan, struct xilinx_axidma_tx_segment *segment) { - xilinx_dma_clean_hw_desc(&segment->hw); + xilinx_dma_clean_hw_desc(chan, segment); list_add_tail(&segment->node, &chan->free_seg_list); } @@ -920,9 +930,9 @@ static void xilinx_dma_free_descriptors(struct xilinx_dma_chan *chan) spin_lock_irqsave(&chan->lock, flags); - xilinx_dma_free_desc_list(chan, &chan->pending_list); xilinx_dma_free_desc_list(chan, &chan->done_list); xilinx_dma_free_desc_list(chan, &chan->active_list); + xilinx_dma_free_desc_list(chan, &chan->pending_list); spin_unlock_irqrestore(&chan->lock, flags); } diff --git a/drivers/dpll/dpll_netlink.c b/drivers/dpll/dpll_netlink.c index 523d76a5fd49..45365214fbef 100644 --- a/drivers/dpll/dpll_netlink.c +++ b/drivers/dpll/dpll_netlink.c @@ -1202,6 +1202,7 @@ dpll_pin_ref_sync_state_set(struct dpll_pin *pin, const enum dpll_pin_state state, struct netlink_ext_ack *extack) { + void *pin_priv, *ref_sync_pin_priv; const struct dpll_pin_ops *ops; enum dpll_pin_state old_state; struct dpll_pin *ref_sync_pin; @@ -1230,9 +1231,15 @@ dpll_pin_ref_sync_state_set(struct dpll_pin *pin, return -EOPNOTSUPP; } dpll = ref->dpll; - ret = ops->ref_sync_get(pin, dpll_pin_on_dpll_priv(dpll, pin), - ref_sync_pin, - dpll_pin_on_dpll_priv(dpll, ref_sync_pin), + pin_priv = dpll_pin_on_dpll_priv(dpll, pin); + ref_sync_pin_priv = dpll_pin_on_dpll_priv(dpll, ref_sync_pin); + /* Pin may have been unregistered from this dpll already */ + if (!ref_sync_pin_priv) { + NL_SET_ERR_MSG(extack, + "reference sync pin not registered with the dpll"); + return -ENODEV; + } + ret = ops->ref_sync_get(pin, pin_priv, ref_sync_pin, ref_sync_pin_priv, &old_state, extack); if (ret) { NL_SET_ERR_MSG(extack, "unable to get old reference sync state"); @@ -1241,9 +1248,7 @@ dpll_pin_ref_sync_state_set(struct dpll_pin *pin, if (state == old_state) return 0; - ret = ops->ref_sync_set(pin, dpll_pin_on_dpll_priv(dpll, pin), - ref_sync_pin, - dpll_pin_on_dpll_priv(dpll, ref_sync_pin), + ret = ops->ref_sync_set(pin, pin_priv, ref_sync_pin, ref_sync_pin_priv, state, extack); if (ret) { NL_SET_ERR_MSG_FMT(extack, diff --git a/drivers/firmware/arm_ffa/driver.c b/drivers/firmware/arm_ffa/driver.c index 8654b3365c9b..28abc808fd32 100644 --- a/drivers/firmware/arm_ffa/driver.c +++ b/drivers/firmware/arm_ffa/driver.c @@ -2225,6 +2225,7 @@ static void ffa_remove(struct platform_device *pdev) static struct platform_driver ffa_driver = { .probe = ffa_probe, .remove = ffa_remove, + .shutdown = ffa_remove, .driver = { .name = FFA_PLATFORM_NAME, }, diff --git a/drivers/firmware/arm_scmi/driver.c b/drivers/firmware/arm_scmi/driver.c index 922777e86d58..a666ad39acdf 100644 --- a/drivers/firmware/arm_scmi/driver.c +++ b/drivers/firmware/arm_scmi/driver.c @@ -477,7 +477,7 @@ void *scmi_notification_instance_data_get(const struct scmi_handle *handle) * - exactly 'next_token' may be NOT available so pick xfer_id >= next_token * using find_next_zero_bit() starting from candidate next_token bit * - * - all tokens ahead upto (MSG_TOKEN_ID_MASK - 1) are used in-flight but we + * - all tokens ahead up to (MSG_TOKEN_ID_MASK - 1) are used in-flight but we * are plenty of free tokens at start, so try a second pass using * find_next_zero_bit() and starting from 0. * diff --git a/drivers/firmware/arm_scpi.c b/drivers/firmware/arm_scpi.c index 2acad5fa5a28..68a730d22037 100644 --- a/drivers/firmware/arm_scpi.c +++ b/drivers/firmware/arm_scpi.c @@ -631,8 +631,8 @@ static struct scpi_dvfs_info *scpi_dvfs_get_info(u8 domain) if (ret) return ERR_PTR(ret); - if (!buf.opp_count) - return ERR_PTR(-ENOENT); + if (!buf.opp_count || buf.opp_count > MAX_DVFS_OPPS) + return ERR_PTR(-EINVAL); info = kmalloc_obj(*info); if (!info) @@ -661,12 +661,15 @@ static struct scpi_dvfs_info *scpi_dvfs_get_info(u8 domain) static int scpi_dev_domain_id(struct device *dev) { struct of_phandle_args clkspec; + int domain; if (of_parse_phandle_with_args(dev->of_node, "clocks", "#clock-cells", 0, &clkspec)) return -EINVAL; - return clkspec.args[0]; + domain = clkspec.args[0]; + of_node_put(clkspec.np); + return domain; } static struct scpi_dvfs_info *scpi_dvfs_info(struct device *dev) diff --git a/drivers/gpio/gpio-virtuser.c b/drivers/gpio/gpio-virtuser.c index 449fb1aed24b..70ddccccc1c4 100644 --- a/drivers/gpio/gpio-virtuser.c +++ b/drivers/gpio/gpio-virtuser.c @@ -692,7 +692,8 @@ static int gpio_virtuser_interrupts_set(void *data, u64 val) atomic_set(&ld->irq, irq); } else { irq = atomic_xchg(&ld->irq, 0); - free_irq(irq, ld); + if (irq) + free_irq(irq, ld); } return 0; diff --git a/drivers/gpio/gpiolib-of.c b/drivers/gpio/gpiolib-of.c index 940b566946ce..f36e4b171fa7 100644 --- a/drivers/gpio/gpiolib-of.c +++ b/drivers/gpio/gpiolib-of.c @@ -788,13 +788,13 @@ static int of_gpio_notify(struct notifier_block *nb, unsigned long action, if (!of_property_read_bool(rd->dn, "gpio-hog")) return NOTIFY_DONE; /* not for us */ - if (of_node_test_and_set_flag(rd->dn, OF_POPULATED)) - return NOTIFY_DONE; - gdev = of_find_gpio_device_by_node(rd->dn->parent); if (!gdev) return NOTIFY_DONE; /* not for us */ + if (of_node_test_and_set_flag(rd->dn, OF_POPULATED)) + return NOTIFY_DONE; + ret = gpiochip_add_hog(gpio_device_get_chip(gdev), of_fwnode_handle(rd->dn)); if (ret < 0) { pr_err("%s: failed to add hogs for %pOF\n", __func__, diff --git a/drivers/gpio/gpiolib-shared.c b/drivers/gpio/gpiolib-shared.c index 495bd3d0ddf0..5f9623e40b0f 100644 --- a/drivers/gpio/gpiolib-shared.c +++ b/drivers/gpio/gpiolib-shared.c @@ -261,10 +261,13 @@ static int gpio_shared_of_traverse(struct device_node *curr) con_id[con_id_len - suffix_len] = '\0'; } - ref = gpio_shared_make_ref(fwnode_handle_get(of_fwnode_handle(curr)), - con_id, args.args[1]); - if (!ref) + struct fwnode_handle *curr_fwnode = + fwnode_handle_get(of_fwnode_handle(curr)); + ref = gpio_shared_make_ref(curr_fwnode, con_id, args.args[1]); + if (!ref) { + fwnode_handle_put(curr_fwnode); return -ENOMEM; + } if (!list_empty(&entry->refs)) pr_debug("GPIO %u at %s is shared by multiple firmware nodes\n", diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu.h b/drivers/gpu/drm/amd/amdgpu/amdgpu.h index 7974f9b7944f..a9c6f5d4a639 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu.h +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu.h @@ -621,6 +621,9 @@ enum amdgpu_enforce_isolation_mode { struct amdgpu_device { struct device *dev; struct pci_dev *pdev; + /* The two ends of the physical PCIe link outside the device. */ + struct pci_dev *link_dev; + struct pci_dev *link_partner; struct drm_device ddev; #ifdef CONFIG_DRM_AMD_ACP diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd.c index 816d8817f0b2..054870e9078d 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd.c +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd.c @@ -330,7 +330,7 @@ void amdgpu_amdkfd_clear_kfd_mapping(struct amdgpu_device *adev) struct kfd_dev *kfd = adev->kfd.dev; unsigned int i; - if (!kfd) + if (!kfd || !kfd->init_complete) return; for (i = 0; i < kfd->num_nodes; i++) { diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c index 104d1d2cbad9..9269e780feb7 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c @@ -1954,18 +1954,17 @@ static void amdgpu_uid_fini(struct amdgpu_device *adev) adev->uid_info = NULL; } -static struct pci_dev *amdgpu_device_find_parent(struct amdgpu_device *adev) +static void amdgpu_device_init_pcie_links(struct amdgpu_device *adev) { - struct pci_dev *parent = adev->pdev; + adev->link_dev = adev->pdev; + adev->link_partner = pci_upstream_bridge(adev->link_dev); - /* skip upstream/downstream switches internal to dGPU */ - while ((parent = pci_upstream_bridge(parent))) { - if (parent->vendor == PCI_VENDOR_ID_ATI) - continue; - break; + /* Skip upstream/downstream switches internal to the dGPU. */ + while (adev->link_partner && + adev->link_partner->vendor == PCI_VENDOR_ID_ATI) { + adev->link_dev = adev->link_partner; + adev->link_partner = pci_upstream_bridge(adev->link_dev); } - - return parent; } /** @@ -1981,7 +1980,6 @@ static struct pci_dev *amdgpu_device_find_parent(struct amdgpu_device *adev) static int amdgpu_device_ip_early_init(struct amdgpu_device *adev) { struct amdgpu_ip_block *ip_block; - struct pci_dev *parent; bool total, skip_bios, early_full_gpu_access = false; uint32_t bios_flags; int i, r; @@ -2077,10 +2075,9 @@ static int amdgpu_device_ip_early_init(struct amdgpu_device *adev) !dev_is_removable(&adev->pdev->dev)) adev->flags |= AMD_IS_PX; - if (!(adev->flags & AMD_IS_APU)) { - parent = amdgpu_device_find_parent(adev); - adev->has_pr3 = parent ? pci_pr3_present(parent) : false; - } + if (!(adev->flags & AMD_IS_APU)) + adev->has_pr3 = adev->link_partner && + pci_pr3_present(adev->link_partner); adev->pm.pp_feature = amdgpu_pp_feature_mask; if (amdgpu_sriov_vf(adev) || sched_policy == KFD_SCHED_POLICY_NO_HWS) @@ -3776,6 +3773,7 @@ int amdgpu_device_init(struct amdgpu_device *adev, adev->shutdown = false; adev->flags = flags; + amdgpu_device_init_pcie_links(adev); if (amdgpu_force_asic_type >= 0 && amdgpu_force_asic_type < CHIP_LAST) adev->asic_type = amdgpu_force_asic_type; @@ -4337,7 +4335,7 @@ void amdgpu_device_fini_hw(struct amdgpu_device *adev) void amdgpu_device_fini_sw(struct amdgpu_device *adev) { - int i, idx; + int i; bool px; amdgpu_device_ip_fini(adev); @@ -4379,11 +4377,9 @@ void amdgpu_device_fini_sw(struct amdgpu_device *adev) if ((adev->pdev->class >> 8) == PCI_CLASS_DISPLAY_VGA) vga_client_unregister(adev->pdev); - if (drm_dev_enter(adev_to_drm(adev), &idx)) { - + if (adev->rmmio) { iounmap(adev->rmmio); adev->rmmio = NULL; - drm_dev_exit(idx); } if (IS_ENABLED(CONFIG_PERF_EVENTS)) @@ -5872,11 +5868,9 @@ static void amdgpu_device_partner_bandwidth(struct amdgpu_device *adev, *width = PCIE_LNK_WIDTH_UNKNOWN; if (amdgpu_device_pcie_dynamic_switching_supported(adev)) { - struct pci_dev *parent = amdgpu_device_find_parent(adev); - - if (parent) { - *speed = pcie_get_speed_cap(parent); - *width = pcie_get_width_cap(parent); + if (adev->link_partner) { + *speed = pcie_get_speed_cap(adev->link_partner); + *width = pcie_get_width_cap(adev->link_partner); } } else { /* use the current speeds rather than max if switching is not supported */ @@ -5898,21 +5892,11 @@ static void amdgpu_device_gpu_bandwidth(struct amdgpu_device *adev, enum pci_bus_speed *speed, enum pcie_link_width *width) { - struct pci_dev *parent = adev->pdev; - if (!speed || !width) return; - /* use the device itself */ - *speed = pcie_get_speed_cap(adev->pdev); - *width = pcie_get_width_cap(adev->pdev); - - /* use the link outside the device */ - parent = amdgpu_device_find_parent(adev); - if (parent) { - *speed = pcie_get_speed_cap(parent); - *width = pcie_get_width_cap(parent); - } + *speed = pcie_get_speed_cap(adev->link_dev); + *width = pcie_get_width_cap(adev->link_dev); } /** diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_dma_buf.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_dma_buf.c index b33c300e26e2..9adf3eed8822 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_dma_buf.c +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_dma_buf.c @@ -43,6 +43,7 @@ #include <linux/dma-buf.h> #include <linux/dma-fence-array.h> #include <linux/pci-p2pdma.h> +#include <linux/pm_runtime.h> static const struct dma_buf_attach_ops amdgpu_dma_buf_attach_ops; @@ -100,15 +101,54 @@ static int amdgpu_dma_buf_attach(struct dma_buf *dmabuf, pci_p2pdma_distance(adev->pdev, attach->dev, false) < 0) attach->peer2peer = false; + /* + * Only allow P2P while the exporter is active, and keep it active + * until detach. With runtime PM disabled take a plain reference so + * the put in detach stays balanced. + */ + if (attach->peer2peer) { + struct device *dev = adev_to_drm(adev)->dev; + int ret = pm_runtime_get_if_active(dev); + + if (!ret) + attach->peer2peer = false; + else if (ret < 0) + pm_runtime_get_noresume(dev); + } + r = dma_resv_lock(bo->tbo.base.resv, NULL); if (r) - return r; + goto err_pm_put; amdgpu_vm_bo_update_shared(bo); dma_resv_unlock(bo->tbo.base.resv); return 0; + +err_pm_put: + if (attach->peer2peer) + pm_runtime_put_autosuspend(adev_to_drm(adev)->dev); + return r; +} + +/** + * amdgpu_dma_buf_detach - &dma_buf_ops.detach implementation + * + * @dmabuf: DMA-buf where we remove the attachment from + * @attach: the attachment to remove + * + * Drop the runtime PM reference taken in amdgpu_dma_buf_attach(). + */ +static void amdgpu_dma_buf_detach(struct dma_buf *dmabuf, + struct dma_buf_attachment *attach) +{ + struct drm_gem_object *obj = dmabuf->priv; + struct amdgpu_bo *bo = gem_to_amdgpu_bo(obj); + struct amdgpu_device *adev = amdgpu_ttm_adev(bo->tbo.bdev); + + if (attach->peer2peer) + pm_runtime_put_autosuspend(adev_to_drm(adev)->dev); } /** @@ -350,6 +390,7 @@ static void amdgpu_dma_buf_vunmap(struct dma_buf *dma_buf, struct iosys_map *map const struct dma_buf_ops amdgpu_dmabuf_ops = { .attach = amdgpu_dma_buf_attach, + .detach = amdgpu_dma_buf_detach, .pin = amdgpu_dma_buf_pin, .unpin = amdgpu_dma_buf_unpin, .map_dma_buf = amdgpu_dma_buf_map, diff --git a/drivers/gpu/drm/amd/amdgpu/nbio_v7_9.c b/drivers/gpu/drm/amd/amdgpu/nbio_v7_9.c index bdfd2917e3ca..def02993b7cf 100644 --- a/drivers/gpu/drm/amd/amdgpu/nbio_v7_9.c +++ b/drivers/gpu/drm/amd/amdgpu/nbio_v7_9.c @@ -535,7 +535,7 @@ static void nbio_v7_9_handle_ras_controller_intr_no_bifring(struct amdgpu_device RAS_CNTLR_INTERRUPT_CLEAR, 1); WREG32_SOC15(NBIO, 0, regBIF_BX0_BIF_DOORBELL_INT_CNTL, bif_doorbell_intr_cntl); - if (!ras->disable_ras_err_cnt_harvest) { + if (ras && !ras->disable_ras_err_cnt_harvest && obj) { /* * clear error status after ras_controller_intr * according to hw team and count ue number diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c index 9811e4e10291..2f78395a0c31 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c @@ -770,10 +770,12 @@ static int create_queue_nocpsch(struct device_queue_manager *dqm, mqd_mgr = dqm->mqd_mgrs[get_mqd_type_from_queue_type( q->properties.type)]; if (qd && !mqd_mgr->restore_mqd) { - pr_debug("restore_mqd not implemented for this GPU\n"); + pr_debug("restore_mqd not implemented for queue type %d\n", + q->properties.type); retval = -EOPNOTSUPP; goto deallocate_vmid; } + if (q->properties.type == KFD_QUEUE_TYPE_COMPUTE) { retval = allocate_hqd(dqm, q); if (retval) @@ -2250,7 +2252,8 @@ static int create_queue_cpsch(struct device_queue_manager *dqm, struct queue *q, mqd_mgr = dqm->mqd_mgrs[get_mqd_type_from_queue_type( q->properties.type)]; if (qd && !mqd_mgr->restore_mqd) { - pr_debug("restore_mqd not implemented for this GPU\n"); + pr_debug("restore_mqd not implemented for queue type %d\n", + q->properties.type); retval = -EOPNOTSUPP; goto out_deallocate_doorbell; } diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v10.c b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v10.c index e034da638c07..4f8a8a1a6186 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v10.c +++ b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v10.c @@ -204,7 +204,7 @@ static void update_mqd(struct mqd_manager *mm, void *mqd, * is safe, giving a maximum field value of 0xA. */ m->cp_hqd_eop_control = q->eop_ring_buffer_size ? min(0xA, - ffs(q->eop_ring_buffer_size / sizeof(unsigned int)) - 1 - 1) : 0; + ffs(q->eop_ring_buffer_size / sizeof(unsigned int) / 4)) : 0; m->cp_hqd_eop_base_addr_lo = lower_32_bits(q->eop_ring_buffer_address >> 8); m->cp_hqd_eop_base_addr_hi = diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v11.c b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v11.c index 350fcbbba4b2..bf015dc5b868 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v11.c +++ b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v11.c @@ -242,7 +242,7 @@ static void update_mqd(struct mqd_manager *mm, void *mqd, * is safe, giving a maximum field value of 0xA. */ m->cp_hqd_eop_control = q->eop_ring_buffer_size ? min(0xA, - ffs(q->eop_ring_buffer_size / sizeof(unsigned int)) - 1 - 1) : 0; + ffs(q->eop_ring_buffer_size / sizeof(unsigned int) / 4)) : 0; m->cp_hqd_eop_base_addr_lo = lower_32_bits(q->eop_ring_buffer_address >> 8); m->cp_hqd_eop_base_addr_hi = diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12.c b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12.c index 7c387fa90076..6ea09b031caf 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12.c +++ b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12.c @@ -217,7 +217,7 @@ static void update_mqd(struct mqd_manager *mm, void *mqd, * is safe, giving a maximum field value of 0xA. */ m->cp_hqd_eop_control = q->eop_ring_buffer_size ? min(0xA, - ffs(q->eop_ring_buffer_size / sizeof(unsigned int)) - 1 - 1) : 0; + ffs(q->eop_ring_buffer_size / sizeof(unsigned int) / 4)) : 0; m->cp_hqd_eop_base_addr_lo = lower_32_bits(q->eop_ring_buffer_address >> 8); m->cp_hqd_eop_base_addr_hi = @@ -380,6 +380,63 @@ static int debugfs_show_mqd_sdma(struct seq_file *m, void *data) #endif +static void restore_mqd(struct mqd_manager *mm, void **mqd, + struct kfd_mem_obj *mqd_mem_obj, uint64_t *gart_addr, + struct queue_properties *qp, const void *mqd_src, + const void *ctl_stack_src, const u32 ctl_stack_size) +{ + u64 addr; + struct v12_compute_mqd *m; + + m = (struct v12_compute_mqd *)mqd_mem_obj->cpu_ptr; + addr = mqd_mem_obj->gpu_addr; + + memset(m, 0, AMDGPU_MQD_SIZE_ALIGN(mm->mqd_size)); + memcpy(m, mqd_src, sizeof(*m)); + + /* Update MQD base address to the newly allocated location */ + m->cp_mqd_base_addr_lo = lower_32_bits(addr); + m->cp_mqd_base_addr_hi = upper_32_bits(addr); + + m->cp_hqd_pq_doorbell_control &= + ~CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET_MASK; + m->cp_hqd_pq_doorbell_control |= + qp->doorbell_off << CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET__SHIFT; + pr_debug("cp_hqd_pq_doorbell_control 0x%x\n", m->cp_hqd_pq_doorbell_control); + + *mqd = m; + if (gart_addr) + *gart_addr = addr; + + qp->is_active = 0; +} + +static void restore_mqd_sdma(struct mqd_manager *mm, void **mqd, + struct kfd_mem_obj *mqd_mem_obj, uint64_t *gart_addr, + struct queue_properties *qp, + const void *mqd_src, + const void *ctl_stack_src, + const u32 ctl_stack_size) +{ + u64 addr; + struct v12_sdma_mqd *m; + + m = (struct v12_sdma_mqd *)mqd_mem_obj->cpu_ptr; + addr = mqd_mem_obj->gpu_addr; + + memset(m, 0, AMDGPU_MQD_SIZE_ALIGN(mm->mqd_size)); + memcpy(m, mqd_src, sizeof(*m)); + + m->sdmax_rlcx_doorbell_offset = + qp->doorbell_off << SDMA0_QUEUE0_DOORBELL_OFFSET__OFFSET__SHIFT; + + *mqd = m; + if (gart_addr) + *gart_addr = addr; + + qp->is_active = 0; +} + struct mqd_manager *mqd_manager_init_v12(enum KFD_MQD_TYPE type, struct kfd_node *dev) { @@ -407,6 +464,7 @@ struct mqd_manager *mqd_manager_init_v12(enum KFD_MQD_TYPE type, mqd->mqd_size = sizeof(struct v12_compute_mqd); mqd->get_wave_state = get_wave_state; mqd->mqd_stride = kfd_mqd_stride; + mqd->restore_mqd = restore_mqd; #if defined(CONFIG_DEBUG_FS) mqd->debugfs_show_mqd = debugfs_show_mqd; #endif @@ -453,6 +511,7 @@ struct mqd_manager *mqd_manager_init_v12(enum KFD_MQD_TYPE type, mqd->is_occupied = kfd_is_occupied_sdma; mqd->mqd_size = sizeof(struct v12_sdma_mqd); mqd->mqd_stride = kfd_mqd_stride; + mqd->restore_mqd = restore_mqd_sdma; #if defined(CONFIG_DEBUG_FS) mqd->debugfs_show_mqd = debugfs_show_mqd_sdma; #endif diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12_1.c b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12_1.c index 431a940f91f3..c709db0210ce 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12_1.c +++ b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12_1.c @@ -295,7 +295,7 @@ static void update_mqd(struct mqd_manager *mm, void *mqd, * is safe, giving a maximum field value of 0xA. */ m->cp_hqd_eop_control = q->eop_ring_buffer_size ? min(0xA, - ffs(q->eop_ring_buffer_size / sizeof(unsigned int)) - 1 - 1) : 0; + ffs(q->eop_ring_buffer_size / sizeof(unsigned int) / 4)) : 0; m->cp_hqd_eop_base_addr_lo = lower_32_bits(q->eop_ring_buffer_address >> 8); m->cp_hqd_eop_base_addr_hi = @@ -641,6 +641,72 @@ static int debugfs_show_mqd_sdma(struct seq_file *m, void *data) #endif +static void restore_mqd_v12_1(struct mqd_manager *mm, void **mqd, + struct kfd_mem_obj *mqd_mem_obj, uint64_t *gart_addr, + struct queue_properties *qp, const void *mqd_src, + const void *ctl_stack_src, const u32 ctl_stack_size) +{ + u64 addr; + struct v12_1_compute_mqd *m; + + /* + * GFX12.1 is multi-XCC capable but this restore handles XCC0 only. + * Multi-XCC CRIU restore is currently unreachable because + * kfd_criu_restore_queue() validates against unscaled mqd_size. + */ + if (NUM_XCC(mm->dev->xcc_mask) > 1) + pr_warn_once("GFX12.1 multi-XCC CRIU restore not fully supported\n"); + + m = (struct v12_1_compute_mqd *)mqd_mem_obj->cpu_ptr; + addr = mqd_mem_obj->gpu_addr; + + memset(m, 0, AMDGPU_MQD_SIZE_ALIGN(mm->mqd_size) * + NUM_XCC(mm->dev->xcc_mask)); + memcpy(m, mqd_src, sizeof(*m)); + + /* Update MQD base address to the newly allocated location */ + m->cp_mqd_base_addr_lo = lower_32_bits(addr); + m->cp_mqd_base_addr_hi = upper_32_bits(addr); + + m->cp_hqd_pq_doorbell_control &= + ~CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET_MASK; + m->cp_hqd_pq_doorbell_control |= + qp->doorbell_off << CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET__SHIFT; + pr_debug("cp_hqd_pq_doorbell_control 0x%x\n", m->cp_hqd_pq_doorbell_control); + + *mqd = m; + if (gart_addr) + *gart_addr = addr; + + qp->is_active = 0; +} + +static void restore_mqd_sdma_v12_1(struct mqd_manager *mm, void **mqd, + struct kfd_mem_obj *mqd_mem_obj, uint64_t *gart_addr, + struct queue_properties *qp, + const void *mqd_src, + const void *ctl_stack_src, + const u32 ctl_stack_size) +{ + u64 addr; + struct v12_sdma_mqd *m; + + m = (struct v12_sdma_mqd *)mqd_mem_obj->cpu_ptr; + addr = mqd_mem_obj->gpu_addr; + + memset(m, 0, AMDGPU_MQD_SIZE_ALIGN(mm->mqd_size)); + memcpy(m, mqd_src, sizeof(*m)); + + m->sdmax_rlcx_doorbell_offset = + qp->doorbell_off << SDMA0_SDMA_QUEUE0_DOORBELL_OFFSET__OFFSET__SHIFT; + + *mqd = m; + if (gart_addr) + *gart_addr = addr; + + qp->is_active = 0; +} + struct mqd_manager *mqd_manager_init_v12_1(enum KFD_MQD_TYPE type, struct kfd_node *dev) { @@ -668,6 +734,7 @@ struct mqd_manager *mqd_manager_init_v12_1(enum KFD_MQD_TYPE type, mqd->mqd_size = sizeof(struct v12_1_compute_mqd); mqd->get_wave_state = get_wave_state_v12_1; mqd->mqd_stride = kfd_mqd_stride; + mqd->restore_mqd = restore_mqd_v12_1; #if defined(CONFIG_DEBUG_FS) mqd->debugfs_show_mqd = debugfs_show_mqd; #endif @@ -714,6 +781,7 @@ struct mqd_manager *mqd_manager_init_v12_1(enum KFD_MQD_TYPE type, mqd->is_occupied = kfd_is_occupied_sdma; mqd->mqd_size = sizeof(struct v12_sdma_mqd); mqd->mqd_stride = kfd_mqd_stride; + mqd->restore_mqd = restore_mqd_sdma_v12_1; #if defined(CONFIG_DEBUG_FS) mqd->debugfs_show_mqd = debugfs_show_mqd_sdma; #endif diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v9.c b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v9.c index b95720198e28..6e6bc1ec0b64 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v9.c +++ b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v9.c @@ -285,6 +285,10 @@ static void update_mqd(struct mqd_manager *mm, void *mqd, 1 << CP_HQD_IB_CONTROL__IB_EXE_DISABLE__SHIFT; /* + * The lowest 6 bits of eop_control store the EOP ring size. If + * their value is X, the ring size is 2^(X + 1) dwords, or + * 2^(X + 3) bytes. + * * HW does not clamp this field correctly. Maximum EOP queue size * is constrained by per-SE EOP done signal count, which is 8-bit. * Limit is 0xFF EOP entries (= 0x7F8 dwords). CP will not submit @@ -296,7 +300,7 @@ static void update_mqd(struct mqd_manager *mm, void *mqd, * */ m->cp_hqd_eop_control = q->eop_ring_buffer_size ? - min(0xA, order_base_2(q->eop_ring_buffer_size / 4) - 1) : 0; + min(0xA, order_base_2(q->eop_ring_buffer_size / 8)) : 0; m->cp_hqd_eop_base_addr_lo = lower_32_bits(q->eop_ring_buffer_address >> 8); diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_vi.c b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_vi.c index 60b87a500698..029572548c14 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_vi.c +++ b/drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_vi.c @@ -208,6 +208,9 @@ static void __update_mqd(struct mqd_manager *mm, void *mqd, mtype << CP_HQD_IB_CONTROL__MTYPE__SHIFT; /* + * The lowest 6 bits of eop_control store the EOP ring size. If + * their value is X, the ring size is 2^(X + 1) dwords, or + * 2^(X + 3) bytes. * HW does not clamp this field correctly. Maximum EOP queue size * is constrained by per-SE EOP done signal count, which is 8-bit. * Limit is 0xFF EOP entries (= 0x7F8 dwords). CP will not submit @@ -215,7 +218,7 @@ static void __update_mqd(struct mqd_manager *mm, void *mqd, * is safe, giving a maximum field value of 0xA. */ m->cp_hqd_eop_control |= q->eop_ring_buffer_size ? min(0xA, - order_base_2(q->eop_ring_buffer_size / 4) - 1) : 0; + order_base_2(q->eop_ring_buffer_size / 8)) : 0; m->cp_hqd_eop_base_addr_lo = lower_32_bits(q->eop_ring_buffer_address >> 8); m->cp_hqd_eop_base_addr_hi = diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c index 08b8605029ab..36d2f86f000a 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c @@ -1420,7 +1420,7 @@ static void dm_gpureset_toggle_interrupts(struct amdgpu_device *adev, if (acrtc && state->stream_status[i].plane_count != 0 && amdgpu_ip_version(adev, DCE_HWIP, 0) == 0) { irq_source = IRQ_TYPE_PFLIP + acrtc->otg_inst; - rc = dc_interrupt_set(adev->dm.dc, irq_source, enable) ? 0 : -EBUSY; + rc = amdgpu_dm_irq_set(adev, irq_source, enable) ? 0 : -EBUSY; if (rc) drm_warn(adev_to_drm(adev), "Failed to %s pflip interrupts\n", enable ? "enable" : "disable"); @@ -1444,7 +1444,7 @@ static void dm_gpureset_toggle_interrupts(struct amdgpu_device *adev, /* During gpu-reset we disable and then enable vblank irq, so * don't use amdgpu_irq_get/put() to avoid refcount change. */ - if (!dc_interrupt_set(adev->dm.dc, irq_source, enable)) + if (!amdgpu_dm_irq_set(adev, irq_source, enable)) drm_warn(adev_to_drm(adev), "Failed to %sable vblank interrupt\n", enable ? "en" : "dis"); } else if (acrtc && state->stream_status[i].plane_count != 0) { diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.h b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.h index 3524931451c8..881c8d1c3cc0 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.h +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.h @@ -553,6 +553,18 @@ struct amdgpu_display_manager { vupdate_params[DC_IRQ_SOURCE_VUPDATE6 - DC_IRQ_SOURCE_VUPDATE1 + 1]; /** + * @irq_reg_lock: + * + * Serializes the read-modify-writes of the HW interrupt control + * registers. Several interrupt sources share one register - e.g. the + * enable and clear bits of both VSTARTUP (vblank) and VUPDATE_NO_LOCK + * live in OTG_GLOBAL_SYNC_STATUS. Therefore, enabling one source must + * not race with acking another. Held only across amdgpu_dm_irq_set() + * and amdgpu_dm_irq_ack(). + */ + spinlock_t irq_reg_lock; + + /** * @dmub_trace_params: * * DMUB trace event IRQ parameters, passed to registered handlers when diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.c index 62eac6e65334..1d941be73561 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.c @@ -31,6 +31,7 @@ #include "amdgpu_dm_psr.h" #include "amdgpu_dm_replay.h" #include "amdgpu_dm_crtc.h" +#include "amdgpu_dm_irq.h" #include "amdgpu_dm_plane.h" #include "amdgpu_dm_trace.h" #include "amdgpu_dm_debugfs.h" @@ -91,7 +92,7 @@ int amdgpu_dm_crtc_set_vupdate_irq(struct drm_crtc *crtc, bool enable) irq_source = IRQ_TYPE_VUPDATE + acrtc->otg_inst; - rc = dc_interrupt_set(adev->dm.dc, irq_source, enable) ? 0 : -EBUSY; + rc = amdgpu_dm_irq_set(adev, irq_source, enable) ? 0 : -EBUSY; DRM_DEBUG_VBL("crtc %d - vupdate irq %sabling: r=%d\n", acrtc->crtc_id, enable ? "en" : "dis", rc); diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c index 298de7b75ca8..ced8b3d2d762 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c @@ -1439,12 +1439,13 @@ void dm_helpers_free_gpu_mem( bool dm_helpers_dmub_outbox_interrupt_control(struct dc_context *ctx, bool enable) { + struct amdgpu_device *adev = ctx->driver_context; enum dc_irq_source irq_source; bool ret; irq_source = DC_IRQ_SOURCE_DMCUB_OUTBOX; - ret = dc_interrupt_set(ctx->dc, irq_source, enable); + ret = amdgpu_dm_irq_set(adev, irq_source, enable); DRM_DEBUG_DRIVER("Dmub trace irq %sabling: r=%d\n", enable ? "en" : "dis", ret); diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.c index d0239a3de2e1..74a8735168aa 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.c @@ -396,6 +396,7 @@ int amdgpu_dm_irq_init(struct amdgpu_device *adev) DRM_DEBUG_KMS("DM_IRQ\n"); spin_lock_init(&adev->dm.irq_handler_list_table_lock); + spin_lock_init(&adev->dm.irq_reg_lock); adev->dm.irq_wq = alloc_workqueue("amdgpu_dm_irq", WQ_UNBOUND | WQ_HIGHPRI, 0); @@ -530,7 +531,7 @@ void amdgpu_dm_irq_suspend(struct amdgpu_device *adev) */ for (src = DC_IRQ_SOURCE_HPD1; src <= DC_IRQ_SOURCE_HPD6RX; src++) { hnd_list_l = &adev->dm.irq_handler_list_low_tab[src]; - dc_interrupt_set(adev->dm.dc, src, false); + amdgpu_dm_irq_set(adev, src, false); DM_IRQ_TABLE_UNLOCK(adev, irq_table_flags); @@ -568,7 +569,7 @@ void amdgpu_dm_irq_resume_early(struct amdgpu_device *adev) hnd_list_l = &adev->dm.irq_handler_list_low_tab[src]; hnd_list_h = &adev->dm.irq_handler_list_high_tab[src]; if (!list_empty(hnd_list_l) || !list_empty(hnd_list_h)) - dc_interrupt_set(adev->dm.dc, src, true); + amdgpu_dm_irq_set(adev, src, true); } DM_IRQ_TABLE_UNLOCK(adev, irq_table_flags); @@ -594,7 +595,7 @@ void amdgpu_dm_irq_resume_late(struct amdgpu_device *adev) hnd_list_l = &adev->dm.irq_handler_list_low_tab[src]; hnd_list_h = &adev->dm.irq_handler_list_high_tab[src]; if (!list_empty(hnd_list_l) || !list_empty(hnd_list_h)) - dc_interrupt_set(adev->dm.dc, src, true); + amdgpu_dm_irq_set(adev, src, true); } DM_IRQ_TABLE_UNLOCK(adev, irq_table_flags); @@ -690,6 +691,23 @@ STATIC_IFN_KUNIT void amdgpu_dm_irq_immediate_work(struct amdgpu_device *adev, } EXPORT_IF_KUNIT(amdgpu_dm_irq_immediate_work); +bool amdgpu_dm_irq_set(struct amdgpu_device *adev, enum dc_irq_source src, + bool enable) +{ + guard(spinlock_irqsave)(&adev->dm.irq_reg_lock); + + return dc_interrupt_set(adev->dm.dc, src, enable); +} +EXPORT_IF_KUNIT(amdgpu_dm_irq_set); + +void amdgpu_dm_irq_ack(struct amdgpu_device *adev, enum dc_irq_source src) +{ + guard(spinlock_irqsave)(&adev->dm.irq_reg_lock); + + dc_interrupt_ack(adev->dm.dc, src); +} +EXPORT_IF_KUNIT(amdgpu_dm_irq_ack); + /** * amdgpu_dm_irq_handler - Generic DM IRQ handler * @adev: amdgpu base driver device containing the DM device @@ -710,7 +728,7 @@ STATIC_IFN_KUNIT int amdgpu_dm_irq_handler(struct amdgpu_device *adev, entry->src_id, entry->src_data[0]); - dc_interrupt_ack(adev->dm.dc, src); + amdgpu_dm_irq_ack(adev, src); /* Call high irq work immediately */ amdgpu_dm_irq_immediate_work(adev, src); @@ -750,7 +768,7 @@ STATIC_IFN_KUNIT int amdgpu_dm_set_hpd_irq_state(struct amdgpu_device *adev, enum dc_irq_source src = amdgpu_dm_hpd_to_dal_irq_source(type); bool st = (state == AMDGPU_IRQ_STATE_ENABLE); - dc_interrupt_set(adev->dm.dc, src, st); + amdgpu_dm_irq_set(adev, src, st); return 0; } EXPORT_IF_KUNIT(amdgpu_dm_set_hpd_irq_state); @@ -785,7 +803,7 @@ static inline int dm_irq_state(struct amdgpu_device *adev, if (dc && dc->caps.ips_support && dc->idle_optimizations_allowed) dc_allow_idle_optimizations(dc, false); - dc_interrupt_set(adev->dm.dc, irq_source, st); + amdgpu_dm_irq_set(adev, irq_source, st); return 0; } @@ -842,7 +860,7 @@ STATIC_IFN_KUNIT int amdgpu_dm_set_dmub_outbox_irq_state(struct amdgpu_device *a enum dc_irq_source irq_source = DC_IRQ_SOURCE_DMCUB_OUTBOX; bool st = (state == AMDGPU_IRQ_STATE_ENABLE); - dc_interrupt_set(adev->dm.dc, irq_source, st); + amdgpu_dm_irq_set(adev, irq_source, st); return 0; } EXPORT_IF_KUNIT(amdgpu_dm_set_dmub_outbox_irq_state); @@ -870,7 +888,7 @@ STATIC_IFN_KUNIT int amdgpu_dm_set_dmub_trace_irq_state(struct amdgpu_device *ad enum dc_irq_source irq_source = DC_IRQ_SOURCE_DMCUB_OUTBOX0; bool st = (state == AMDGPU_IRQ_STATE_ENABLE); - dc_interrupt_set(adev->dm.dc, irq_source, st); + amdgpu_dm_irq_set(adev, irq_source, st); return 0; } EXPORT_IF_KUNIT(amdgpu_dm_set_dmub_trace_irq_state); @@ -937,9 +955,7 @@ EXPORT_IF_KUNIT(amdgpu_dm_set_irq_funcs); void amdgpu_dm_outbox_init(struct amdgpu_device *adev) { - dc_interrupt_set(adev->dm.dc, - DC_IRQ_SOURCE_DMCUB_OUTBOX, - true); + amdgpu_dm_irq_set(adev, DC_IRQ_SOURCE_DMCUB_OUTBOX, true); } EXPORT_IF_KUNIT(amdgpu_dm_outbox_init); @@ -962,7 +978,7 @@ void amdgpu_dm_hpd_init(struct amdgpu_device *adev) /* First, clear all hpd and hpdrx interrupts */ for (i = DC_IRQ_SOURCE_HPD1; i <= DC_IRQ_SOURCE_HPD6RX; i++) { - if (!dc_interrupt_set(adev->dm.dc, i, false)) + if (!amdgpu_dm_irq_set(adev, i, false)) drm_err(dev, "Failed to clear hpd(rx) source=%d on init\n", i); } @@ -991,7 +1007,7 @@ void amdgpu_dm_hpd_init(struct amdgpu_device *adev) * of dm. Note that only hpd interrupt types are registered with * base driver; hpd_rx types aren't. IOW, amdgpu_irq_get/put on * hpd_rx isn't available. DM currently controls hpd_rx - * explicitly with dc_interrupt_set() + * explicitly with amdgpu_dm_irq_set() */ if (dc_link->irq_source_hpd != DC_IRQ_SOURCE_INVALID) { irq_type = dc_link->irq_source_hpd - DC_IRQ_SOURCE_HPD1; @@ -1000,23 +1016,21 @@ void amdgpu_dm_hpd_init(struct amdgpu_device *adev) * and what bios reports as the # of connectors with hpd * sources. Since the # of hpd source types registered * with base driver == mode_info.num_hpd, we have to - * fallback to dc_interrupt_set for the remaining types. + * fallback to amdgpu_dm_irq_set for the remaining types. */ if (irq_type < adev->mode_info.num_hpd) { if (amdgpu_irq_get(adev, &adev->hpd_irq, irq_type)) drm_err(dev, "DM_IRQ: Failed get HPD for source=%d)!\n", dc_link->irq_source_hpd); } else { - dc_interrupt_set(adev->dm.dc, - dc_link->irq_source_hpd, - true); + amdgpu_dm_irq_set(adev, dc_link->irq_source_hpd, + true); } } if (dc_link->irq_source_hpd_rx != DC_IRQ_SOURCE_INVALID) { - dc_interrupt_set(adev->dm.dc, - dc_link->irq_source_hpd_rx, - true); + amdgpu_dm_irq_set(adev, dc_link->irq_source_hpd_rx, + true); } } drm_connector_list_iter_end(&iter); @@ -1061,16 +1075,14 @@ void amdgpu_dm_hpd_fini(struct amdgpu_device *adev) drm_err(dev, "DM_IRQ: Failed put HPD for source=%d!\n", dc_link->irq_source_hpd); } else { - dc_interrupt_set(adev->dm.dc, - dc_link->irq_source_hpd, - false); + amdgpu_dm_irq_set(adev, dc_link->irq_source_hpd, + false); } } if (dc_link->irq_source_hpd_rx != DC_IRQ_SOURCE_INVALID) { - dc_interrupt_set(adev->dm.dc, - dc_link->irq_source_hpd_rx, - false); + amdgpu_dm_irq_set(adev, dc_link->irq_source_hpd_rx, + false); } } drm_connector_list_iter_end(&iter); diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.h b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.h index 4c200a9614a7..bc16ecc67329 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.h +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.h @@ -89,6 +89,34 @@ void amdgpu_dm_irq_unregister_interrupt(struct amdgpu_device *adev, enum dc_irq_source irq_source, void *ih_index); +/** + * amdgpu_dm_irq_set - enable or disable a DC interrupt source. + * + * @adev: AMD DRM device + * @src: DC interrupt source to toggle + * @enable: true to enable the source, false to disable it + * + * DM-wide replacement for dc_interrupt_set(). As locking is DM's + * responsibility, this is a thin wrapper serializes the underlying + * read-modify-write against the other interrupt sources sharing HW control + * registers with @src, so DM must never call dc_interrupt_set() directly. + * + * Returns: true if the source was toggled. + */ +bool amdgpu_dm_irq_set(struct amdgpu_device *adev, enum dc_irq_source src, + bool enable); + +/** + * amdgpu_dm_irq_ack - acknowledge a DC interrupt source. + * + * @adev: AMD DRM device + * @src: DC interrupt source to acknowledge + * + * DM-wide replacement for dc_interrupt_ack(), serialized the same way as + * amdgpu_dm_irq_set(). + */ +void amdgpu_dm_irq_ack(struct amdgpu_device *adev, enum dc_irq_source src); + void amdgpu_dm_set_irq_funcs(struct amdgpu_device *adev); void amdgpu_dm_outbox_init(struct amdgpu_device *adev); diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_crtc_test.c b/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_crtc_test.c index 0d998f204250..ae0f4da96252 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_crtc_test.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_crtc_test.c @@ -436,7 +436,7 @@ static void dm_test_crtc_set_vupdate_irq_no_otg(struct kunit *test) * dm_test_crtc_set_vupdate_irq_dc_busy - Test vupdate irq when DC rejects request * @test: The KUnit test context * - * With an OTG instance assigned but no DC attached, dc_interrupt_set() returns + * With an OTG instance assigned but no DC attached, amdgpu_dm_irq_set() returns * false and the function must report the request as busy (-EBUSY). */ static void dm_test_crtc_set_vupdate_irq_dc_busy(struct kunit *test) @@ -453,12 +453,12 @@ static void dm_test_crtc_set_vupdate_irq_dc_busy(struct kunit *test) acrtc->base.dev = &adev->ddev; acrtc->otg_inst = 0; - /* adev->dm.dc is NULL, so dc_interrupt_set() returns false. */ + /* adev->dm.dc is NULL, so amdgpu_dm_irq_set() returns false. */ KUNIT_EXPECT_EQ(test, amdgpu_dm_crtc_set_vupdate_irq(&acrtc->base, true), -EBUSY); } -/* Per-source funcs let dc_interrupt_set() succeed without register access. */ +/* Per-source funcs let amdgpu_dm_irq_set() succeed without register access. */ static bool dm_test_vupdate_irq_src_set(struct irq_service *irq_service, const struct irq_source_info *info, bool enable) @@ -477,7 +477,7 @@ static struct irq_source_info_funcs dm_test_vupdate_irq_src_funcs = { .ack = dm_test_vupdate_irq_src_ack, }; -/* A .set that fails so dc_interrupt_set() reports the source as busy. */ +/* A .set that fails so amdgpu_dm_irq_set() reports the source as busy. */ static bool dm_test_vupdate_irq_src_set_busy(struct irq_service *irq_service, const struct irq_source_info *info, bool enable) @@ -519,7 +519,9 @@ static void dm_test_crtc_set_vupdate_irq_enable(struct kunit *test) irqs = kunit_kzalloc(test, sizeof(*irqs), GFP_KERNEL); KUNIT_ASSERT_NOT_ERR_OR_NULL(test, irqs); - /* Populate the per-source info table so dc_interrupt_set() succeeds. */ + /* + * Populate the per-source info table so amdgpu_dm_irq_set() succeeds. + */ info = kunit_kzalloc(test, sizeof(*info) * DAL_IRQ_SOURCES_NUMBER, GFP_KERNEL); KUNIT_ASSERT_NOT_ERR_OR_NULL(test, info); @@ -1018,7 +1020,9 @@ static void dm_test_crtc_enable_vblank_vupdate_busy(struct kunit *test) irqs = kunit_kzalloc(test, sizeof(*irqs), GFP_KERNEL); KUNIT_ASSERT_NOT_ERR_OR_NULL(test, irqs); - /* Per-source .set fails so dc_interrupt_set() reports the source busy. */ + /* + * Per-source .set fails so amdgpu_dm_irq_set() reports the source busy. + */ info = kunit_kzalloc(test, sizeof(*info) * DAL_IRQ_SOURCES_NUMBER, GFP_KERNEL); KUNIT_ASSERT_NOT_ERR_OR_NULL(test, info); diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_helpers_test.c b/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_helpers_test.c index 82e0c984693c..639512bea275 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_helpers_test.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_helpers_test.c @@ -2442,17 +2442,21 @@ static void dm_test_is_dp_sink_present_null_priv(struct kunit *test) * dm_test_dmub_outbox_interrupt_control_null_dc - Test outbox irq control with NULL dc * @test: The KUnit test context * - * dc_interrupt_set() is NULL-safe and returns false when dc is NULL, so the + * amdgpu_dm_irq_set() is NULL-safe and returns false when dc is NULL, so the * helper returns false without touching real interrupt hardware. */ static void dm_test_dmub_outbox_interrupt_control_null_dc(struct kunit *test) { + struct amdgpu_device *adev; struct dc_context *ctx; + adev = kunit_kzalloc(test, sizeof(*adev), GFP_KERNEL); + KUNIT_ASSERT_NOT_NULL(test, adev); ctx = kunit_kzalloc(test, sizeof(*ctx), GFP_KERNEL); KUNIT_ASSERT_NOT_NULL(test, ctx); + ctx->driver_context = adev; - /* ctx->dc is NULL → dc_interrupt_set returns false */ + /* adev->dm.dc is NULL → amdgpu_dm_irq_set returns false */ KUNIT_EXPECT_FALSE(test, dm_helpers_dmub_outbox_interrupt_control(ctx, true)); KUNIT_EXPECT_FALSE(test, dm_helpers_dmub_outbox_interrupt_control(ctx, false)); } diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_irq_test.c b/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_irq_test.c index 861ee9eaa032..95322d8c8613 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_irq_test.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/tests/amdgpu_dm_irq_test.c @@ -268,7 +268,7 @@ static bool dm_test_irq_src_ack(struct irq_service *irq_service, return true; } -/* Per-source funcs let dc_interrupt_set() succeed without register access. */ +/* Per-source funcs let amdgpu_dm_irq_set() succeed without register access. */ static struct irq_source_info_funcs dm_test_irq_src_funcs = { .set = dm_test_irq_src_set, .ack = dm_test_irq_src_ack, @@ -290,7 +290,7 @@ static struct dc *dm_test_alloc_dc_with_irq_service(struct kunit *test, KUNIT_ASSERT_NOT_ERR_OR_NULL(test, irqs); /* - * Populate the per-source info table so dc_interrupt_set()/_ack() + * Populate the per-source info table so amdgpu_dm_irq_set()/_ack() * succeed without touching hardware registers. */ info = kunit_kzalloc(test, sizeof(*info) * DAL_IRQ_SOURCES_NUMBER, @@ -1227,7 +1227,7 @@ static void dm_test_irq_suspend_empty(struct kunit *test) KUNIT_ASSERT_EQ(test, amdgpu_dm_irq_init(adev), 0); /* - * With no registered handlers the HW dc_interrupt_set() calls are + * With no registered handlers the amdgpu_dm_irq_set() calls are * skipped, so suspend must complete without touching the (absent) DC. */ amdgpu_dm_irq_suspend(adev); @@ -1275,11 +1275,11 @@ static void dm_test_irq_resume_late_empty(struct kunit *test) } /** - * dm_test_irq_suspend_registered - Test suspend reaches the dc_interrupt_set path + * dm_test_irq_suspend_registered - Test suspend reaches the irq set path * @test: The KUnit test context * * Registers a low-context HPD handler so the handler list is non-empty, - * forcing amdgpu_dm_irq_suspend() to call dc_interrupt_set() (NULL-safe with + * forcing amdgpu_dm_irq_suspend() to call amdgpu_dm_irq_set() (NULL-safe with * no DC) and flush_work() on the registered handler. */ static void dm_test_irq_suspend_registered(struct kunit *test) @@ -1330,11 +1330,11 @@ static void dm_test_irq_suspend_disables_polling(struct kunit *test) } /** - * dm_test_irq_resume_early_registered - Test early resume reaches dc_interrupt_set + * dm_test_irq_resume_early_registered - Test early resume reaches irq set * @test: The KUnit test context * * Registers a low-context HPD RX handler so early resume calls - * dc_interrupt_set() for the short-pulse interrupt source. + * amdgpu_dm_irq_set() for the short-pulse interrupt source. */ static void dm_test_irq_resume_early_registered(struct kunit *test) { @@ -1358,10 +1358,10 @@ static void dm_test_irq_resume_early_registered(struct kunit *test) } /** - * dm_test_irq_resume_late_registered - Test late resume reaches dc_interrupt_set + * dm_test_irq_resume_late_registered - Test late resume reaches irq set * @test: The KUnit test context * - * Registers a low-context HPD handler so late resume calls dc_interrupt_set() + * Registers a low-context HPD handler so late resume calls amdgpu_dm_irq_set() * for the HPD interrupt source. */ static void dm_test_irq_resume_late_registered(struct kunit *test) @@ -1592,7 +1592,7 @@ static void dm_test_set_crtc_irq_state_enable(struct kunit *test) /* * otg_inst >= 0 computes the irq source and reaches the NULL-safe - * dc_interrupt_set(); the ips_support branch is skipped (dc == NULL). + * amdgpu_dm_irq_set(); the ips_support branch is skipped (dc == NULL). */ acrtc->otg_inst = 3; adev->mode_info.crtcs[0] = acrtc; @@ -1671,8 +1671,8 @@ static void dm_test_set_vupdate_irq_state_enable(struct kunit *test) * * With a non-NULL DC that advertises IPS support and currently allows idle * optimizations, dm_irq_state() must call dc_allow_idle_optimizations() before - * dc_interrupt_set(). disable_idle_power_optimizations makes that call a safe - * early return, and per-source stub funcs let dc_interrupt_set() succeed. + * amdgpu_dm_irq_set(). disable_idle_power_optimizations makes that call a safe + * early return, and per-source stub funcs let amdgpu_dm_irq_set() succeed. */ static void dm_test_set_crtc_irq_state_allows_idle(struct kunit *test) { @@ -1891,7 +1891,7 @@ static void dm_test_set_hpd_irq_state_null_dc(struct kunit *test) adev = kunit_kzalloc(test, sizeof(*adev), GFP_KERNEL); KUNIT_ASSERT_NOT_ERR_OR_NULL(test, adev); - /* dc_interrupt_set() is a no-op when dc is NULL, so both states + /* amdgpu_dm_irq_set() is a no-op when dc is NULL, so both states * return 0 without dereferencing the (absent) DC. */ KUNIT_EXPECT_EQ(test, amdgpu_dm_set_hpd_irq_state(adev, NULL, AMDGPU_HPD_1, @@ -1951,7 +1951,7 @@ static void dm_test_outbox_init_null_dc(struct kunit *test) adev = kunit_kzalloc(test, sizeof(*adev), GFP_KERNEL); KUNIT_ASSERT_NOT_ERR_OR_NULL(test, adev); - /* Single dc_interrupt_set() call must be skipped when dc is NULL. */ + /* Single amdgpu_dm_irq_set() call must be skipped when dc is NULL. */ amdgpu_dm_outbox_init(adev); } @@ -1969,7 +1969,7 @@ static void dm_test_hpd_init_empty_connectors(struct kunit *test) /* * With an empty connector list the per-connector loop is skipped and - * the initial clear loop relies on dc_interrupt_set() being a no-op + * the initial clear loop relies on amdgpu_dm_irq_set() being a no-op * for a NULL dc, so init must complete without touching the DC. */ amdgpu_dm_hpd_init(adev); @@ -2004,7 +2004,7 @@ static void dm_test_hpd_init_fini_with_connectors(struct kunit *test) /* * num_hpd = 0 forces irq_type >= num_hpd so the loop takes the HW - * fallback (dc_interrupt_set()) instead of amdgpu_irq_get(); with a + * fallback (amdgpu_dm_irq_set()) instead of amdgpu_irq_get(); with a * NULL dc that fallback is a safe no-op. */ adev->mode_info.num_hpd = 0; @@ -2090,7 +2090,7 @@ static void dm_test_hpd_init_fini_irq_ref(struct kunit *test) /* * num_hpd >= 1 makes irq_type (0) < num_hpd, so the loop takes the * amdgpu_irq_get()/amdgpu_irq_put() branch instead of the - * dc_interrupt_set() fallback. The mock device has irq.installed == + * amdgpu_dm_irq_set() fallback. The mock device has irq.installed == * false, so both calls fail early with -ENOENT (logging an error) * without touching the base-driver irq state. */ diff --git a/drivers/gpu/drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c b/drivers/gpu/drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c index cb163902e12e..d669b47af120 100644 --- a/drivers/gpu/drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c +++ b/drivers/gpu/drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c @@ -1064,10 +1064,12 @@ bool dcn30_apply_idle_power_optimizations(struct dc *dc, bool enable) */ unsigned int denom = refresh_hz * 6528; unsigned int stutter_period = dc->current_state->perf_params.stutter_period_us; + uint64_t num = (1000000LL + 2 * stutter_period * refresh_hz) * + (100LL + dc->debug.mall_additional_timer_percent); + uint64_t tmr_ticks; - tmr_delay = (uint32_t)(div_u64(((1000000LL + 2 * stutter_period * refresh_hz) * - (100LL + dc->debug.mall_additional_timer_percent) + denom - 1), - denom) - 64LL); + tmr_ticks = div_u64(num + denom - 1, denom); + tmr_delay = tmr_ticks > 64 ? (uint32_t)(tmr_ticks - 64) : 0; /* In some cases the stutter period is really big (tiny modes) in these * cases MALL cant be enabled, So skip these cases to avoid a ASSERT() @@ -1089,9 +1091,8 @@ bool dcn30_apply_idle_power_optimizations(struct dc *dc, bool enable) } denom *= 2; - tmr_delay = (uint32_t)(div_u64(((1000000LL + 2 * stutter_period * refresh_hz) * - (100LL + dc->debug.mall_additional_timer_percent) + denom - 1), - denom) - 64LL); + tmr_ticks = div_u64(num + denom - 1, denom); + tmr_delay = tmr_ticks > 64 ? (uint32_t)(tmr_ticks - 64) : 0; } /* Copy HW cursor */ diff --git a/drivers/gpu/drm/amd/display/dc/hwss/dcn50/dcn50_hwseq.c b/drivers/gpu/drm/amd/display/dc/hwss/dcn50/dcn50_hwseq.c index a7f8fd03faea..e549556b9679 100644 --- a/drivers/gpu/drm/amd/display/dc/hwss/dcn50/dcn50_hwseq.c +++ b/drivers/gpu/drm/amd/display/dc/hwss/dcn50/dcn50_hwseq.c @@ -67,7 +67,8 @@ static void dcn50_initialize_min_clocks(struct dc *dc) * audio corruption. Read current DISPCLK from DENTIST and request the same * freq to ensure that the timing is valid and unchanged. */ - clocks->dispclk_khz = dc->clk_mgr->funcs->get_dispclk_from_dentist(dc->clk_mgr); + if (dc->clk_mgr->funcs->get_dispclk_from_dentist) + clocks->dispclk_khz = dc->clk_mgr->funcs->get_dispclk_from_dentist(dc->clk_mgr); } clocks->ref_dtbclk_khz = dc->clk_mgr->bw_params->clk_table.entries[0].dtbclk_mhz * 1000; clocks->fclk_p_state_change_support = true; @@ -639,7 +640,8 @@ void dcn50_init_hw(struct dc *dc) dc->res_pool->hubbub->funcs->allow_self_refresh_control(dc->res_pool->hubbub, !dc->res_pool->hubbub->ctx->dc->debug.disable_stutter); - dcn50_initialize_min_clocks(dc); + if (dc->clk_mgr && dc->clk_mgr->funcs) + dcn50_initialize_min_clocks(dc); /* On HW init, allow idle optimizations after pipes have been turned off. * diff --git a/drivers/gpu/drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.c b/drivers/gpu/drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.c index 72c2d3ca52f6..61ad6efa7a74 100644 --- a/drivers/gpu/drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.c +++ b/drivers/gpu/drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.c @@ -643,7 +643,8 @@ void dcn60_init_hw(struct dc *dc) dc->res_pool->hubbub->funcs->allow_self_refresh_control(dc->res_pool->hubbub, !dc->res_pool->hubbub->ctx->dc->debug.disable_stutter); - dcn401_initialize_min_clocks(dc); + if (dc->clk_mgr && dc->clk_mgr->funcs) + dcn401_initialize_min_clocks(dc); /* On HW init, allow idle optimizations after pipes have been turned off. * @@ -1001,6 +1002,7 @@ static void dcn60_build_hubbub_perfmon_sequence( /** * dcn60_update_probe_status - Set the valid flag on a latched probe result. * @status: result sink whose u was written by the GET BLS step during execute + * @probe: current probe state used to determine measurement type and validity */ static void dcn60_update_probe_status(struct dc_probe_status *status) { @@ -1024,6 +1026,7 @@ static void dcn60_update_probe_status(struct dc_probe_status *status) /** * is_probe_measurement_type_for_hubbub - Returns true if the probe type is * served by the hubbub perfmon block on DCN60. + * @type: the probe measurement type to classify */ static bool is_probe_measurement_type_for_hubbub(enum dc_probe_type type) { diff --git a/drivers/gpu/drm/amd/pm/swsmu/smu14/smu_v14_0_2_ppt.c b/drivers/gpu/drm/amd/pm/swsmu/smu14/smu_v14_0_2_ppt.c index 56a5c11bc196..ed99e61f18e1 100644 --- a/drivers/gpu/drm/amd/pm/swsmu/smu14/smu_v14_0_2_ppt.c +++ b/drivers/gpu/drm/amd/pm/swsmu/smu14/smu_v14_0_2_ppt.c @@ -2142,6 +2142,7 @@ static void smu_v14_0_2_init_msg_ctl(struct smu_context *smu) static ssize_t smu_v14_0_2_get_gpu_metrics(struct smu_context *smu, void **table) { + uint32_t mp1_ver = amdgpu_ip_version(smu->adev, MP1_HWIP, 0); struct gpu_metrics_v1_3 *gpu_metrics = (struct gpu_metrics_v1_3 *)smu_driver_table_ptr( smu, SMU_DRIVER_TABLE_GPU_METRICS); @@ -2171,6 +2172,8 @@ static ssize_t smu_v14_0_2_get_gpu_metrics(struct smu_context *smu, metrics->Vcn1ActivityPercentage); gpu_metrics->average_socket_power = metrics->AverageSocketPower; + if (mp1_ver == IP_VERSION(14, 0, 3) && smu->smc_fw_version >= 0x00685000) + gpu_metrics->energy_accumulator = metrics->EnergyAccumulator; if (metrics->AverageGfxActivity <= SMU_14_0_2_BUSY_THRESHOLD) gpu_metrics->average_gfxclk_frequency = metrics->AverageGfxclkFrequencyPostDs; diff --git a/drivers/gpu/drm/ci/xfails/msm-apq8016-fails.txt b/drivers/gpu/drm/ci/xfails/msm-apq8016-fails.txt index 4546363447ff..0f3d85e4845e 100644 --- a/drivers/gpu/drm/ci/xfails/msm-apq8016-fails.txt +++ b/drivers/gpu/drm/ci/xfails/msm-apq8016-fails.txt @@ -7,3 +7,10 @@ kms_hdmi_inject@inject-4k,Fail kms_lease@lease-uevent,Fail msm/msm_mapping@memptrs,Fail msm/msm_mapping@ring,Fail + +# Started failing with v7.3-rc2 backmerge +# https://gitlab.freedesktop.org/drm/msm/-/work_items/104 +kms_cursor_legacy@single-move,Fail +kms_cursor_legacy@torture-bo,Fail +kms_cursor_legacy@forked-bo,Fail +kms_cursor_legacy@torture-move,Fail diff --git a/drivers/gpu/drm/drm_atomic_uapi.c b/drivers/gpu/drm/drm_atomic_uapi.c index 5ea593b3a98e..68d59e17ffc2 100644 --- a/drivers/gpu/drm/drm_atomic_uapi.c +++ b/drivers/gpu/drm/drm_atomic_uapi.c @@ -1462,10 +1462,12 @@ static int prepare_signaling(struct drm_device *dev, struct dma_fence *fence; struct drm_out_fence_state *f; + ret = -ENOMEM; + f = krealloc(*fence_state, sizeof(**fence_state) * (*num_fences + 1), GFP_KERNEL); if (!f) - return -ENOMEM; + goto err_free_event; memset(&f[*num_fences], 0, sizeof(*f)); @@ -1474,12 +1476,12 @@ static int prepare_signaling(struct drm_device *dev, fence = drm_crtc_create_fence(crtc); if (!fence) - return -ENOMEM; + goto err_free_event; ret = setup_out_fence(&f[(*num_fences)++], fence); if (ret) { dma_fence_put(fence); - return ret; + goto err_free_event; } crtc_state->event->base.fence = fence; @@ -1535,6 +1537,11 @@ static int prepare_signaling(struct drm_device *dev, } return 0; + +err_free_event: + drm_event_cancel_free(dev, &crtc_state->event->base); + crtc_state->event = NULL; + return ret; } static void complete_signaling(struct drm_device *dev, diff --git a/drivers/gpu/drm/gud/gud_pipe.c b/drivers/gpu/drm/gud/gud_pipe.c index 5ef887d8485a..aa7792966287 100644 --- a/drivers/gpu/drm/gud/gud_pipe.c +++ b/drivers/gpu/drm/gud/gud_pipe.c @@ -482,6 +482,9 @@ int gud_plane_atomic_check(struct drm_plane *plane, if (!new_plane_state->visible) return 0; + if (gdrm->flags & GUD_DISPLAY_FLAG_FULL_UPDATE) + new_plane_state->ignore_damage_clips = true; + if (old_plane_state->rotation != new_plane_state->rotation) crtc_state->mode_changed = true; @@ -562,8 +565,8 @@ int gud_plane_atomic_check(struct drm_plane *plane, goto out; } - req->properties[num_properties + i].prop = cpu_to_le16(prop); - req->properties[num_properties + i].val = cpu_to_le64(val); + req->properties[num_properties].prop = cpu_to_le16(prop); + req->properties[num_properties].val = cpu_to_le64(val); num_properties++; } diff --git a/drivers/gpu/drm/i915/display/intel_cursor.c b/drivers/gpu/drm/i915/display/intel_cursor.c index 86bb96ac449b..0673f16f6fd0 100644 --- a/drivers/gpu/drm/i915/display/intel_cursor.c +++ b/drivers/gpu/drm/i915/display/intel_cursor.c @@ -530,18 +530,13 @@ static int i9xx_check_cursor(struct intel_crtc_state *crtc_state, } static void i9xx_cursor_disable_sel_fetch_arm(struct intel_dsb *dsb, - struct intel_plane *plane) + struct intel_plane *plane, + const struct intel_crtc_state *crtc_state) { struct intel_display *display = to_intel_display(plane); enum pipe pipe = plane->pipe; - /* - * Clear this whenever the hardware has selective fetch, not just when - * the current state uses it. The cursor may have been enabled with - * selective fetch earlier and had its enable bit orphaned when the - * feature was switched off. - */ - if (!HAS_PSR2_SEL_FETCH(display)) + if (!crtc_state->enable_psr2_sel_fetch) return; intel_de_write_dsb(display, dsb, SEL_FETCH_CUR_CTL(pipe), 0); @@ -591,7 +586,7 @@ static void i9xx_cursor_update_sel_fetch_arm(struct intel_dsb *dsb, if (crtc_state->enable_psr2_su_region_et) wa_16021440873(dsb, plane, crtc_state, plane_state); else - i9xx_cursor_disable_sel_fetch_arm(dsb, plane); + i9xx_cursor_disable_sel_fetch_arm(dsb, plane, crtc_state); } } @@ -700,7 +695,7 @@ static void i9xx_cursor_update_arm(struct intel_dsb *dsb, if (plane_state) i9xx_cursor_update_sel_fetch_arm(dsb, plane, crtc_state, plane_state); else - i9xx_cursor_disable_sel_fetch_arm(dsb, plane); + i9xx_cursor_disable_sel_fetch_arm(dsb, plane, crtc_state); if (plane->cursor.base != base || plane->cursor.size != fbc_ctl || diff --git a/drivers/gpu/drm/i915/display/intel_dp_link_caps.c b/drivers/gpu/drm/i915/display/intel_dp_link_caps.c index 7b6cc6055da8..98657aa4d3d5 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_link_caps.c +++ b/drivers/gpu/drm/i915/display/intel_dp_link_caps.c @@ -426,12 +426,15 @@ calc_allowed_config_filter(struct intel_dp_link_caps *link_caps, const struct intel_dp_link_config *forced_params) { struct intel_dp_link_caps_filter allowed_configs = INTEL_DP_LINK_CAPS_FILTER_NONE; + struct intel_display *display = to_intel_display(link_caps->dp); struct intel_dp_link_caps_order order = bw_desc_config_order(); struct intel_dp_link_caps_iter iter; struct intel_dp_link_config config; iter_start(&iter, link_caps, order, enabled_configs); for_each_dp_link_config(&iter, &config) { + int config_idx; + if (forced_params->rate && forced_params->rate != config.rate) continue; @@ -446,7 +449,11 @@ calc_allowed_config_filter(struct intel_dp_link_caps *link_caps, if (config.lane_count > max_limits->lane_count) continue; - allowed_configs.config_mask |= BIT(iter_pos_to_idx(link_caps, order, iter.pos)); + config_idx = iter_pos_to_idx(link_caps, order, iter.pos); + if (drm_WARN_ON(display->drm, config_idx < 0)) + continue; + + allowed_configs.config_mask |= BIT(config_idx); } intel_dp_link_caps_iter_end(&iter); diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c index 5cda1ab90e40..07a683293352 100644 --- a/drivers/gpu/drm/i915/display/skl_universal_plane.c +++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c @@ -879,18 +879,13 @@ skl_plane_disable_arm(struct intel_dsb *dsb, } static void icl_plane_disable_sel_fetch_arm(struct intel_dsb *dsb, - struct intel_plane *plane) + struct intel_plane *plane, + const struct intel_crtc_state *crtc_state) { struct intel_display *display = to_intel_display(plane); enum pipe pipe = plane->pipe; - /* - * Clear this whenever the hardware has selective fetch, not just when - * the current state uses it. The plane may have been enabled with - * selective fetch earlier and had its enable bit orphaned when the - * feature was switched off. - */ - if (!HAS_PSR2_SEL_FETCH(display)) + if (!crtc_state->enable_psr2_sel_fetch) return; intel_de_write_dsb(display, dsb, SEL_FETCH_PLANE_CTL(pipe, plane->id), 0); @@ -926,7 +921,7 @@ icl_plane_disable_arm(struct intel_dsb *dsb, skl_write_plane_wm(dsb, plane, crtc_state); - icl_plane_disable_sel_fetch_arm(dsb, plane); + icl_plane_disable_sel_fetch_arm(dsb, plane, crtc_state); if (plane_has_normalizer(plane)) intel_de_write_dsb(display, dsb, @@ -1646,7 +1641,7 @@ static void icl_plane_update_sel_fetch_arm(struct intel_dsb *dsb, intel_de_write_dsb(display, dsb, SEL_FETCH_PLANE_CTL(pipe, plane->id), SEL_FETCH_PLANE_CTL_ENABLE); else - icl_plane_disable_sel_fetch_arm(dsb, plane); + icl_plane_disable_sel_fetch_arm(dsb, plane, crtc_state); } static void diff --git a/drivers/gpu/drm/loongson/lsdc_plane.c b/drivers/gpu/drm/loongson/lsdc_plane.c index bea42215796d..bcc0ffa17bdf 100644 --- a/drivers/gpu/drm/loongson/lsdc_plane.c +++ b/drivers/gpu/drm/loongson/lsdc_plane.c @@ -7,6 +7,7 @@ #include <drm/drm_atomic.h> #include <drm/drm_atomic_helper.h> +#include <drm/drm_blend.h> #include <drm/drm_framebuffer.h> #include <drm/drm_gem_atomic_helper.h> #include <drm/drm_print.h> @@ -765,7 +766,7 @@ int ls7a1000_cursor_plane_init(struct drm_device *ddev, drm_plane_helper_add(plane, &ls7a1000_cursor_plane_helper_funcs); - return 0; + return drm_plane_create_blend_mode_property(plane, BIT(DRM_MODE_BLEND_COVERAGE)); } int ls7a2000_cursor_plane_init(struct drm_device *ddev, @@ -790,5 +791,5 @@ int ls7a2000_cursor_plane_init(struct drm_device *ddev, drm_plane_helper_add(plane, &ls7a2000_cursor_plane_helper_funcs); - return 0; + return drm_plane_create_blend_mode_property(plane, BIT(DRM_MODE_BLEND_COVERAGE)); } diff --git a/drivers/gpu/drm/msm/Makefile b/drivers/gpu/drm/msm/Makefile index d0c3a4c6703b..0b8f2cafed19 100644 --- a/drivers/gpu/drm/msm/Makefile +++ b/drivers/gpu/drm/msm/Makefile @@ -177,7 +177,6 @@ quiet_cmd_headergen = GENHDR $@ cmd_headergen = mkdir -p $(obj)/generated && $(PYTHON3) $(src)/registers/gen_header.py \ $(headergen-opts) --rnn $(src)/registers --xml $< c-defines > $@ -# TODO how to do this for a2xx/a5xx which have different .xml arg? quiet_cmd_headergen_json = GENHDRJSN $@ cmd_headergen_json = mkdir -p $(obj)/generated && $(PYTHON3) $(src)/registers/gen_header.py \ $(headergen-opts) --rnn $(src)/registers --xml $(filter %.xml,$^) perfcntrs --json $< > $@ diff --git a/drivers/gpu/drm/msm/adreno/a6xx_gmu.c b/drivers/gpu/drm/msm/adreno/a6xx_gmu.c index 27cac853975f..6d49c51df1a2 100644 --- a/drivers/gpu/drm/msm/adreno/a6xx_gmu.c +++ b/drivers/gpu/drm/msm/adreno/a6xx_gmu.c @@ -320,7 +320,7 @@ static int a6xx_gmu_start(struct a6xx_gmu *gmu) gmu_write(gmu, REG_A6XX_GMU_CM3_SYSRESET, 0); ret = gmu_poll_timeout(gmu, REG_A6XX_GMU_CM3_FW_INIT_RESULT, val, - (val & mask) == reset_val, 100, 10000); + (val & mask) == reset_val, 100, 100000); if (ret) DRM_DEV_ERROR(gmu->dev, "GMU firmware initialization timed out\n"); diff --git a/drivers/gpu/drm/msm/adreno/a6xx_gpu.c b/drivers/gpu/drm/msm/adreno/a6xx_gpu.c index f9de9329dee3..081e79ea4652 100644 --- a/drivers/gpu/drm/msm/adreno/a6xx_gpu.c +++ b/drivers/gpu/drm/msm/adreno/a6xx_gpu.c @@ -23,9 +23,15 @@ static u64 a6xx_gmu_get_timestamp(struct msm_gpu *gpu) u64 count_hi, count_lo, temp; do { - count_hi = gmu_read(&a6xx_gpu->gmu, REG_A6XX_GMU_ALWAYS_ON_COUNTER_H); - count_lo = gmu_read(&a6xx_gpu->gmu, REG_A6XX_GMU_ALWAYS_ON_COUNTER_L); - temp = gmu_read(&a6xx_gpu->gmu, REG_A6XX_GMU_ALWAYS_ON_COUNTER_H); + if (adreno_is_a750_family(adreno_gpu)) { + count_hi = gmu_read(&a6xx_gpu->gmu, REG_A7XX_GMU_CX_AO_COUNTER_H); + count_lo = gmu_read(&a6xx_gpu->gmu, REG_A7XX_GMU_CX_AO_COUNTER_L); + temp = gmu_read(&a6xx_gpu->gmu, REG_A7XX_GMU_CX_AO_COUNTER_H); + } else { + count_hi = gmu_read(&a6xx_gpu->gmu, REG_A6XX_GMU_ALWAYS_ON_COUNTER_H); + count_lo = gmu_read(&a6xx_gpu->gmu, REG_A6XX_GMU_ALWAYS_ON_COUNTER_L); + temp = gmu_read(&a6xx_gpu->gmu, REG_A6XX_GMU_ALWAYS_ON_COUNTER_H); + } } while (unlikely(count_hi != temp)); return (count_hi << 32) | count_lo; diff --git a/drivers/gpu/drm/msm/adreno/adreno_device.c b/drivers/gpu/drm/msm/adreno/adreno_device.c index 7f20320ef66a..05c77fe27e62 100644 --- a/drivers/gpu/drm/msm/adreno/adreno_device.c +++ b/drivers/gpu/drm/msm/adreno/adreno_device.c @@ -25,7 +25,7 @@ MODULE_PARM_DESC(disable_acd, "Forcefully disable GPU ACD"); module_param_unsafe(disable_acd, bool, 0400); static bool skip_gpu; -MODULE_PARM_DESC(no_gpu, "Disable GPU driver register (0=enable GPU driver register (default), 1=skip GPU driver register"); +MODULE_PARM_DESC(skip_gpu, "Disable GPU driver register (0=enable GPU driver register (default), 1=skip GPU driver register"); module_param(skip_gpu, bool, 0400); extern const struct adreno_gpulist a2xx_gpulist; diff --git a/drivers/gpu/drm/msm/adreno/adreno_gpu.c b/drivers/gpu/drm/msm/adreno/adreno_gpu.c index 8cd2020d4b7e..5832dc25d6bf 100644 --- a/drivers/gpu/drm/msm/adreno/adreno_gpu.c +++ b/drivers/gpu/drm/msm/adreno/adreno_gpu.c @@ -52,6 +52,12 @@ static int zap_shader_load_mdt(struct msm_gpu *gpu, const char *fwname, return -ENODEV; } + /* We need PAS to be able to load the firmware */ + if (!qcom_pas_is_available()) { + DRM_DEV_ERROR(dev, "PAS is not available\n"); + return -EPROBE_DEFER; + } + ret = of_reserved_mem_region_to_resource(np, 0, &r); if (ret) { zap_available = false; @@ -170,18 +176,11 @@ out: int adreno_zap_shader_load(struct msm_gpu *gpu, u32 pasid) { struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu); - struct platform_device *pdev = gpu->pdev; /* Short cut if we determine the zap shader isn't available/needed */ if (!zap_available) return -ENODEV; - /* We need PAS to be able to load the firmware */ - if (!qcom_pas_is_available()) { - DRM_DEV_ERROR(&pdev->dev, "PAS is not available\n"); - return -EPROBE_DEFER; - } - return zap_shader_load_mdt(gpu, adreno_gpu->info->zapfw, pasid); } @@ -1262,6 +1261,8 @@ void adreno_gpu_cleanup(struct adreno_gpu *adreno_gpu) for (i = 0; i < ARRAY_SIZE(adreno_gpu->info->fw); i++) release_firmware(adreno_gpu->fw[i]); + pm_runtime_dont_use_autosuspend(&gpu->pdev->dev); + if (priv && pm_runtime_enabled(&priv->gpu_pdev->dev)) pm_runtime_disable(&priv->gpu_pdev->dev); diff --git a/drivers/gpu/drm/msm/disp/dpu1/dpu_hw_ctl.c b/drivers/gpu/drm/msm/disp/dpu1/dpu_hw_ctl.c index 36a497f1d6c1..e8729fe1cfef 100644 --- a/drivers/gpu/drm/msm/disp/dpu1/dpu_hw_ctl.c +++ b/drivers/gpu/drm/msm/disp/dpu1/dpu_hw_ctl.c @@ -118,6 +118,7 @@ static inline void dpu_hw_ctl_clear_pending_flush(struct dpu_hw_ctl *ctx) ctx->pending_intf_flush_mask = 0; ctx->pending_wb_flush_mask = 0; ctx->pending_cwb_flush_mask = 0; + ctx->pending_periph_flush_mask = 0; ctx->pending_merge_3d_flush_mask = 0; ctx->pending_dsc_flush_mask = 0; ctx->pending_cdm_flush_mask = 0; diff --git a/drivers/gpu/drm/msm/dp/dp_display.c b/drivers/gpu/drm/msm/dp/dp_display.c index bc646d172abe..4dcbd9b99d06 100644 --- a/drivers/gpu/drm/msm/dp/dp_display.c +++ b/drivers/gpu/drm/msm/dp/dp_display.c @@ -756,6 +756,7 @@ enum drm_mode_status msm_dp_display_mode_valid(struct msm_dp *dp, struct msm_dp_link_info *link_info; u32 mode_rate_khz = 0, supported_rate_khz = 0, mode_bpp = 0; int mode_pclk_khz = mode->clock; + int link_pclk_khz; bool is_yuv_420; if (!dp || !mode_pclk_khz || !dp->connector) { @@ -775,6 +776,8 @@ enum drm_mode_status msm_dp_display_mode_valid(struct msm_dp *dp, if (is_yuv_420 && !msm_dp_display->panel->vsc_sdp_supported) return MODE_NO_420; + link_pclk_khz = is_yuv_420 ? mode_pclk_khz / 2 : mode_pclk_khz; + if (is_yuv_420 || msm_dp_display->wide_bus_supported) mode_pclk_khz /= 2; @@ -786,9 +789,9 @@ enum drm_mode_status msm_dp_display_mode_valid(struct msm_dp *dp, mode_bpp = default_bpp; mode_bpp = msm_dp_panel_get_mode_bpp(msm_dp_display->panel, - mode_bpp, mode_pclk_khz); + mode_bpp, link_pclk_khz); - mode_rate_khz = mode_pclk_khz * mode_bpp; + mode_rate_khz = link_pclk_khz * mode_bpp; supported_rate_khz = link_info->num_lanes * link_info->rate * 8; if (mode_rate_khz > supported_rate_khz) @@ -1458,6 +1461,20 @@ void msm_dp_display_atomic_disable(struct msm_dp *dp) msm_dp_display = container_of(dp, struct msm_dp_display_private, msm_dp_display); + /* + * If .atomic_enable() bailed out - link training failure is the common + * case - the mainlink was never brought up and ->power_on stayed false. + * Driving the PUSH_IDLE pattern into a controller that was never + * enabled times out, and .atomic_post_disable() then drops the + * controller's runtime-PM reference without tearing the PHY back down, + * because msm_dp_display_disable() returns early on !power_on. On + * glymur (Snapdragon X2 Elite) that combination is answered by a + * TrustZone-level SOCCP/ADSP force-stop and a silent SoC reset. + * There is nothing to push idle, so leave it alone. + */ + if (!dp->power_on) + return; + msm_dp_ctrl_push_idle(msm_dp_display->ctrl); } diff --git a/drivers/gpu/drm/msm/dsi/dsi_host.c b/drivers/gpu/drm/msm/dsi/dsi_host.c index 7e4e3718b536..b292dfd266d1 100644 --- a/drivers/gpu/drm/msm/dsi/dsi_host.c +++ b/drivers/gpu/drm/msm/dsi/dsi_host.c @@ -129,7 +129,7 @@ struct msm_dsi_host { struct clk *dsi_pll_pixel_clk; unsigned long byte_clk_rate; - unsigned long byte_intf_clk_rate; + bool byte_intf_clk_div_2; unsigned long pixel_clk_rate; unsigned long esc_clk_rate; @@ -382,8 +382,20 @@ int msm_dsi_runtime_resume(struct device *dev) int dsi_link_clk_set_rate_6g(struct msm_dsi_host *msm_host) { + unsigned long byte_intf_clk_rate; + long rounded_byte_clk_rate; int ret; + rounded_byte_clk_rate = clk_round_rate(msm_host->byte_clk, + msm_host->byte_clk_rate); + if (rounded_byte_clk_rate < 0) { + pr_err("%s: failed to round byte clock rate, %ld\n", + __func__, rounded_byte_clk_rate); + return rounded_byte_clk_rate; + } + + msm_host->byte_clk_rate = rounded_byte_clk_rate; + DBG("Set clk rates: pclk=%lu, byteclk=%lu", msm_host->pixel_clk_rate, msm_host->byte_clk_rate); @@ -401,7 +413,11 @@ int dsi_link_clk_set_rate_6g(struct msm_dsi_host *msm_host) } if (msm_host->byte_intf_clk) { - ret = clk_set_rate(msm_host->byte_intf_clk, msm_host->byte_intf_clk_rate); + byte_intf_clk_rate = msm_host->byte_clk_rate; + if (msm_host->byte_intf_clk_div_2) + byte_intf_clk_rate /= 2; + + ret = clk_set_rate(msm_host->byte_intf_clk, byte_intf_clk_rate); if (ret) { pr_err("%s: Failed to set rate byte intf clk, %d\n", __func__, ret); @@ -669,24 +685,12 @@ static void dsi_calc_pclk(struct msm_dsi_host *msm_host, bool is_bonded_dsi) int dsi_calc_clk_rate_6g(struct msm_dsi_host *msm_host, bool is_bonded_dsi) { - long rounded_byte_clk_rate; - if (!msm_host->mode) { pr_err("%s: mode not set\n", __func__); return -EINVAL; } dsi_calc_pclk(msm_host, is_bonded_dsi); - - rounded_byte_clk_rate = clk_round_rate(msm_host->byte_clk, - msm_host->byte_clk_rate); - if (rounded_byte_clk_rate < 0) { - pr_err("%s: failed to round byte clock rate, %ld\n", - __func__, rounded_byte_clk_rate); - return rounded_byte_clk_rate; - } - - msm_host->byte_clk_rate = rounded_byte_clk_rate; msm_host->esc_clk_rate = clk_get_rate(msm_host->esc_clk); return 0; } @@ -2495,9 +2499,7 @@ int msm_dsi_host_power_on(struct mipi_dsi_host *host, goto unlock_ret; } - msm_host->byte_intf_clk_rate = msm_host->byte_clk_rate; - if (phy_shared_timings->byte_intf_clk_div_2) - msm_host->byte_intf_clk_rate /= 2; + msm_host->byte_intf_clk_div_2 = phy_shared_timings->byte_intf_clk_div_2; msm_dsi_sfpb_config(msm_host, true); diff --git a/drivers/gpu/drm/msm/hdmi/hdmi_phy.c b/drivers/gpu/drm/msm/hdmi/hdmi_phy.c index eb1088755cb3..77dce35cd45e 100644 --- a/drivers/gpu/drm/msm/hdmi/hdmi_phy.c +++ b/drivers/gpu/drm/msm/hdmi/hdmi_phy.c @@ -168,13 +168,13 @@ static int msm_hdmi_phy_probe(struct platform_device *pdev) ret = msm_hdmi_phy_resource_enable(phy); if (ret) - return ret; + goto err_pm_disable; ret = msm_hdmi_phy_pll_init(pdev, phy->cfg->type); if (ret) { DRM_DEV_ERROR(dev, "couldn't init PLL\n"); msm_hdmi_phy_resource_disable(phy); - return ret; + goto err_pm_disable; } msm_hdmi_phy_resource_disable(phy); @@ -182,6 +182,10 @@ static int msm_hdmi_phy_probe(struct platform_device *pdev) platform_set_drvdata(pdev, phy); return 0; + +err_pm_disable: + pm_runtime_disable(dev); + return ret; } static void msm_hdmi_phy_remove(struct platform_device *pdev) diff --git a/drivers/gpu/drm/msm/msm_drv.c b/drivers/gpu/drm/msm/msm_drv.c index db1b655dd055..f3d2eaa04f14 100644 --- a/drivers/gpu/drm/msm/msm_drv.c +++ b/drivers/gpu/drm/msm/msm_drv.c @@ -55,7 +55,7 @@ MODULE_PARM_DESC(modeset, "Use kernel modesetting [KMS] (1=on (default), 0=disab module_param(modeset, bool, 0600); static bool separate_gpu_kms; -MODULE_PARM_DESC(separate_gpu_drm, "Use separate DRM device for the GPU (0=single DRM device for both GPU and display (default), 1=two DRM devices)"); +MODULE_PARM_DESC(separate_gpu_kms, "Use separate DRM device for the GPU (0=single DRM device for both GPU and display (default), 1=two DRM devices)"); module_param(separate_gpu_kms, bool, 0400); DECLARE_FAULT_ATTR(fail_gem_alloc); diff --git a/drivers/gpu/drm/msm/msm_fbdev.c b/drivers/gpu/drm/msm/msm_fbdev.c index 89ca9da3e1f2..dd6d6c507d77 100644 --- a/drivers/gpu/drm/msm/msm_fbdev.c +++ b/drivers/gpu/drm/msm/msm_fbdev.c @@ -155,6 +155,7 @@ int msm_fbdev_driver_fbdev_probe(struct drm_fb_helper *helper, helper->fb = buffer->fb; fbi->fbops = &msm_fb_ops; + fbi->flags |= FBINFO_VIRTFB; /* system memory */ drm_fb_helper_fill_info(fbi, helper, sizes); diff --git a/drivers/gpu/drm/msm/msm_gem.h b/drivers/gpu/drm/msm/msm_gem.h index dff60cbc9d95..7c6a8c01f910 100644 --- a/drivers/gpu/drm/msm/msm_gem.h +++ b/drivers/gpu/drm/msm/msm_gem.h @@ -68,6 +68,9 @@ struct msm_gem_vm { /** @base: Inherit from drm_gpuvm. */ struct drm_gpuvm base; + /** @rcu: RCU-delayed free so an exported sched fence->sched stays valid. */ + struct rcu_head rcu; + /** * @sched: Scheduler used for asynchronous VM_BIND request. * diff --git a/drivers/gpu/drm/msm/msm_gem_vma.c b/drivers/gpu/drm/msm/msm_gem_vma.c index c11d021581e0..1badec3caa7b 100644 --- a/drivers/gpu/drm/msm/msm_gem_vma.c +++ b/drivers/gpu/drm/msm/msm_gem_vma.c @@ -166,7 +166,7 @@ msm_gem_vm_free(struct drm_gpuvm *gpuvm) dma_fence_put(vm->last_fence); put_pid(vm->pid); kfree(vm->log); - kfree(vm); + kfree_rcu(vm, rcu); } /** diff --git a/drivers/gpu/drm/msm/msm_ringbuffer.c b/drivers/gpu/drm/msm/msm_ringbuffer.c index 59c69aa75649..38e1e6866301 100644 --- a/drivers/gpu/drm/msm/msm_ringbuffer.c +++ b/drivers/gpu/drm/msm/msm_ringbuffer.c @@ -140,5 +140,5 @@ void msm_ringbuffer_destroy(struct msm_ringbuffer *ring) msm_gem_kernel_put(ring->bo, ring->gpu->vm); - kfree(ring); + kfree_rcu(ring, rcu); } diff --git a/drivers/gpu/drm/msm/msm_ringbuffer.h b/drivers/gpu/drm/msm/msm_ringbuffer.h index 3631ec283c6e..0bfb6b0f42fa 100644 --- a/drivers/gpu/drm/msm/msm_ringbuffer.h +++ b/drivers/gpu/drm/msm/msm_ringbuffer.h @@ -55,6 +55,7 @@ struct msm_ringbuffer { /* * The job scheduler for this ring. */ + struct rcu_head rcu; struct drm_gpu_scheduler sched; bool sched_initialized; diff --git a/drivers/gpu/drm/msm/registers/adreno/a6xx_gmu.xml b/drivers/gpu/drm/msm/registers/adreno/a6xx_gmu.xml index 33404eb18fd0..b3082738c99d 100644 --- a/drivers/gpu/drm/msm/registers/adreno/a6xx_gmu.xml +++ b/drivers/gpu/drm/msm/registers/adreno/a6xx_gmu.xml @@ -141,6 +141,8 @@ xsi:schemaLocation="https://gitlab.freedesktop.org/freedreno/ rules-fd.xsd"> <reg32 offset="0x1f9f0" name="GMU_BOOT_KMD_LM_HANDSHAKE"/> <reg32 offset="0x1f957" name="GMU_LLM_GLM_SLEEP_CTRL"/> <reg32 offset="0x1f958" name="GMU_LLM_GLM_SLEEP_STATUS"/> + <reg32 offset="0x1f880" name="GMU_CX_AO_COUNTER_L" variants="A7XX"/> + <reg32 offset="0x1f881" name="GMU_CX_AO_COUNTER_H" variants="A7XX"/> <reg32 offset="0x1f888" name="GMU_ALWAYS_ON_COUNTER_L" variants="A6XX-A7XX"/> <reg32 offset="0x1f840" name="GMU_ALWAYS_ON_COUNTER_L" variants="A8XX-"/> <reg32 offset="0x1f889" name="GMU_ALWAYS_ON_COUNTER_H" variants="A6XX-A7XX"/> diff --git a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/fbsr.c b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/fbsr.c index f128330f30d7..40bf83ea33ac 100644 --- a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/fbsr.c +++ b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/fbsr.c @@ -208,7 +208,7 @@ r535_fbsr_resume(struct nvkm_gsp *gsp) } static int -r535_fbsr_suspend(struct nvkm_gsp *gsp, bool runtime) +r535_fbsr_suspend(struct nvkm_gsp *gsp) { struct nvkm_subdev *subdev = &gsp->subdev; struct nvkm_device *device = subdev->device; diff --git a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/gsp.c b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/gsp.c index f544afa12b6b..94925f1590ea 100644 --- a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/gsp.c +++ b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r535/gsp.c @@ -1749,7 +1749,7 @@ r535_gsp_fini(struct nvkm_gsp *gsp, enum nvkm_suspend_state suspend) sr->sysmemAddrOfSuspendResumeData = gsp->sr.radix3.lvl0.addr; sr->sizeOfSuspendResumeData = len; - ret = rm->api->fbsr->suspend(gsp, suspend == NVKM_RUNTIME_SUSPEND); + ret = rm->api->fbsr->suspend(gsp); if (ret) { nvkm_gsp_mem_dtor(&gsp->sr.meta); nvkm_gsp_radix3_dtor(gsp, &gsp->sr.radix3); @@ -1761,8 +1761,12 @@ r535_gsp_fini(struct nvkm_gsp *gsp, enum nvkm_suspend_state suspend) * TODO: Debug the GSP firmware / RPC handling to find out why * without this Turing (but none of the other architectures) * ends up resetting all channels after resume. + * Additionally, runtime suspend on other architectures quickly + * becomes unreliable without this sleep. If you're experiencing + * issues with runtime suspend, try bumping this delay up and + * sending a patch if it fixes your GPU. */ - msleep(50); + msleep(200); } ret = r535_gsp_rpc_unloading_guest_driver(gsp, suspend); diff --git a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/fbsr.c b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/fbsr.c index 8ef8b4f65588..af5aa5065c3d 100644 --- a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/fbsr.c +++ b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/fbsr.c @@ -62,7 +62,7 @@ r570_fbsr_resume(struct nvkm_gsp *gsp) } static int -r570_fbsr_init(struct nvkm_gsp *gsp, struct sg_table *sgt, u64 size, bool runtime) +r570_fbsr_init(struct nvkm_gsp *gsp, struct sg_table *sgt, u64 size) { NV2080_CTRL_INTERNAL_FBSR_INIT_PARAMS *ctrl; struct nvkm_gsp_object memlist; @@ -81,7 +81,7 @@ r570_fbsr_init(struct nvkm_gsp *gsp, struct sg_table *sgt, u64 size, bool runtim ctrl->hClient = gsp->internal.client.object.handle; ctrl->hSysMem = memlist.handle; ctrl->sysmemAddrOfSuspendResumeData = gsp->sr.meta.addr; - ctrl->bEnteringGcoffState = runtime ? 1 : 0; + ctrl->bEnteringGcoffState = 0; ret = nvkm_gsp_rm_ctrl_wr(&gsp->internal.device.subdevice, ctrl); if (ret) @@ -92,7 +92,7 @@ r570_fbsr_init(struct nvkm_gsp *gsp, struct sg_table *sgt, u64 size, bool runtim } static int -r570_fbsr_suspend(struct nvkm_gsp *gsp, bool runtime) +r570_fbsr_suspend(struct nvkm_gsp *gsp) { struct nvkm_subdev *subdev = &gsp->subdev; struct nvkm_device *device = subdev->device; @@ -133,7 +133,7 @@ r570_fbsr_suspend(struct nvkm_gsp *gsp, bool runtime) return ret; /* Initialise FBSR on RM. */ - ret = r570_fbsr_init(gsp, &gsp->sr.fbsr, size, runtime); + ret = r570_fbsr_init(gsp, &gsp->sr.fbsr, size); if (ret) { nvkm_gsp_sg_free(device, &gsp->sr.fbsr); return ret; diff --git a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/gsp.c b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/gsp.c index 1488771c63fc..b45781cd0dfd 100644 --- a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/gsp.c +++ b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/gsp.c @@ -207,7 +207,8 @@ r570_gsp_set_rmargs(struct nvkm_gsp *gsp, bool resume) args->srInitArguments.bInPMTransition = 0; } else { args->srInitArguments.oldLevel = NV2080_CTRL_GPU_SET_POWER_STATE_GPU_LEVEL_3; - args->srInitArguments.flags = 0; + args->srInitArguments.flags = + GPU_STATE_FLAGS_PRESERVING | GPU_STATE_FLAGS_PM_TRANSITION; args->srInitArguments.bInPMTransition = 1; } diff --git a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/nvrm/gsp.h b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/nvrm/gsp.h index b6075021e74f..c458569af9d7 100644 --- a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/nvrm/gsp.h +++ b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/r570/nvrm/gsp.h @@ -523,6 +523,14 @@ typedef struct #define NV2080_CTRL_GPU_SET_POWER_STATE_GPU_LEVEL_3 (0x00000003U) +#define GPU_STATE_FLAGS_PRESERVING BIT(0) // GPU state is preserved +#define GPU_STATE_FLAGS_VGA_TRANSITION BIT(1) // To be used with GPU_STATE_FLAGS_PRESERVING. +#define GPU_STATE_FLAGS_PM_TRANSITION BIT(2) // To be used with GPU_STATE_FLAGS_PRESERVING. +#define GPU_STATE_FLAGS_PM_SUSPEND BIT(3) +#define GPU_STATE_FLAGS_PM_HIBERNATE BIT(4) +#define GPU_STATE_FLAGS_GC6_TRANSITION BIT(5) // To be used with GPU_STATE_FLAGS_PRESERVING. +#define GPU_STATE_FLAGS_FAST_UNLOAD BIT(6) // Used during windows restart, skips stateDestroy steps + typedef struct { // Magic for verification by secure ucode diff --git a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/rm.h b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/rm.h index fcd0221dcea1..e9ac47d86b69 100644 --- a/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/rm.h +++ b/drivers/gpu/drm/nouveau/nvkm/subdev/gsp/rm/rm.h @@ -79,7 +79,7 @@ struct nvkm_rm_api { } *device; const struct nvkm_rm_api_fbsr { - int (*suspend)(struct nvkm_gsp *, bool runtime); + int (*suspend)(struct nvkm_gsp *); void (*resume)(struct nvkm_gsp *); } *fbsr; diff --git a/drivers/gpu/drm/scheduler/sched_entity.c b/drivers/gpu/drm/scheduler/sched_entity.c index a4a7efdbf229..673ca9cbf362 100644 --- a/drivers/gpu/drm/scheduler/sched_entity.c +++ b/drivers/gpu/drm/scheduler/sched_entity.c @@ -559,9 +559,10 @@ struct drm_sched_job *drm_sched_entity_pop_job(struct drm_sched_entity *entity) */ smp_wmb(); + spin_lock(&entity->lock); spsc_queue_pop(&entity->job_queue); - drm_sched_rq_pop_entity(entity); + spin_unlock(&entity->lock); /* Jobs and entities might have different lifecycles. Since we're * removing the job from the entities queue, set the jobs entity pointer @@ -647,6 +648,9 @@ void drm_sched_entity_push_job(struct drm_sched_job *sched_job) * Make sure to set the submit_ts first, to avoid a race. */ sched_job->submit_ts = submit_ts = ktime_get(); + + spin_lock(&entity->lock); + first = spsc_queue_push(&entity->job_queue, &sched_job->queue_node); /* first job wakes up scheduler */ @@ -657,5 +661,7 @@ void drm_sched_entity_push_job(struct drm_sched_job *sched_job) if (sched) drm_sched_wakeup(sched); } + + spin_unlock(&entity->lock); } EXPORT_SYMBOL(drm_sched_entity_push_job); diff --git a/drivers/gpu/drm/scheduler/sched_rq.c b/drivers/gpu/drm/scheduler/sched_rq.c index 0464d324d98d..23f46ec610e7 100644 --- a/drivers/gpu/drm/scheduler/sched_rq.c +++ b/drivers/gpu/drm/scheduler/sched_rq.c @@ -257,19 +257,17 @@ static ktime_t drm_sched_entity_get_job_ts(struct drm_sched_entity *entity) struct drm_gpu_scheduler * drm_sched_rq_add_entity(struct drm_sched_entity *entity, ktime_t ts) { + struct drm_sched_rq *rq = entity->rq; struct drm_gpu_scheduler *sched; - struct drm_sched_rq *rq; /* Add the entity to the run queue */ - spin_lock(&entity->lock); - if (entity->stopped) { - spin_unlock(&entity->lock); + lockdep_assert_held(&entity->lock); + if (entity->stopped) { DRM_ERROR("Trying to push to a killed entity\n"); return NULL; } - rq = entity->rq; spin_lock(&rq->lock); sched = rq->sched; @@ -289,7 +287,6 @@ drm_sched_rq_add_entity(struct drm_sched_entity *entity, ktime_t ts) drm_sched_rq_update_fifo_locked(entity, rq, ts); spin_unlock(&rq->lock); - spin_unlock(&entity->lock); return sched; } @@ -343,16 +340,17 @@ drm_sched_rq_next_rr_ts(struct drm_sched_rq *rq, */ void drm_sched_rq_pop_entity(struct drm_sched_entity *entity) { + struct drm_sched_rq *rq = entity->rq; struct drm_sched_job *next_job; - struct drm_sched_rq *rq; + + lockdep_assert_held(&entity->lock); + + spin_lock(&rq->lock); /* * Update the entity's location in the min heap according to * the timestamp of the next job, if any. */ - spin_lock(&entity->lock); - rq = entity->rq; - spin_lock(&rq->lock); next_job = drm_sched_entity_queue_peek(entity); if (next_job) { ktime_t ts; @@ -375,8 +373,8 @@ void drm_sched_rq_pop_entity(struct drm_sched_entity *entity) drm_sched_entity_save_vruntime(entity, min_vruntime); } } + spin_unlock(&rq->lock); - spin_unlock(&entity->lock); } /** diff --git a/drivers/gpu/drm/ttm/ttm_bo.c b/drivers/gpu/drm/ttm/ttm_bo.c index ef56c18ded1b..9b85b5f388d4 100644 --- a/drivers/gpu/drm/ttm/ttm_bo.c +++ b/drivers/gpu/drm/ttm/ttm_bo.c @@ -1434,7 +1434,7 @@ ttm_bo_swapout_cb(struct ttm_lru_walk *walk, struct ttm_buffer_object *bo) if (ttm_tt_is_populated(tt)) { ret = ttm_tt_swapout(bdev, tt, swapout_walk->gfp_flags); - if (!ret) { + if (ret > 0) { spin_lock(&bdev->lru_lock); ttm_resource_del_bulk_move_unevictable(bo->resource, bo); ttm_resource_move_to_lru_tail(bo->resource); diff --git a/drivers/gpu/drm/vc4/vc4_kms.c b/drivers/gpu/drm/vc4/vc4_kms.c index b17e73bce384..82a7f2ac4c7e 100644 --- a/drivers/gpu/drm/vc4/vc4_kms.c +++ b/drivers/gpu/drm/vc4/vc4_kms.c @@ -1188,7 +1188,7 @@ int vc4_kms_load(struct drm_device *dev) drm_mode_config_reset(dev); - drm_kms_helper_poll_init(dev); + drmm_kms_helper_poll_init(dev); return 0; } diff --git a/drivers/gpu/drm/verisilicon/vs_cursor_plane.c b/drivers/gpu/drm/verisilicon/vs_cursor_plane.c index fa4f601dd0c8..da456dced88a 100644 --- a/drivers/gpu/drm/verisilicon/vs_cursor_plane.c +++ b/drivers/gpu/drm/verisilicon/vs_cursor_plane.c @@ -11,6 +11,7 @@ #include <drm/drm_atomic.h> #include <drm/drm_atomic_helper.h> +#include <drm/drm_blend.h> #include <drm/drm_crtc.h> #include <drm/drm_fourcc.h> #include <drm/drm_framebuffer.h> @@ -267,6 +268,7 @@ struct drm_plane *vs_cursor_plane_init(struct drm_device *drm_dev, return plane; drm_plane_helper_add(plane, &vs_cursor_plane_helper_funcs); + drm_plane_create_blend_mode_property(plane, BIT(DRM_MODE_BLEND_COVERAGE)); return plane; } diff --git a/drivers/gpu/drm/verisilicon/vs_hwdb.c b/drivers/gpu/drm/verisilicon/vs_hwdb.c index 2a0f7c59afa3..ebf6f843bc88 100644 --- a/drivers/gpu/drm/verisilicon/vs_hwdb.c +++ b/drivers/gpu/drm/verisilicon/vs_hwdb.c @@ -10,82 +10,50 @@ #include "vs_dc_top_regs.h" #include "vs_hwdb.h" -static const u32 vs_formats_array_no_yuv444[] = { +static const u32 vs_primary_formats_array_no_yuv444[] = { DRM_FORMAT_XRGB4444, DRM_FORMAT_XBGR4444, DRM_FORMAT_RGBX4444, DRM_FORMAT_BGRX4444, - DRM_FORMAT_ARGB4444, - DRM_FORMAT_ABGR4444, - DRM_FORMAT_RGBA4444, - DRM_FORMAT_BGRA4444, DRM_FORMAT_XRGB1555, DRM_FORMAT_XBGR1555, DRM_FORMAT_RGBX5551, DRM_FORMAT_BGRX5551, - DRM_FORMAT_ARGB1555, - DRM_FORMAT_ABGR1555, - DRM_FORMAT_RGBA5551, - DRM_FORMAT_BGRA5551, DRM_FORMAT_RGB565, DRM_FORMAT_BGR565, DRM_FORMAT_XRGB8888, DRM_FORMAT_XBGR8888, DRM_FORMAT_RGBX8888, DRM_FORMAT_BGRX8888, - DRM_FORMAT_ARGB8888, - DRM_FORMAT_ABGR8888, - DRM_FORMAT_RGBA8888, - DRM_FORMAT_BGRA8888, - DRM_FORMAT_ARGB2101010, - DRM_FORMAT_ABGR2101010, - DRM_FORMAT_RGBA1010102, - DRM_FORMAT_BGRA1010102, /* TODO: non-RGB formats */ }; -static const u32 vs_formats_array_with_yuv444[] = { +static const u32 vs_primary_formats_array_with_yuv444[] = { DRM_FORMAT_XRGB4444, DRM_FORMAT_XBGR4444, DRM_FORMAT_RGBX4444, DRM_FORMAT_BGRX4444, - DRM_FORMAT_ARGB4444, - DRM_FORMAT_ABGR4444, - DRM_FORMAT_RGBA4444, - DRM_FORMAT_BGRA4444, DRM_FORMAT_XRGB1555, DRM_FORMAT_XBGR1555, DRM_FORMAT_RGBX5551, DRM_FORMAT_BGRX5551, - DRM_FORMAT_ARGB1555, - DRM_FORMAT_ABGR1555, - DRM_FORMAT_RGBA5551, - DRM_FORMAT_BGRA5551, DRM_FORMAT_RGB565, DRM_FORMAT_BGR565, DRM_FORMAT_XRGB8888, DRM_FORMAT_XBGR8888, DRM_FORMAT_RGBX8888, DRM_FORMAT_BGRX8888, - DRM_FORMAT_ARGB8888, - DRM_FORMAT_ABGR8888, - DRM_FORMAT_RGBA8888, - DRM_FORMAT_BGRA8888, - DRM_FORMAT_ARGB2101010, - DRM_FORMAT_ABGR2101010, - DRM_FORMAT_RGBA1010102, - DRM_FORMAT_BGRA1010102, /* TODO: non-RGB formats */ }; static const struct vs_formats vs_formats_no_yuv444 = { - .array = vs_formats_array_no_yuv444, - .num = ARRAY_SIZE(vs_formats_array_no_yuv444) + .primary_array = vs_primary_formats_array_no_yuv444, + .primary_num = ARRAY_SIZE(vs_primary_formats_array_no_yuv444) }; static const struct vs_formats vs_formats_with_yuv444 = { - .array = vs_formats_array_with_yuv444, - .num = ARRAY_SIZE(vs_formats_array_with_yuv444) + .primary_array = vs_primary_formats_array_with_yuv444, + .primary_num = ARRAY_SIZE(vs_primary_formats_array_with_yuv444) }; static struct vs_chip_identity vs_chip_identities[] = { diff --git a/drivers/gpu/drm/verisilicon/vs_hwdb.h b/drivers/gpu/drm/verisilicon/vs_hwdb.h index 2065ecb73043..616076d931a5 100644 --- a/drivers/gpu/drm/verisilicon/vs_hwdb.h +++ b/drivers/gpu/drm/verisilicon/vs_hwdb.h @@ -10,8 +10,8 @@ #include <linux/types.h> struct vs_formats { - const u32 *array; - unsigned int num; + const u32 *primary_array; + unsigned int primary_num; }; struct vs_chip_identity { diff --git a/drivers/gpu/drm/verisilicon/vs_primary_plane.c b/drivers/gpu/drm/verisilicon/vs_primary_plane.c index 1f2be41ae496..8e9449169301 100644 --- a/drivers/gpu/drm/verisilicon/vs_primary_plane.c +++ b/drivers/gpu/drm/verisilicon/vs_primary_plane.c @@ -168,8 +168,8 @@ struct drm_plane *vs_primary_plane_init(struct drm_device *drm_dev, struct vs_dc plane = drmm_universal_plane_alloc(drm_dev, struct drm_plane, dev, 0, &vs_primary_plane_funcs, - dc->identity.formats->array, - dc->identity.formats->num, + dc->identity.formats->primary_array, + dc->identity.formats->primary_num, NULL, DRM_PLANE_TYPE_PRIMARY, NULL); diff --git a/drivers/gpu/drm/xe/xe_i2c.c b/drivers/gpu/drm/xe/xe_i2c.c index 399099ff0dbc..f4f381988289 100644 --- a/drivers/gpu/drm/xe/xe_i2c.c +++ b/drivers/gpu/drm/xe/xe_i2c.c @@ -318,8 +318,10 @@ void xe_i2c_pm_resume(struct xe_device *xe, bool d3cold) static void xe_i2c_remove(void *data) { struct xe_i2c *i2c = data; + struct xe_device *xe = tile_to_xe(i2c->mmio->tile); unsigned int i; + xe_i2c_irq_reset(xe); xe_amc_exit(i2c); for (i = 0; i < XE_I2C_MAX_CLIENTS; i++) { @@ -329,6 +331,7 @@ static void xe_i2c_remove(void *data) bus_unregister_notifier(&i2c_bus_type, &i2c->bus_notifier); xe_i2c_unregister_adapter(i2c); + xe->i2c = NULL; } /** diff --git a/drivers/gpu/drm/xe/xe_mmio_gem.c b/drivers/gpu/drm/xe/xe_mmio_gem.c index 3741ae60f532..5ffe03d36190 100644 --- a/drivers/gpu/drm/xe/xe_mmio_gem.c +++ b/drivers/gpu/drm/xe/xe_mmio_gem.c @@ -5,9 +5,9 @@ #include "xe_mmio_gem.h" +#include <linux/dma-resv.h> #include <drm/drm_drv.h> #include <drm/drm_gem.h> -#include <drm/drm_managed.h> #include "xe_device_types.h" @@ -37,12 +37,24 @@ static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *); struct xe_mmio_gem { struct drm_gem_object base; phys_addr_t phys_addr; + struct page *dummy_page; /* protected by the GEM's dma_resv */ + bool destroyed; /* protected by the GEM's dma_resv */ }; +static int xe_mmio_gem_vm_may_split(struct vm_area_struct *area, unsigned long addr) +{ + /* + * Forbid splitting. Together with VM_DONTEXPAND, this keeps the VMA + * matching the GEM object exactly. + */ + return -EINVAL; +} + static const struct vm_operations_struct vm_ops = { .open = drm_gem_vm_open, .close = drm_gem_vm_close, .fault = xe_mmio_gem_vm_fault, + .may_split = xe_mmio_gem_vm_may_split, }; static const struct drm_gem_object_funcs xe_mmio_gem_funcs = { @@ -121,6 +133,8 @@ static void xe_mmio_gem_free(struct drm_gem_object *base) { struct xe_mmio_gem *obj = to_xe_mmio_gem(base); + if (obj->dummy_page) + __free_page(obj->dummy_page); drm_gem_object_release(base); kfree(obj); } @@ -128,15 +142,31 @@ static void xe_mmio_gem_free(struct drm_gem_object *base) /** * xe_mmio_gem_destroy - Destroy the GEM object that exposes an MMIO region * @gem: the GEM object to destroy + * @file: DRM file descriptor previously passed to xe_mmio_gem_create() * * This function releases resources associated with the GEM object created by * xe_mmio_gem_create(). * * See: "Exposing MMIO regions to userspace" */ -void xe_mmio_gem_destroy(struct xe_mmio_gem *gem) +void xe_mmio_gem_destroy(struct xe_mmio_gem *gem, struct drm_file *file) { - xe_mmio_gem_free(&gem->base); + struct drm_gem_object *base = &gem->base; + struct drm_device *dev = base->dev; + + drm_vma_node_revoke(&base->vma_node, file); + + dma_resv_lock(base->resv, NULL); + gem->destroyed = true; + dma_resv_unlock(base->resv); + /* + * Setting 'destroyed' under lock takes care of the subsequent faults. + * Zap the existing PTEs to cut off access to the real MMIO through + * currently mapped pages. + */ + drm_vma_node_unmap(&base->vma_node, dev->anon_inode->i_mapping); + + drm_gem_object_put(base); } static int xe_mmio_gem_mmap(struct drm_gem_object *base, struct vm_area_struct *vma) @@ -147,8 +177,6 @@ static int xe_mmio_gem_mmap(struct drm_gem_object *base, struct vm_area_struct * if ((vma->vm_flags & VM_SHARED) == 0) return -EINVAL; - /* Set vm_pgoff (used as a fake buffer offset by DRM) to 0 */ - vma->vm_pgoff = 0; vma->vm_page_prot = pgprot_noncached(vma_get_page_prot(vma)); vm_flags_set(vma, VM_IO | VM_PFNMAP | VM_DONTEXPAND | VM_DONTDUMP | VM_DONTCOPY | VM_NORESERVE); @@ -157,51 +185,47 @@ static int xe_mmio_gem_mmap(struct drm_gem_object *base, struct vm_area_struct * return 0; } -static void xe_mmio_gem_release_dummy_page(struct drm_device *dev, void *res) +static int alloc_dummy_page_if_needed(struct drm_gem_object *base) { - __free_page((struct page *)res); + struct xe_mmio_gem *obj = to_xe_mmio_gem(base); + + dma_resv_assert_held(base->resv); + if (!obj->dummy_page) + obj->dummy_page = alloc_page(GFP_KERNEL | __GFP_ZERO); + + return obj->dummy_page ? 0 : -ENOMEM; } -static vm_fault_t xe_mmio_gem_vm_fault_dummy_page(struct vm_area_struct *vma) +static vm_fault_t xe_mmio_gem_vm_fault_dummy_page(struct vm_fault *vmf) { + struct vm_area_struct *vma = vmf->vma; struct drm_gem_object *base = vma->vm_private_data; - struct drm_device *dev = base->dev; - vm_fault_t ret = VM_FAULT_NOPAGE; - struct page *page; + struct xe_mmio_gem *obj = to_xe_mmio_gem(base); unsigned long pfn; - unsigned long i; - page = alloc_page(GFP_KERNEL | __GFP_ZERO); - if (!page) + if (alloc_dummy_page_if_needed(base)) return VM_FAULT_OOM; - if (drmm_add_action_or_reset(dev, xe_mmio_gem_release_dummy_page, page)) - return VM_FAULT_OOM; - - pfn = page_to_pfn(page); - - /* Map the entire VMA to the same dummy page */ - for (i = 0; i < base->size; i += PAGE_SIZE) { - unsigned long addr = vma->vm_start + i; - - ret = vmf_insert_pfn(vma, addr, pfn); - if (ret & VM_FAULT_ERROR) - break; - } + pfn = page_to_pfn(obj->dummy_page); - return ret; + return vmf_insert_pfn_prot(vma, vmf->address, pfn, + vm_get_page_prot(vma->vm_flags)); } -static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *vmf) +static vm_fault_t xe_mmio_gem_vm_fault_locked(struct vm_fault *vmf) { struct vm_area_struct *vma = vmf->vma; struct drm_gem_object *base = vma->vm_private_data; struct xe_mmio_gem *obj = to_xe_mmio_gem(base); struct drm_device *dev = base->dev; vm_fault_t ret = VM_FAULT_NOPAGE; - unsigned long i; + unsigned long addr, pfn; int idx; + dma_resv_assert_held(base->resv); + if (obj->destroyed) + return VM_FAULT_SIGBUS; + if (!drm_dev_enter(dev, &idx)) { /* * Provide a dummy page to avoid SIGBUS for events such as hot-unplug. @@ -209,18 +233,30 @@ static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *vmf) * It is assumed the userspace will receive the notification via some * other channel (e.g. drm uevent). */ - return xe_mmio_gem_vm_fault_dummy_page(vma); + return xe_mmio_gem_vm_fault_dummy_page(vmf); } - for (i = 0; i < base->size; i += PAGE_SIZE) { - unsigned long addr = vma->vm_start + i; - unsigned long phys_addr = obj->phys_addr + i; - - ret = vmf_insert_pfn(vma, addr, PHYS_PFN(phys_addr)); + pfn = PHYS_PFN(obj->phys_addr); + for (addr = vma->vm_start; addr < vma->vm_end; addr += PAGE_SIZE) { + ret = vmf_insert_pfn(vma, addr, pfn); if (ret & VM_FAULT_ERROR) break; + + pfn++; } drm_dev_exit(idx); return ret; } + +static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *vmf) +{ + struct vm_area_struct *vma = vmf->vma; + struct drm_gem_object *base = vma->vm_private_data; + vm_fault_t ret; + + dma_resv_lock(base->resv, NULL); + ret = xe_mmio_gem_vm_fault_locked(vmf); + dma_resv_unlock(base->resv); + return ret; +} diff --git a/drivers/gpu/drm/xe/xe_mmio_gem.h b/drivers/gpu/drm/xe/xe_mmio_gem.h index 4b76d5586ebb..80d7795f07c8 100644 --- a/drivers/gpu/drm/xe/xe_mmio_gem.h +++ b/drivers/gpu/drm/xe/xe_mmio_gem.h @@ -15,6 +15,6 @@ struct xe_mmio_gem; struct xe_mmio_gem *xe_mmio_gem_create(struct xe_device *xe, struct drm_file *file, phys_addr_t phys_addr, size_t size); u64 xe_mmio_gem_mmap_offset(struct xe_mmio_gem *gem); -void xe_mmio_gem_destroy(struct xe_mmio_gem *gem); +void xe_mmio_gem_destroy(struct xe_mmio_gem *gem, struct drm_file *file); #endif /* _XE_MMIO_GEM_H_ */ diff --git a/drivers/gpu/drm/xe/xe_shrinker.c b/drivers/gpu/drm/xe/xe_shrinker.c index 83374cd57660..deb4378c1ec1 100644 --- a/drivers/gpu/drm/xe/xe_shrinker.c +++ b/drivers/gpu/drm/xe/xe_shrinker.c @@ -54,13 +54,40 @@ xe_shrinker_mod_pages(struct xe_shrinker *shrinker, long shrinkable, long purgea write_unlock(&shrinker->lock); } -static s64 __xe_shrinker_walk(struct xe_device *xe, +static bool __xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker) +{ + struct xe_device *xe = shrinker->xe; + + if (xe_pm_runtime_get_if_active(xe)) + return true; + + if (xe_rpm_reclaim_safe(xe) && !ttm_bo_shrink_avoid_wait()) { + xe_pm_runtime_get(xe); + return true; + } + + queue_work(xe->unordered_wq, &shrinker->pm_worker); + + return false; +} + +static void xe_shrinker_runtime_pm_put(struct xe_shrinker *shrinker, bool runtime_pm) +{ + if (runtime_pm) + xe_pm_runtime_put(shrinker->xe); +} + +static int __xe_shrinker_walk(struct xe_shrinker *shrinker, struct ttm_operation_ctx *ctx, const struct xe_bo_shrink_flags flags, - unsigned long to_scan, unsigned long *scanned) + unsigned long to_scan, unsigned long *scanned, + unsigned long *freed) { + struct xe_device *xe = shrinker->xe; unsigned int mem_type; - s64 freed = 0, lret; + bool rpm = false; + int ret = 0; + s64 lret; for (mem_type = XE_PL_SYSTEM; mem_type <= XE_PL_TT; ++mem_type) { struct ttm_resource_manager *man = ttm_manager_type(&xe->ttm, mem_type); @@ -74,23 +101,35 @@ static s64 __xe_shrinker_walk(struct xe_device *xe, if (!man || !man->use_tt) continue; + if (mem_type != XE_PL_SYSTEM && !rpm && + xe_device_is_l2_flush_optimized(xe)) { + if (!__xe_shrinker_runtime_pm_get(shrinker)) + break; + rpm = true; + } + ttm_bo_lru_for_each_reserved_guarded(&curs, man, &arg, ttm_bo) { if (!ttm_bo_shrink_suitable(ttm_bo, ctx)) continue; lret = xe_bo_shrink(ctx, ttm_bo, flags, scanned); - if (lret < 0) - return lret; + if (lret < 0) { + ret = lret; + goto out; + } - freed += lret; + *freed += lret; if (*scanned >= to_scan) - break; + goto out; } /* Trylocks should never error, just fail. */ xe_assert(xe, !IS_ERR(ttm_bo)); } - return freed; +out: + xe_shrinker_runtime_pm_put(shrinker, rpm); + + return ret; } /* @@ -99,40 +138,36 @@ static s64 __xe_shrinker_walk(struct xe_device *xe, * add writeback. This avoids stalls and explicit writebacks with light or * moderate memory pressure. */ -static s64 xe_shrinker_walk(struct xe_device *xe, +static int xe_shrinker_walk(struct xe_shrinker *shrinker, struct ttm_operation_ctx *ctx, const struct xe_bo_shrink_flags flags, - unsigned long to_scan, unsigned long *scanned) + unsigned long to_scan, unsigned long *scanned, + unsigned long *freed) { bool no_wait_gpu = true; struct xe_bo_shrink_flags save_flags = flags; - s64 lret, freed; + int ret; swap(no_wait_gpu, ctx->no_wait_gpu); save_flags.writeback = false; - lret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned); + ret = __xe_shrinker_walk(shrinker, ctx, save_flags, to_scan, scanned, + freed); swap(no_wait_gpu, ctx->no_wait_gpu); - if (lret < 0 || *scanned >= to_scan) - return lret; + if (ret || *scanned >= to_scan) + return ret; - freed = lret; if (!ctx->no_wait_gpu) { - lret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned); - if (lret < 0) - return lret; - freed += lret; - if (*scanned >= to_scan) - return freed; + ret = __xe_shrinker_walk(shrinker, ctx, save_flags, to_scan, scanned, + freed); + if (ret || *scanned >= to_scan) + return ret; } - if (flags.writeback) { - lret = __xe_shrinker_walk(xe, ctx, flags, to_scan, scanned); - if (lret < 0) - return lret; - freed += lret; - } + if (flags.writeback) + ret = __xe_shrinker_walk(shrinker, ctx, flags, to_scan, scanned, + freed); - return freed; + return ret; } static unsigned long @@ -180,22 +215,7 @@ static bool xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker, bool force, return false; } - if (!xe_pm_runtime_get_if_active(xe)) { - if (xe_rpm_reclaim_safe(xe) && !ttm_bo_shrink_avoid_wait()) { - xe_pm_runtime_get(xe); - return true; - } - queue_work(xe->unordered_wq, &shrinker->pm_worker); - return false; - } - - return true; -} - -static void xe_shrinker_runtime_pm_put(struct xe_shrinker *shrinker, bool runtime_pm) -{ - if (runtime_pm) - xe_pm_runtime_put(shrinker->xe); + return __xe_shrinker_runtime_pm_get(shrinker); } static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_control *sc) @@ -214,7 +234,6 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con bool runtime_pm; bool purgeable; bool can_backup = !!(sc->gfp_mask & __GFP_FS); - s64 lret; nr_to_scan = sc->nr_to_scan; @@ -225,12 +244,9 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con /* Might need runtime PM. Try to wake early if it looks like it. */ runtime_pm = xe_shrinker_runtime_pm_get(shrinker, false, nr_to_scan, can_backup); - if (purgeable && nr_scanned < nr_to_scan) { - lret = xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags, - nr_to_scan, &nr_scanned); - if (lret >= 0) - freed += lret; - } + if (purgeable && nr_scanned < nr_to_scan) + xe_shrinker_walk(shrinker, &ctx, shrink_flags, + nr_to_scan, &nr_scanned, &freed); sc->nr_scanned = nr_scanned; if (nr_scanned >= nr_to_scan || !can_backup) @@ -242,10 +258,8 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con shrink_flags.purge = false; - lret = xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags, - nr_to_scan, &nr_scanned); - if (lret >= 0) - freed += lret; + xe_shrinker_walk(shrinker, &ctx, shrink_flags, + nr_to_scan, &nr_scanned, &freed); sc->nr_scanned = nr_scanned; out: diff --git a/drivers/hwmon/cgbc-hwmon.c b/drivers/hwmon/cgbc-hwmon.c index 3aff4e092132..230062a46907 100644 --- a/drivers/hwmon/cgbc-hwmon.c +++ b/drivers/hwmon/cgbc-hwmon.c @@ -41,45 +41,60 @@ enum cgbc_sensor_types { static const char * const cgbc_hwmon_labels_temp[] = { "CPU Temperature", - "Box Temperature", + "Case Temperature", "Ambient Temperature", - "Board Temperature", - "Carrier Temperature", - "Chipset Temperature", - "Video Temperature", + "CPU Board Temperature", + "Carrier Board Temperature", + "System Chipset Temperature", + "Video Controller/Board Temperature", "Other Temperature", - "TOPDIM Temperature", - "BOTTOMDIM Temperature", + "Top DIMM 0 Temperature", + "Bottom DIMM 0 Temperature", + "Alternate Board Temperature", + "Top DIMM 1 Temperature", + "Top DIMM 2 Temperature", + "Top DIMM 3 Temperature", + "Top DIMM 4 Temperature", + "Top DIMM 5 Temperature", + "Top DIMM 6 Temperature", + "Top DIMM 7 Temperature", + "Bottom DIMM 1 Temperature", }; +static const char * const cgbc_hwmon_labels_in[] = { + "CPU Core Voltage", + "DC Runtime Voltage", + "DC Standby Voltage", + "CMOS Battery Voltage", + "Battery Supply Voltage", + "AC Voltage", + "Other Voltage", + "5V Runtime Voltage", + "5V Standby Voltage", + "3V3 Runtime Voltage", + "3V3 Standby Voltage", + "VCore A Voltage", + "VCore B Voltage", + "12V Runtime Voltage", + "12V Standby Voltage", +}; + +/* + * Current sensors are a bit special, they don't have consecutive IDs like + * other types of sensors. So they need to be defined explicitly. + */ static const struct { - enum hwmon_sensor_types type; const char *label; -} cgbc_hwmon_labels_in[] = { - { hwmon_in, "CPU Voltage" }, - { hwmon_in, "DC Runtime Voltage" }, - { hwmon_in, "DC Standby Voltage" }, - { hwmon_in, "CMOS Battery Voltage" }, - { hwmon_in, "Battery Voltage" }, - { hwmon_in, "AC Voltage" }, - { hwmon_in, "Other Voltage" }, - { hwmon_in, "5V Voltage" }, - { hwmon_in, "5V Standby Voltage" }, - { hwmon_in, "3V3 Voltage" }, - { hwmon_in, "3V3 Standby Voltage" }, - { hwmon_in, "VCore A Voltage" }, - { hwmon_in, "VCore B Voltage" }, - { hwmon_in, "12V Voltage" }, - { hwmon_curr, "DC Current" }, - { hwmon_curr, "5V Current" }, - { hwmon_curr, "12V Current" }, + int id; +} cgbc_hwmon_labels_curr[] = { + { "DC Runtime Current", 0x12 }, + { "5V Runtime Current", 0x18 }, + { "12V Runtime Current", 0x1E }, }; -#define CGBC_HWMON_NB_IN_SENSORS 14 - static const char * const cgbc_hwmon_labels_fan[] = { "CPU Fan", - "Box Fan", + "Case Fan", "Ambient Fan", "Chipset Fan", "Video Fan", @@ -114,7 +129,8 @@ static int cgbc_hwmon_probe_sensors(struct device *dev, struct cgbc_hwmon_data * for (i = 0; i < nb_sensors; i++) { enum cgbc_sensor_types type; - unsigned int channel; + unsigned int channel, id; + int j; /* * No need to request data for the first sensor. @@ -128,32 +144,49 @@ static int cgbc_hwmon_probe_sensors(struct device *dev, struct cgbc_hwmon_data * } type = FIELD_GET(CGBC_HWMON_TYPE_MASK, data[1]); - channel = FIELD_GET(CGBC_HWMON_ID_MASK, data[1]) - 1; + id = FIELD_GET(CGBC_HWMON_ID_MASK, data[1]); + channel = id - 1; if (type == CGBC_HWMON_TYPE_TEMP && channel < ARRAY_SIZE(cgbc_hwmon_labels_temp)) { sensor->type = hwmon_temp; sensor->label = cgbc_hwmon_labels_temp[channel]; - } else if (type == CGBC_HWMON_TYPE_IN && - channel < ARRAY_SIZE(cgbc_hwmon_labels_in)) { + } else if (type == CGBC_HWMON_TYPE_IN) { /* * The Board Controller doesn't differentiate current and voltage sensors. - * Get the sensor type from cgbc_hwmon_labels_in[channel].type instead. + * First check if it is a current sensor. */ - sensor->type = cgbc_hwmon_labels_in[channel].type; - sensor->label = cgbc_hwmon_labels_in[channel].label; + for (j = 0; j < ARRAY_SIZE(cgbc_hwmon_labels_curr); j++) { + if (id == cgbc_hwmon_labels_curr[j].id) { + sensor->type = hwmon_curr; + sensor->label = cgbc_hwmon_labels_curr[j].label; + channel = j; + } + } + + /* If it's not a current sensor, it may be a voltage sensor. */ + if (!sensor->label && channel < ARRAY_SIZE(cgbc_hwmon_labels_in)) { + sensor->type = hwmon_in; + sensor->label = cgbc_hwmon_labels_in[channel]; + } } else if (type == CGBC_HWMON_TYPE_FAN && channel < ARRAY_SIZE(cgbc_hwmon_labels_fan)) { sensor->type = hwmon_fan; sensor->label = cgbc_hwmon_labels_fan[channel]; - } else { - dev_warn(dev, "Board Controller returned an unknown sensor (type=%d, channel=%d), ignore it", - type, channel); + } + + if (!sensor->label) { + dev_warn(dev, "Board Controller returned an unknown sensor (bc_type=%d, bc_id=%d), ignore it", + type, id); continue; } sensor->active = FIELD_GET(CGBC_HWMON_ACTIVE_BIT, data[1]); sensor->channel = channel; sensor->index = i; + + dev_dbg(dev, "Found sensor: bc_type=%d, bc_id=%d, hwmon_type=%d, hwmon_channel=%d, hwmon_label='%s', active=%d\n", + type, id, sensor->type, sensor->channel, sensor->label, sensor->active); + sensor++; hwmon->nb_sensors++; } @@ -167,14 +200,6 @@ static struct cgbc_hwmon_sensor *cgbc_hwmon_find_sensor(struct cgbc_hwmon_data * struct cgbc_hwmon_sensor *sensor = NULL; int i; - /* - * The Board Controller doesn't differentiate current and voltage sensors. - * The channel value (from the Board Controller point of view) shall be computed for current - * sensors. - */ - if (type == hwmon_curr) - channel += CGBC_HWMON_NB_IN_SENSORS; - for (i = 0; i < hwmon->nb_sensors; i++) { if (hwmon->sensors[i].type == type && hwmon->sensors[i].channel == channel) { sensor = &hwmon->sensors[i]; @@ -240,7 +265,12 @@ static const struct hwmon_channel_info * const cgbc_hwmon_info[] = { HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, - HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL), + HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, + HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, + HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, + HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, + HWMON_T_INPUT | HWMON_T_LABEL, HWMON_T_INPUT | HWMON_T_LABEL, + HWMON_T_INPUT | HWMON_T_LABEL), HWMON_CHANNEL_INFO(in, HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL, @@ -248,7 +278,8 @@ static const struct hwmon_channel_info * const cgbc_hwmon_info[] = { HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL, - HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL), + HWMON_I_INPUT | HWMON_I_LABEL, HWMON_I_INPUT | HWMON_I_LABEL, + HWMON_I_INPUT | HWMON_I_LABEL), HWMON_CHANNEL_INFO(curr, HWMON_C_INPUT | HWMON_C_LABEL, HWMON_C_INPUT | HWMON_C_LABEL, HWMON_C_INPUT | HWMON_C_LABEL), diff --git a/drivers/hwmon/gpio-fan.c b/drivers/hwmon/gpio-fan.c index df8bd9707605..3f78375eeb44 100644 --- a/drivers/hwmon/gpio-fan.c +++ b/drivers/hwmon/gpio-fan.c @@ -68,7 +68,7 @@ static irqreturn_t fan_alarm_irq_handler(int irq, void *dev_id) schedule_work(&fan_data->alarm_work); - return IRQ_NONE; + return IRQ_HANDLED; } static ssize_t fan1_alarm_show(struct device *dev, @@ -103,7 +103,7 @@ static int fan_alarm_init(struct gpio_fan_data *fan_data) irq_set_irq_type(alarm_irq, IRQ_TYPE_EDGE_BOTH); return devm_request_irq(dev, alarm_irq, fan_alarm_irq_handler, - IRQF_SHARED, "GPIO fan alarm", fan_data); + 0, "GPIO fan alarm", fan_data); } /* diff --git a/drivers/hwmon/hp-wmi-sensors.c b/drivers/hwmon/hp-wmi-sensors.c index 03c684ba83bd..cfa0d6ad0fb3 100644 --- a/drivers/hwmon/hp-wmi-sensors.c +++ b/drivers/hwmon/hp-wmi-sensors.c @@ -526,14 +526,12 @@ static int check_wobj(const union acpi_object *wobj, for (prop = 0; prop <= last_prop; prop++) { type = elements[prop].type; valid_type = property_map[prop]; - if (type != valid_type) { - if (type == ACPI_TYPE_BUFFER && - valid_type == ACPI_TYPE_STRING && - is_raw_wmi_string(elements[prop].buffer.pointer, - elements[prop].buffer.length)) - continue; + if (type == ACPI_TYPE_BUFFER && + is_raw_wmi_string(elements[prop].buffer.pointer, + elements[prop].buffer.length)) + type = ACPI_TYPE_STRING; + if (type != valid_type) return -EINVAL; - } } return 0; @@ -579,6 +577,7 @@ static int check_numeric_sensor_wobj(const union acpi_object *wobj, int prop = HP_WMI_PROPERTY_NAME; acpi_object_type valid_type; union acpi_object *elements; + union acpi_object *element; u32 elem_count; int last_prop; bool is_new; @@ -602,13 +601,20 @@ static int check_numeric_sensor_wobj(const union acpi_object *wobj, elem_count > HP_WMI_MAX_PROPERTIES) return -EINVAL; - type = elements[HP_WMI_PROPERTY_SIZE].type; + element = &elements[HP_WMI_PROPERTY_SIZE]; + type = element->type; switch (type) { case ACPI_TYPE_INTEGER: is_new = true; last_prop = HP_WMI_PROPERTY_RATE_UNITS; break; + case ACPI_TYPE_BUFFER: + if (!is_raw_wmi_string(element->buffer.pointer, + element->buffer.length)) + return -EINVAL; + fallthrough; + case ACPI_TYPE_STRING: is_new = false; last_prop = HP_WMI_PROPERTY_CURRENT_READING; @@ -631,6 +637,10 @@ static int check_numeric_sensor_wobj(const union acpi_object *wobj, for (i = 0; i < elem_count && prop <= last_prop; i++, prop++) { type = elements[i].type; valid_type = hp_wmi_property_map[prop]; + if (type == ACPI_TYPE_BUFFER && + is_raw_wmi_string(elements[i].buffer.pointer, + elements[i].buffer.length)) + type = ACPI_TYPE_STRING; if (type != valid_type) return -EINVAL; @@ -651,6 +661,10 @@ static int check_numeric_sensor_wobj(const union acpi_object *wobj, /* PossibleStates[0] has already been type-checked. */ for (j = 0; i + 1 < elem_count && j + 1 < count; j++) { type = elements[++i].type; + if (type == ACPI_TYPE_BUFFER && + is_raw_wmi_string(elements[i].buffer.pointer, + elements[i].buffer.length)) + type = ACPI_TYPE_STRING; if (type != valid_type) return -EINVAL; } @@ -1247,7 +1261,9 @@ static int fungible_show(struct seq_file *seqf, enum hp_wmi_property prop) break; case HP_WMI_PROPERTY_CURRENT_STATE: + mutex_lock(&state->lock); seq_printf(seqf, "%s\n", nsensor->current_state); + mutex_unlock(&state->lock); break; case HP_WMI_PROPERTY_UNIT_MODIFIER: diff --git a/drivers/hwmon/k10temp.c b/drivers/hwmon/k10temp.c index 75a45010d687..3e7e63edc6a3 100644 --- a/drivers/hwmon/k10temp.c +++ b/drivers/hwmon/k10temp.c @@ -523,7 +523,7 @@ static int k10temp_probe(struct pci_dev *pdev, const struct pci_device_id *id) } } else if (boot_cpu_data.x86 == 0x1a) { switch (boot_cpu_data.x86_model) { - case 0x00 ... 0x2f: /* Zen5 Turin */ + case 0x00 ... 0x1f: /* Zen5 Turin */ data->ccd_offset = 0x1F0; k10temp_get_ccd_support(data, 16); break; diff --git a/drivers/hwmon/pmbus/pmbus.h b/drivers/hwmon/pmbus/pmbus.h index 2cd3216b3cd9..920c1102ab6d 100644 --- a/drivers/hwmon/pmbus/pmbus.h +++ b/drivers/hwmon/pmbus/pmbus.h @@ -242,6 +242,7 @@ enum pmbus_regs { /* * OPERATION */ +#define PB_OPERATION_CONTROL_V_SRC GENMASK(5, 4) #define PB_OPERATION_CONTROL_ON BIT(7) /* @@ -386,7 +387,7 @@ enum pmbus_sensor_classes { }; #define PMBUS_PAGES 32 /* Per PMBus specification */ -#define PMBUS_PHASES 10 /* Maximum number of phases per page */ +#define PMBUS_PHASES 16 /* Maximum number of phases per page */ /* Functionality bit mask */ #define PMBUS_HAVE_VIN BIT(0) diff --git a/drivers/hwmon/pmbus/tps53679.c b/drivers/hwmon/pmbus/tps53679.c index 31e54608b3c9..6c25701b36fc 100644 --- a/drivers/hwmon/pmbus/tps53679.c +++ b/drivers/hwmon/pmbus/tps53679.c @@ -187,7 +187,7 @@ static int tps53676_identify(struct i2c_client *client, return -EIO; for (i = 0; i < 2 * TPS53676_MAX_PHASES; i += 2) { if (buf[i + 1] & 0x80) { - if (buf[i] & 0x08) + if (buf[i] & BIT(4)) phases_b++; else phases_a++; @@ -200,6 +200,15 @@ static int tps53676_identify(struct i2c_client *client, if (phases_b > 0) { info->pages = 2; info->phases[1] = phases_b; + } else { + /* + * pmbus_set_page() does not update the PAGE register on + * single-page devices, so select page 0 explicitly in case + * the boot firmware left the device on another page. + */ + ret = i2c_smbus_write_byte_data(client, PMBUS_PAGE, 0); + if (ret < 0) + return ret; } return 0; } diff --git a/drivers/hwmon/pwm-fan.c b/drivers/hwmon/pwm-fan.c index 3b87f65bae05..c633d7f6464c 100644 --- a/drivers/hwmon/pwm-fan.c +++ b/drivers/hwmon/pwm-fan.c @@ -483,7 +483,6 @@ static void pwm_fan_cleanup(void *__ctx) { struct pwm_fan_ctx *ctx = __ctx; - timer_delete_sync(&ctx->rpm_timer); if (ctx->pwm_shutdown) { ctx->enable_mode = pwm_enable_reg_enable; __set_pwm(ctx, ctx->pwm_shutdown); @@ -494,6 +493,13 @@ static void pwm_fan_cleanup(void *__ctx) } } +static void pwm_fan_timer_cleanup(void *__ctx) +{ + struct pwm_fan_ctx *ctx = __ctx; + + timer_shutdown_sync(&ctx->rpm_timer); +} + static int pwm_fan_probe(struct platform_device *pdev) { struct thermal_cooling_device *cdev; @@ -644,6 +650,10 @@ static int pwm_fan_probe(struct platform_device *pdev) } if (ctx->tach_count > 0) { + ret = devm_add_action_or_reset(dev, pwm_fan_timer_cleanup, ctx); + if (ret) + return ret; + ctx->sample_start = ktime_get(); mod_timer(&ctx->rpm_timer, jiffies + HZ); @@ -700,6 +710,7 @@ static void pwm_fan_shutdown(struct platform_device *pdev) { struct pwm_fan_ctx *ctx = platform_get_drvdata(pdev); + pwm_fan_timer_cleanup(ctx); pwm_fan_cleanup(ctx); } diff --git a/drivers/hwmon/w83791d.c b/drivers/hwmon/w83791d.c index 4a777430af5c..4b07a25ae59e 100644 --- a/drivers/hwmon/w83791d.c +++ b/drivers/hwmon/w83791d.c @@ -1415,6 +1415,7 @@ static void w83791d_remove(struct i2c_client *client) struct w83791d_data *data = i2c_get_clientdata(client); hwmon_device_unregister(data->hwmon_dev); + sysfs_remove_group(&client->dev.kobj, &w83791d_group_fanpwm45); sysfs_remove_group(&client->dev.kobj, &w83791d_group); } diff --git a/drivers/hwmon/w83793.c b/drivers/hwmon/w83793.c index a548586369e1..c6ef04c69856 100644 --- a/drivers/hwmon/w83793.c +++ b/drivers/hwmon/w83793.c @@ -1928,7 +1928,9 @@ exit_remove: for (i = 0; i < ARRAY_SIZE(w83793_temp); i++) device_remove_file(dev, &w83793_temp[i].dev_attr); free_mem: - kfree(data); + mutex_lock(&watchdog_data_mutex); + kref_put(&data->kref, w83793_release_resources); + mutex_unlock(&watchdog_data_mutex); exit: return err; } diff --git a/drivers/i2c/busses/i2c-at91-core.c b/drivers/i2c/busses/i2c-at91-core.c index b64adef778d4..8ca4556d9664 100644 --- a/drivers/i2c/busses/i2c-at91-core.c +++ b/drivers/i2c/busses/i2c-at91-core.c @@ -255,6 +255,7 @@ static int at91_twi_probe(struct platform_device *pdev) if (rc) { pm_runtime_disable(dev->dev); pm_runtime_set_suspended(dev->dev); + at91_twi_dma_release(dev); return rc; } @@ -270,6 +271,8 @@ static void at91_twi_remove(struct platform_device *pdev) i2c_del_adapter(&dev->adapter); + at91_twi_dma_release(dev); + pm_runtime_disable(dev->dev); pm_runtime_set_suspended(dev->dev); } diff --git a/drivers/i2c/busses/i2c-at91-master.c b/drivers/i2c/busses/i2c-at91-master.c index 894cedbca99f..68238cc8aee0 100644 --- a/drivers/i2c/busses/i2c-at91-master.c +++ b/drivers/i2c/busses/i2c-at91-master.c @@ -817,11 +817,21 @@ static int at91_twi_configure_dma(struct at91_twi_dev *dev, u32 phy_addr) error: if (ret != -EPROBE_DEFER) dev_info(dev->dev, "can't get DMA channel, continue without DMA support\n"); + at91_twi_dma_release(dev); + return ret; +} + +void at91_twi_dma_release(struct at91_twi_dev *dev) +{ + struct at91_twi_dma *dma = &dev->dma; + if (dma->chan_rx) dma_release_channel(dma->chan_rx); if (dma->chan_tx) dma_release_channel(dma->chan_tx); - return ret; + dma->chan_rx = NULL; + dma->chan_tx = NULL; + dev->use_dma = false; } static int at91_init_twi_recovery_gpio(struct platform_device *pdev, diff --git a/drivers/i2c/busses/i2c-at91.h b/drivers/i2c/busses/i2c-at91.h index 942e9c3973bb..d68fcbbc3e0f 100644 --- a/drivers/i2c/busses/i2c-at91.h +++ b/drivers/i2c/busses/i2c-at91.h @@ -172,6 +172,7 @@ void at91_twi_irq_restore(struct at91_twi_dev *dev); void at91_init_twi_bus(struct at91_twi_dev *dev); void at91_init_twi_bus_master(struct at91_twi_dev *dev); +void at91_twi_dma_release(struct at91_twi_dev *dev); int at91_twi_probe_master(struct platform_device *pdev, u32 phy_addr, struct at91_twi_dev *dev); diff --git a/drivers/i2c/busses/i2c-imx.c b/drivers/i2c/busses/i2c-imx.c index 19ec056b00af..ff5a91158df9 100644 --- a/drivers/i2c/busses/i2c-imx.c +++ b/drivers/i2c/busses/i2c-imx.c @@ -1880,6 +1880,8 @@ static int i2c_imx_probe(struct platform_device *pdev) clk_notifier_unregister: clk_notifier_unregister(i2c_imx->clk, &i2c_imx->clk_change_nb); + if (i2c_imx->dma) + i2c_imx_dma_free(i2c_imx); free_irq(irq, i2c_imx); rpm_disable: pm_runtime_put_noidle(&pdev->dev); @@ -1920,6 +1922,7 @@ static void i2c_imx_remove(struct platform_device *pdev) pm_runtime_put_noidle(&pdev->dev); pm_runtime_disable(&pdev->dev); + pm_runtime_dont_use_autosuspend(&pdev->dev); } static int i2c_imx_runtime_suspend(struct device *dev) diff --git a/drivers/i2c/busses/i2c-qcom-cci.c b/drivers/i2c/busses/i2c-qcom-cci.c index 25b6e4e9e3fa..873e901a23d7 100644 --- a/drivers/i2c/busses/i2c-qcom-cci.c +++ b/drivers/i2c/busses/i2c-qcom-cci.c @@ -497,10 +497,14 @@ static const struct dev_pm_ops qcom_cci_pm = { SET_RUNTIME_PM_OPS(cci_suspend_runtime, cci_resume_runtime, NULL) }; +static void cci_put_of_node(void *data) +{ + of_node_put(data); +} + static int cci_probe(struct platform_device *pdev) { struct device *dev = &pdev->dev; - struct device_node *child; struct resource *r; struct cci *cci; int ret, i; @@ -516,7 +520,7 @@ static int cci_probe(struct platform_device *pdev) if (!cci->data) return -ENOENT; - for_each_available_child_of_node(dev->of_node, child) { + for_each_available_child_of_node_scoped(dev->of_node, child) { struct cci_master *master; u32 idx; @@ -537,6 +541,9 @@ static int cci_probe(struct platform_device *pdev) master->adap.algo = &cci_algo; master->adap.dev.parent = dev; master->adap.dev.of_node = of_node_get(child); + ret = devm_add_action_or_reset(dev, cci_put_of_node, child); + if (ret) + return ret; master->master = idx; master->cci = cci; @@ -606,10 +613,8 @@ static int cci_probe(struct platform_device *pdev) continue; ret = i2c_add_adapter(&cci->master[i].adap); - if (ret < 0) { - of_node_put(cci->master[i].adap.dev.of_node); + if (ret < 0) goto error_i2c; - } } return 0; @@ -617,10 +622,8 @@ static int cci_probe(struct platform_device *pdev) error_i2c: for (--i ; i >= 0; i--) { - if (cci->master[i].cci) { + if (cci->master[i].cci) i2c_del_adapter(&cci->master[i].adap); - of_node_put(cci->master[i].adap.dev.of_node); - } } disable_clocks: cci_disable_clocks(cci); @@ -636,7 +639,6 @@ static void cci_remove(struct platform_device *pdev) for (i = 0; i < cci->data->num_masters; i++) { if (cci->master[i].cci) { i2c_del_adapter(&cci->master[i].adap); - of_node_put(cci->master[i].adap.dev.of_node); cci_halt(cci, i); } } diff --git a/drivers/i2c/busses/i2c-qcom-geni.c b/drivers/i2c/busses/i2c-qcom-geni.c index 00013b41a6f5..e9e41a174c20 100644 --- a/drivers/i2c/busses/i2c-qcom-geni.c +++ b/drivers/i2c/busses/i2c-qcom-geni.c @@ -1189,8 +1189,10 @@ static int geni_i2c_probe(struct platform_device *pdev) return ret; ret = i2c_add_adapter(&gi2c->adap); - if (ret) + if (ret) { + release_gpi_dma(gi2c); return dev_err_probe(dev, ret, "Error adding i2c adapter\n"); + } dev_dbg(dev, "Geni-I2C adaptor successfully added\n"); diff --git a/drivers/i2c/i2c-atr.c b/drivers/i2c/i2c-atr.c index e6d2af659d81..ca29633dcd62 100644 --- a/drivers/i2c/i2c-atr.c +++ b/drivers/i2c/i2c-atr.c @@ -855,6 +855,7 @@ int i2c_atr_add_adapter(struct i2c_atr *atr, struct i2c_atr_adap_desc *desc) ret = i2c_add_adapter(&chan->adap); if (ret) { + atr->adapter[chan_id] = NULL; dev_err(dev, "failed to add atr-adapter %u (error=%d)\n", chan_id, ret); goto err_free_alias_pool; diff --git a/drivers/infiniband/core/iwpm_util.c b/drivers/infiniband/core/iwpm_util.c index 990cf928b32a..51af8c1c49f9 100644 --- a/drivers/infiniband/core/iwpm_util.c +++ b/drivers/infiniband/core/iwpm_util.c @@ -314,10 +314,6 @@ struct iwpm_nlmsg_request *iwpm_get_nlmsg_request(__u32 nlmsg_seq, if (!nlmsg_request) return NULL; - spin_lock_irqsave(&iwpm_nlmsg_req_lock, flags); - list_add_tail(&nlmsg_request->inprocess_list, &iwpm_nlmsg_req_list); - spin_unlock_irqrestore(&iwpm_nlmsg_req_lock, flags); - kref_init(&nlmsg_request->kref); kref_get(&nlmsg_request->kref); nlmsg_request->nlmsg_seq = nlmsg_seq; @@ -326,6 +322,11 @@ struct iwpm_nlmsg_request *iwpm_get_nlmsg_request(__u32 nlmsg_seq, nlmsg_request->err_code = 0; sema_init(&nlmsg_request->sem, 1); down(&nlmsg_request->sem); + + spin_lock_irqsave(&iwpm_nlmsg_req_lock, flags); + list_add_tail(&nlmsg_request->inprocess_list, &iwpm_nlmsg_req_list); + spin_unlock_irqrestore(&iwpm_nlmsg_req_lock, flags); + return nlmsg_request; } diff --git a/drivers/infiniband/core/mad.c b/drivers/infiniband/core/mad.c index e0b3b36b8b14..3c91f00d09c6 100644 --- a/drivers/infiniband/core/mad.c +++ b/drivers/infiniband/core/mad.c @@ -2059,6 +2059,8 @@ static void ib_mad_complete_recv(struct ib_mad_agent_private *mad_agent_priv, int ret; INIT_LIST_HEAD(&mad_recv_wc->rmpp_list); + list_add(&mad_recv_wc->recv_buf.list, &mad_recv_wc->rmpp_list); + ret = ib_mad_enforce_security(mad_agent_priv, mad_recv_wc->wc->pkey_index); if (ret) { @@ -2067,7 +2069,6 @@ static void ib_mad_complete_recv(struct ib_mad_agent_private *mad_agent_priv, return; } - list_add(&mad_recv_wc->recv_buf.list, &mad_recv_wc->rmpp_list); if (is_kernel_rmpp_data_response(mad_agent_priv, mad_recv_wc)) { spin_lock_irqsave(&mad_agent_priv->lock, flags); mad_send_wr = ib_find_send_mad(mad_agent_priv, mad_recv_wc); diff --git a/drivers/infiniband/core/rdma_core.c b/drivers/infiniband/core/rdma_core.c index fd5651c003ae..a7cbe643e33c 100644 --- a/drivers/infiniband/core/rdma_core.c +++ b/drivers/infiniband/core/rdma_core.c @@ -69,7 +69,6 @@ void ib_uverbs_release_file(struct kref *ref) if (file->disassociate_page) __free_pages(file->disassociate_page, 0); - mutex_destroy(&file->disassociation_lock); mutex_destroy(&file->umap_lock); mutex_destroy(&file->ucontext_lock); kfree(file); diff --git a/drivers/infiniband/core/ucma.c b/drivers/infiniband/core/ucma.c index 4929636f7c53..a15182f7a7c5 100644 --- a/drivers/infiniband/core/ucma.c +++ b/drivers/infiniband/core/ucma.c @@ -1556,9 +1556,10 @@ static ssize_t ucma_process_join(struct ucma_file *file, mutex_lock(&ctx->mutex); ret = rdma_join_multicast(ctx->cm_id, (struct sockaddr *)&mc->addr, join_state, mc); - mutex_unlock(&ctx->mutex); - if (ret) + if (ret) { + mutex_unlock(&ctx->mutex); goto err_xa_erase; + } resp.id = mc->id; if (copy_to_user(u64_to_user_ptr(cmd->response), @@ -1566,6 +1567,7 @@ static ssize_t ucma_process_join(struct ucma_file *file, ret = -EFAULT; goto err_leave_multicast; } + mutex_unlock(&ctx->mutex); xa_store(&multicast_table, mc->id, mc, 0); @@ -1573,7 +1575,6 @@ static ssize_t ucma_process_join(struct ucma_file *file, return 0; err_leave_multicast: - mutex_lock(&ctx->mutex); rdma_leave_multicast(ctx->cm_id, (struct sockaddr *) &mc->addr); mutex_unlock(&ctx->mutex); ucma_cleanup_mc_events(mc); diff --git a/drivers/infiniband/core/uverbs_flow.c b/drivers/infiniband/core/uverbs_flow.c index 1528a294f7f8..de5a2769f084 100644 --- a/drivers/infiniband/core/uverbs_flow.c +++ b/drivers/infiniband/core/uverbs_flow.c @@ -26,6 +26,7 @@ out: return resources; err: + kfree(resources->collection); kfree(resources->counters); kfree(resources); diff --git a/drivers/infiniband/core/uverbs_main.c b/drivers/infiniband/core/uverbs_main.c index 0d88b2ee68ff..2a046c888dfa 100644 --- a/drivers/infiniband/core/uverbs_main.c +++ b/drivers/infiniband/core/uverbs_main.c @@ -644,12 +644,15 @@ static int ib_uverbs_mmap(struct file *filp, struct vm_area_struct *vma) goto out; } - mutex_lock(&file->disassociation_lock); + if (!down_read_trylock(&file->hw_destroy_rwsem)) { + ret = -EIO; + goto out; + } vma->vm_ops = &rdma_umap_ops; ret = ucontext->device->ops.mmap(ucontext, vma); - mutex_unlock(&file->disassociation_lock); + up_read(&file->hw_destroy_rwsem); out: srcu_read_unlock(&file->device->disassociate_srcu, srcu_key); return ret; @@ -671,7 +674,6 @@ static void rdma_umap_open(struct vm_area_struct *vma) /* We are racing with disassociation */ if (!down_read_trylock(&ufile->hw_destroy_rwsem)) goto out_zap; - mutex_lock(&ufile->disassociation_lock); /* * Disassociation already completed, the VMA should already be zapped. @@ -684,12 +686,10 @@ static void rdma_umap_open(struct vm_area_struct *vma) goto out_unlock; rdma_umap_priv_init(priv, vma, opriv->entry); - mutex_unlock(&ufile->disassociation_lock); up_read(&ufile->hw_destroy_rwsem); return; out_unlock: - mutex_unlock(&ufile->disassociation_lock); up_read(&ufile->hw_destroy_rwsem); out_zap: /* @@ -773,7 +773,7 @@ void uverbs_user_mmap_disassociate(struct ib_uverbs_file *ufile) { struct rdma_umap_priv *priv, *next_priv; - mutex_lock(&ufile->disassociation_lock); + lockdep_assert_held_write(&ufile->hw_destroy_rwsem); while (1) { struct mm_struct *mm = NULL; @@ -799,10 +799,8 @@ void uverbs_user_mmap_disassociate(struct ib_uverbs_file *ufile) break; } mutex_unlock(&ufile->umap_lock); - if (!mm) { - mutex_unlock(&ufile->disassociation_lock); + if (!mm) return; - } /* * The umap_lock is nested under mmap_lock since it used within @@ -832,8 +830,6 @@ void uverbs_user_mmap_disassociate(struct ib_uverbs_file *ufile) mmap_read_unlock(mm); mmput(mm); } - - mutex_unlock(&ufile->disassociation_lock); } /** @@ -851,8 +847,11 @@ void rdma_user_mmap_disassociate(struct ib_device *device) mutex_lock(&uverbs_dev->lists_mutex); list_for_each_entry(ufile, &uverbs_dev->uverbs_file_list, list) { - if (ufile->ucontext) + if (ufile->ucontext) { + down_write(&ufile->hw_destroy_rwsem); uverbs_user_mmap_disassociate(ufile); + up_write(&ufile->hw_destroy_rwsem); + } } mutex_unlock(&uverbs_dev->lists_mutex); } @@ -927,8 +926,6 @@ static int ib_uverbs_open(struct inode *inode, struct file *filp) mutex_init(&file->umap_lock); INIT_LIST_HEAD(&file->umaps); - mutex_init(&file->disassociation_lock); - filp->private_data = file; list_add_tail(&file->list, &dev->uverbs_file_list); mutex_unlock(&dev->lists_mutex); diff --git a/drivers/infiniband/core/verbs.c b/drivers/infiniband/core/verbs.c index 04abc80c1327..c43e25d37266 100644 --- a/drivers/infiniband/core/verbs.c +++ b/drivers/infiniband/core/verbs.c @@ -2058,11 +2058,13 @@ int ib_get_eth_speed(struct ib_device *dev, u32 port_num, u16 *speed, u8 *width) return -ENODEV; rtnl_lock(); - rc = __ethtool_get_link_ksettings(netdev, &lksettings); - rtnl_unlock(); - - dev_put(netdev); + if (READ_ONCE(netdev->reg_state) != NETREG_REGISTERED) { + dev_put(netdev); + rtnl_unlock(); + return -ENODEV; + } + rc = __ethtool_get_link_ksettings(netdev, &lksettings); if (!rc && lksettings.base.speed != (u32)SPEED_UNKNOWN) { netdev_speed = lksettings.base.speed; } else { @@ -2071,6 +2073,8 @@ int ib_get_eth_speed(struct ib_device *dev, u32 port_num, u16 *speed, u8 *width) pr_warn("%s speed is unknown, defaulting to %u\n", netdev->name, netdev_speed); } + dev_put(netdev); + rtnl_unlock(); ib_get_width_and_speed(netdev_speed, lksettings.lanes, speed, width); diff --git a/drivers/infiniband/hw/bnxt_re/main.c b/drivers/infiniband/hw/bnxt_re/main.c index ce72db1b4bc3..17654a9e23fe 100644 --- a/drivers/infiniband/hw/bnxt_re/main.c +++ b/drivers/infiniband/hw/bnxt_re/main.c @@ -356,9 +356,13 @@ static int bnxt_re_update_qp1_tos_dscp(struct bnxt_re_dev *rdev) return bnxt_qplib_modify_qp(&rdev->qplib_res, &qp->qplib_qp); } -static void bnxt_re_init_dcb_wq(struct bnxt_re_dev *rdev) +static int bnxt_re_init_dcb_wq(struct bnxt_re_dev *rdev) { rdev->dcb_wq = create_singlethread_workqueue("bnxt_re_dcb_wq"); + if (!rdev->dcb_wq) + return -ENOMEM; + + return 0; } static void bnxt_re_uninit_dcb_wq(struct bnxt_re_dev *rdev) @@ -2339,7 +2343,9 @@ static int bnxt_re_dev_init(struct bnxt_re_dev *rdev, u8 op_type) } bnxt_re_debugfs_add_pdev(rdev); - bnxt_re_init_dcb_wq(rdev); + rc = bnxt_re_init_dcb_wq(rdev); + if (rc) + goto fail; bnxt_re_net_register_async_event(rdev); if (!rdev->is_virtfn) diff --git a/drivers/infiniband/hw/bnxt_re/uapi.c b/drivers/infiniband/hw/bnxt_re/uapi.c index feaf98631fc5..a407c6bc469b 100644 --- a/drivers/infiniband/hw/bnxt_re/uapi.c +++ b/drivers/infiniband/hw/bnxt_re/uapi.c @@ -462,7 +462,6 @@ static int UVERBS_HANDLER(BNXT_RE_METHOD_DBR_ALLOC)(struct uverbs_attr_bundle *a uobj->object = obj; uverbs_finalize_uobj_create(attrs, BNXT_RE_ALLOC_DBR_HANDLE); - dbr.umdbr = dpi->umdbr; dbr.dpi = dpi->dpi; ret = uverbs_copy_to_struct_or_zero(attrs, BNXT_RE_ALLOC_DBR_ATTR, &dbr, sizeof(dbr)); @@ -525,7 +524,6 @@ static int UVERBS_HANDLER(BNXT_RE_METHOD_GET_DEFAULT_DBR)(struct uverbs_attr_bun return PTR_ERR(ib_uctx); uctx = container_of(ib_uctx, struct bnxt_re_ucontext, ib_uctx); - dpi.umdbr = uctx->dpi.umdbr; dpi.dpi = uctx->dpi.dpi; ret = uverbs_copy_to_struct_or_zero(attrs, BNXT_RE_DEFAULT_DBR_ATTR, @@ -543,7 +541,7 @@ DECLARE_UVERBS_NAMED_METHOD(BNXT_RE_METHOD_DBR_ALLOC, UA_MANDATORY), UVERBS_ATTR_PTR_OUT(BNXT_RE_ALLOC_DBR_ATTR, UVERBS_ATTR_STRUCT(struct bnxt_re_db_region, - umdbr), + reserved2), UA_MANDATORY), UVERBS_ATTR_PTR_OUT(BNXT_RE_ALLOC_DBR_OFFSET, UVERBS_ATTR_TYPE(u64), @@ -563,7 +561,7 @@ DECLARE_UVERBS_NAMED_OBJECT(BNXT_RE_OBJECT_DBR, DECLARE_UVERBS_NAMED_METHOD(BNXT_RE_METHOD_GET_DEFAULT_DBR, UVERBS_ATTR_PTR_OUT(BNXT_RE_DEFAULT_DBR_ATTR, UVERBS_ATTR_STRUCT(struct bnxt_re_db_region, - umdbr), + reserved2), UA_MANDATORY)); DECLARE_UVERBS_GLOBAL_METHODS(BNXT_RE_OBJECT_DEFAULT_DBR, diff --git a/drivers/infiniband/hw/efa/efa_com.c b/drivers/infiniband/hw/efa/efa_com.c index 583b1cf0d721..04c6d63c449e 100644 --- a/drivers/infiniband/hw/efa/efa_com.c +++ b/drivers/infiniband/hw/efa/efa_com.c @@ -850,7 +850,7 @@ int efa_com_admin_init(struct efa_com_dev *edev, aq->dmadev = edev->dmadev; aq->efa_dev = edev->efa_dev; - set_bit(EFA_AQ_STATE_POLLING_BIT, &aq->state); + efa_com_set_admin_polling_mode(edev, true); sema_init(&aq->avail_cmds, aq->depth); @@ -868,8 +868,6 @@ int efa_com_admin_init(struct efa_com_dev *edev, if (err) goto err_destroy_sq; - efa_com_set_admin_polling_mode(edev, false); - err = efa_com_admin_init_aenq(edev, aenq_handlers); if (err) goto err_destroy_cq; @@ -1254,7 +1252,7 @@ static void efa_com_destroy_eq(struct efa_com_dev *edev, err); } -static void efa_com_arm_eq(struct efa_com_dev *edev, struct efa_com_eq *eeq) +void efa_com_arm_eq(struct efa_com_dev *edev, struct efa_com_eq *eeq) { u32 val = 0; @@ -1343,7 +1341,6 @@ int efa_com_eq_init(struct efa_com_dev *edev, struct efa_com_eq *eeq, eeq->phase = 1; eeq->depth = params.depth; eeq->cb = cb; - efa_com_arm_eq(edev, eeq); return 0; diff --git a/drivers/infiniband/hw/efa/efa_com.h b/drivers/infiniband/hw/efa/efa_com.h index 0341704d0921..98fb6a42a6cb 100644 --- a/drivers/infiniband/hw/efa/efa_com.h +++ b/drivers/infiniband/hw/efa/efa_com.h @@ -169,6 +169,7 @@ int efa_com_admin_init(struct efa_com_dev *edev, void efa_com_admin_destroy(struct efa_com_dev *edev); int efa_com_eq_init(struct efa_com_dev *edev, struct efa_com_eq *eeq, efa_eqe_handler cb, u16 depth, u8 msix_vec); +void efa_com_arm_eq(struct efa_com_dev *edev, struct efa_com_eq *eeq); void efa_com_eq_destroy(struct efa_com_dev *edev, struct efa_com_eq *eeq); int efa_com_dev_reset(struct efa_com_dev *edev, enum efa_regs_reset_reason_types reset_reason); diff --git a/drivers/infiniband/hw/efa/efa_main.c b/drivers/infiniband/hw/efa/efa_main.c index 4cd80727dcd2..753ed8430b69 100644 --- a/drivers/infiniband/hw/efa/efa_main.c +++ b/drivers/infiniband/hw/efa/efa_main.c @@ -302,28 +302,30 @@ static void efa_set_host_info(struct efa_dev *dev) static void efa_destroy_eq(struct efa_dev *dev, struct efa_eq *eq) { - efa_com_eq_destroy(&dev->edev, &eq->eeq); efa_free_irq(dev, &eq->irq); + efa_com_eq_destroy(&dev->edev, &eq->eeq); } static int efa_create_eq(struct efa_dev *dev, struct efa_eq *eq, u32 msix_vec) { int err; - efa_setup_comp_irq(dev, eq, msix_vec); - err = efa_request_irq(dev, &eq->irq); + err = efa_com_eq_init(&dev->edev, &eq->eeq, efa_process_eqe, + dev->dev_attr.max_eq_depth, msix_vec); if (err) return err; - err = efa_com_eq_init(&dev->edev, &eq->eeq, efa_process_eqe, - dev->dev_attr.max_eq_depth, msix_vec); + efa_setup_comp_irq(dev, eq, msix_vec); + err = efa_request_irq(dev, &eq->irq); if (err) - goto err_free_comp_irq; + goto err_destroy_eq; + + efa_com_arm_eq(&dev->edev, &eq->eeq); return 0; -err_free_comp_irq: - efa_free_irq(dev, &eq->irq); +err_destroy_eq: + efa_com_eq_destroy(&dev->edev, &eq->eeq); return err; } @@ -619,18 +621,21 @@ static struct efa_dev *efa_probe_device(struct pci_dev *pdev) edev->aq.msix_vector_idx = dev->admin_msix_vector_idx; edev->aenq.msix_vector_idx = dev->admin_msix_vector_idx; - err = efa_set_mgmnt_irq(dev); + err = efa_com_admin_init(edev, &aenq_handlers); if (err) goto err_disable_msix; - err = efa_com_admin_init(edev, &aenq_handlers); + err = efa_set_mgmnt_irq(dev); if (err) - goto err_free_mgmnt_irq; + goto err_destroy_admin; + + efa_com_set_admin_polling_mode(edev, false); return dev; -err_free_mgmnt_irq: - efa_free_irq(dev, &dev->admin_irq); +err_destroy_admin: + efa_com_dev_reset(edev, EFA_REGS_RESET_INIT_ERR); + efa_com_admin_destroy(edev); err_disable_msix: efa_disable_msix(dev); err_reg_read_destroy: @@ -654,8 +659,8 @@ static void efa_remove_device(struct pci_dev *pdev, edev = &dev->edev; efa_com_dev_reset(edev, reset_reason); - efa_com_admin_destroy(edev); efa_free_irq(dev, &dev->admin_irq); + efa_com_admin_destroy(edev); efa_disable_msix(dev); efa_com_mmio_reg_read_destroy(edev); devm_iounmap(&pdev->dev, edev->reg_bar); diff --git a/drivers/infiniband/hw/erdma/erdma_main.c b/drivers/infiniband/hw/erdma/erdma_main.c index 7e87a815e853..445182c6bc5d 100644 --- a/drivers/infiniband/hw/erdma/erdma_main.c +++ b/drivers/infiniband/hw/erdma/erdma_main.c @@ -572,8 +572,8 @@ static int erdma_ib_device_add(struct pci_dev *pdev) INIT_LIST_HEAD(&dev->cep_list); spin_lock_init(&dev->lock); - xa_init_flags(&dev->qp_xa, XA_FLAGS_ALLOC1); - xa_init_flags(&dev->cq_xa, XA_FLAGS_ALLOC1); + xa_init_flags(&dev->qp_xa, XA_FLAGS_ALLOC1 | XA_FLAGS_LOCK_IRQ); + xa_init_flags(&dev->cq_xa, XA_FLAGS_ALLOC1 | XA_FLAGS_LOCK_IRQ); dev->next_alloc_cqn = 1; dev->next_alloc_qpn = 1; diff --git a/drivers/infiniband/hw/erdma/erdma_verbs.c b/drivers/infiniband/hw/erdma/erdma_verbs.c index 65b1af1e6623..f18b88bba281 100644 --- a/drivers/infiniband/hw/erdma/erdma_verbs.c +++ b/drivers/infiniband/hw/erdma/erdma_verbs.c @@ -1021,15 +1021,15 @@ int erdma_create_qp(struct ib_qp *ibqp, struct ib_qp_init_attr *attrs, init_completion(&qp->safe_free); if (qp->ibqp.qp_type == IB_QPT_GSI) { - old_entry = xa_store(&dev->qp_xa, 1, qp, GFP_KERNEL); + old_entry = xa_store_irq(&dev->qp_xa, 1, qp, GFP_KERNEL); if (xa_is_err(old_entry)) ret = xa_err(old_entry); else qp->ibqp.qp_num = 1; } else { - ret = xa_alloc_cyclic(&dev->qp_xa, &qp->ibqp.qp_num, qp, - XA_LIMIT(1, dev->attrs.max_qp - 1), - &dev->next_alloc_qpn, GFP_KERNEL); + ret = xa_alloc_cyclic_irq(&dev->qp_xa, &qp->ibqp.qp_num, qp, + XA_LIMIT(1, dev->attrs.max_qp - 1), + &dev->next_alloc_qpn, GFP_KERNEL); } if (ret < 0) { @@ -1089,7 +1089,7 @@ err_out_cmd: else free_kernel_qp(qp); err_out_xa: - xa_erase(&dev->qp_xa, QP_ID(qp)); + xa_erase_irq(&dev->qp_xa, QP_ID(qp)); err_out: return ret; } @@ -1993,9 +1993,9 @@ int erdma_create_cq(struct ib_cq *ibcq, const struct ib_cq_init_attr *attr, refcount_set(&cq->refcount, 1); init_completion(&cq->free); - ret = xa_alloc_cyclic(&dev->cq_xa, &cq->cqn, cq, - XA_LIMIT(1, dev->attrs.max_cq - 1), - &dev->next_alloc_cqn, GFP_KERNEL); + ret = xa_alloc_cyclic_irq(&dev->cq_xa, &cq->cqn, cq, + XA_LIMIT(1, dev->attrs.max_cq - 1), + &dev->next_alloc_cqn, GFP_KERNEL); if (ret < 0) return ret; @@ -2041,7 +2041,7 @@ err_free_res: } err_out_xa: - xa_erase(&dev->cq_xa, cq->cqn); + xa_erase_irq(&dev->cq_xa, cq->cqn); return ret; } diff --git a/drivers/infiniband/hw/hfi1/file_ops.c b/drivers/infiniband/hw/hfi1/file_ops.c index dc548e6802e2..1a36f995c4f6 100644 --- a/drivers/infiniband/hw/hfi1/file_ops.c +++ b/drivers/infiniband/hw/hfi1/file_ops.c @@ -326,6 +326,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) void *memvirt = NULL; dma_addr_t memdma = 0; u8 subctxt, mapio = 0, vmf = 0, type; + size_t memdmalen = 0; ssize_t memlen = 0; int ret = 0; u16 ctxt; @@ -371,7 +372,9 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) mapio = 1; break; case PIO_CRED: { + struct credit_return_base *cr = &dd->cr_base[uctxt->sc->node]; u64 cr_page_offset; + if (flags & VM_WRITE) { ret = -EPERM; goto done; @@ -381,11 +384,18 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) * second or third page allocated for credit returns (if number * of enabled contexts > 64 and 128 respectively). */ - cr_page_offset = ((u64)uctxt->sc->hw_free - - (u64)dd->cr_base[uctxt->numa_id].va) & - PAGE_MASK; - memvirt = dd->cr_base[uctxt->numa_id].va + cr_page_offset; - memdma = dd->cr_base[uctxt->numa_id].dma + cr_page_offset; + cr_page_offset = ((u64)uctxt->sc->hw_free - (u64)cr->va) & + PAGE_MASK; + /* + * dma_mmap_coherent() describes the whole coherent buffer and + * selects the page within it with vma->vm_pgoff, so pass the + * base of the allocation and its length and let vm_pgoff pick + * the page. + */ + vma->vm_pgoff = cr_page_offset >> PAGE_SHIFT; + memvirt = cr->va; + memdma = cr->dma; + memdmalen = TXE_NUM_CONTEXTS * sizeof(struct credit_return); memlen = PAGE_SIZE; flags &= ~VM_MAYWRITE; flags |= VM_DONTCOPY | VM_DONTEXPAND; @@ -567,7 +577,8 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) ret = 0; } else if (memdma) { ret = dma_mmap_coherent(&dd->pcidev->dev, vma, - memvirt, memdma, memlen); + memvirt, memdma, + memdmalen ? memdmalen : memlen); } else if (mapio) { ret = io_remap_pfn_range(vma, vma->vm_start, PFN_DOWN(memaddr), diff --git a/drivers/infiniband/hw/irdma/verbs.c b/drivers/infiniband/hw/irdma/verbs.c index 9cfd84dd6869..5d71300da0ff 100644 --- a/drivers/infiniband/hw/irdma/verbs.c +++ b/drivers/infiniband/hw/irdma/verbs.c @@ -4281,7 +4281,7 @@ static int irdma_post_send(struct ib_qp *ibqp, stag_info.total_len = iwmr->ibmr.length; stag_info.reg_addr_pa = *palloc->level1.addr; stag_info.first_pm_pbl_index = palloc->level1.idx; - stag_info.local_fence = ib_wr->send_flags & IB_SEND_FENCE; + stag_info.local_fence = true; if (iwmr->npages > IRDMA_MIN_PAGES_PER_FMR) stag_info.chunk_size = 1; err = irdma_sc_mr_fast_register(&iwqp->sc_qp, &stag_info, diff --git a/drivers/infiniband/hw/mlx4/sysfs.c b/drivers/infiniband/hw/mlx4/sysfs.c index e688ad66a895..5438224bf325 100644 --- a/drivers/infiniband/hw/mlx4/sysfs.c +++ b/drivers/infiniband/hw/mlx4/sysfs.c @@ -751,11 +751,13 @@ err_add: kobject_put(p); } kobject_put(dev->dev_ports_parent[slave]); + dev->dev_ports_parent[slave] = NULL; err_ports: kobject_put(dev->pkeys.device_parent[slave]); /* extra put for the device_parent create_and_add */ kobject_put(dev->pkeys.device_parent[slave]); + dev->pkeys.device_parent[slave] = NULL; fail_dev: kobject_put(dev->iov_parent); @@ -785,6 +787,8 @@ static void unregister_pkey_tree(struct mlx4_ib_dev *device) return; for (slave = device->dev->persist->num_vfs; slave >= 0; --slave) { + if (!device->pkeys.device_parent[slave]) + continue; list_for_each_entry_safe(p, t, &device->pkeys.pkey_port_list[slave], entry) { diff --git a/drivers/infiniband/hw/mlx5/main.c b/drivers/infiniband/hw/mlx5/main.c index 373ee1f42d4a..a457647edacd 100644 --- a/drivers/infiniband/hw/mlx5/main.c +++ b/drivers/infiniband/hw/mlx5/main.c @@ -1683,11 +1683,8 @@ static int mlx5_ib_query_port_speed_rep(struct mlx5_ib_dev *dev, u32 port_num, struct mlx5_core_dev *mdev; u16 op_mod; - if (!dev->port[port_num - 1].rep) { - mlx5_ib_warn(dev, "Representor doesn't exist for port %u\n", - port_num); - return -EINVAL; - } + if (!dev->port[port_num - 1].rep) + return -ENODEV; rep = dev->port[port_num - 1].rep; mdev = mlx5_eswitch_get_core_dev(rep->esw); diff --git a/drivers/infiniband/sw/rxe/rxe_mcast.c b/drivers/infiniband/sw/rxe/rxe_mcast.c index acd03bd87794..5ca9211ab33b 100644 --- a/drivers/infiniband/sw/rxe/rxe_mcast.c +++ b/drivers/infiniband/sw/rxe/rxe_mcast.c @@ -175,7 +175,9 @@ struct rxe_mcg *rxe_lookup_mcg(struct rxe_dev *rxe, union ib_gid *mgid) * @mgid: multicast address as a gid * @mcg: new mcg object * - * Context: caller should hold rxe->mcg lock + * Initializes the mcg fields. The mcg is private and not yet visible in + * mcg_tree, so this may run without rxe->mcg_lock; __rxe_publish_mcg() + * makes it visible under the lock once it is ready. */ static void __rxe_init_mcg(struct rxe_dev *rxe, union ib_gid *mgid, struct rxe_mcg *mcg) @@ -184,13 +186,22 @@ static void __rxe_init_mcg(struct rxe_dev *rxe, union ib_gid *mgid, memcpy(&mcg->mgid, mgid, sizeof(mcg->mgid)); INIT_LIST_HEAD(&mcg->qp_list); mcg->rxe = rxe; +} - /* caller holds a ref on mcg but that will be - * dropped when mcg goes out of scope. We need to take a ref - * on the pointer that will be saved in the red-black tree - * by __rxe_insert_mcg and used to lookup mcg from mgid later. - * Inserting mcg makes it visible to outside so this should - * be done last after the object is ready. +/** + * __rxe_publish_mcg - make a fully initialized mcg visible in mcg_tree + * @mcg: the mcg object + * + * Context: caller must hold rxe->mcg_lock and a reference on mcg + */ +static void __rxe_publish_mcg(struct rxe_mcg *mcg) +{ + /* caller holds a ref on mcg but that will be dropped when mcg goes + * out of scope. We need to take a ref on the pointer that will be + * saved in the red-black tree by __rxe_insert_mcg and used to lookup + * mcg from mgid later. Inserting mcg makes it visible to outside so + * this is done last after the object is ready and the multicast + * address has been programmed. */ kref_get(&mcg->ref_cnt); __rxe_insert_mcg(mcg); @@ -228,26 +239,37 @@ static struct rxe_mcg *rxe_get_mcg(struct rxe_dev *rxe, union ib_gid *mgid) err = -ENOMEM; goto err_dec; } + __rxe_init_mcg(rxe, mgid, mcg); + + /* program the multicast address while mcg is still private, before + * it is inserted into mcg_tree. dev_mc_add() may sleep so this must + * run outside mcg_lock. On failure mcg was never published, so a + * plain free is correct and the tree is untouched. + */ + err = rxe_mcast_add(rxe, mgid); + if (err) { + kfree(mcg); + goto err_dec; + } spin_lock_bh(&rxe->mcg_lock); - /* re-check to see if someone else just added it */ + /* re-check to see if someone else just added it while we were adding + * the multicast address; if so use theirs and drop ours + */ tmp = __rxe_lookup_mcg(rxe, mgid); if (tmp) { spin_unlock_bh(&rxe->mcg_lock); + rxe_mcast_del(rxe, mgid); atomic_dec(&rxe->mcg_num); kfree(mcg); return tmp; } - __rxe_init_mcg(rxe, mgid, mcg); + __rxe_publish_mcg(mcg); spin_unlock_bh(&rxe->mcg_lock); - /* add mcast address outside of lock */ - err = rxe_mcast_add(rxe, mgid); - if (!err) - return mcg; + return mcg; - kfree(mcg); err_dec: atomic_dec(&rxe->mcg_num); return ERR_PTR(err); diff --git a/drivers/infiniband/sw/rxe/rxe_mr.c b/drivers/infiniband/sw/rxe/rxe_mr.c index 875eceb55fdf..71d9ea477289 100644 --- a/drivers/infiniband/sw/rxe/rxe_mr.c +++ b/drivers/infiniband/sw/rxe/rxe_mr.c @@ -33,7 +33,8 @@ int mr_check_range(struct rxe_mr *mr, u64 iova, size_t length) case IB_MR_TYPE_USER: case IB_MR_TYPE_MEM_REG: if (iova < mr->ibmr.iova || - iova + length > mr->ibmr.iova + mr->ibmr.length) { + length > mr->ibmr.length || + iova - mr->ibmr.iova > mr->ibmr.length - length) { rxe_dbg_mr(mr, "iova/length out of range\n"); return -EINVAL; } diff --git a/drivers/infiniband/sw/rxe/rxe_odp.c b/drivers/infiniband/sw/rxe/rxe_odp.c index e870efa7a0a3..ab21b620e94c 100644 --- a/drivers/infiniband/sw/rxe/rxe_odp.c +++ b/drivers/infiniband/sw/rxe/rxe_odp.c @@ -120,19 +120,23 @@ int rxe_odp_mr_init_user(struct rxe_dev *rxe, u64 start, u64 length, } static inline bool rxe_check_pagefault(struct ib_umem_odp *umem_odp, u64 iova, - int length) + int length, bool write) { bool need_fault = false; + u64 access = HMM_PFN_VALID; u64 addr; int idx; + if (write) + access |= HMM_PFN_WRITE; + addr = iova & (~(BIT(umem_odp->page_shift) - 1)); /* Skim through all pages that are to be accessed. */ while (addr < iova + length) { idx = (addr - ib_umem_start(umem_odp)) >> umem_odp->page_shift; - if (!(umem_odp->map.pfn_list[idx] & HMM_PFN_VALID)) { + if ((umem_odp->map.pfn_list[idx] & access) != access) { need_fault = true; break; } @@ -155,6 +159,7 @@ static unsigned long rxe_odp_iova_to_page_offset(struct ib_umem_odp *umem_odp, u static int rxe_odp_map_range_and_lock(struct rxe_mr *mr, u64 iova, int length, u32 flags) { struct ib_umem_odp *umem_odp = to_ib_umem_odp(mr->umem); + bool write = !(flags & RXE_PAGEFAULT_RDONLY); bool need_fault; int err; @@ -163,7 +168,7 @@ static int rxe_odp_map_range_and_lock(struct rxe_mr *mr, u64 iova, int length, u mutex_lock(&umem_odp->umem_mutex); - need_fault = rxe_check_pagefault(umem_odp, iova, length); + need_fault = rxe_check_pagefault(umem_odp, iova, length, write); if (need_fault) { mutex_unlock(&umem_odp->umem_mutex); @@ -173,7 +178,7 @@ static int rxe_odp_map_range_and_lock(struct rxe_mr *mr, u64 iova, int length, u if (err < 0) return err; - need_fault = rxe_check_pagefault(umem_odp, iova, length); + need_fault = rxe_check_pagefault(umem_odp, iova, length, write); if (need_fault) { mutex_unlock(&umem_odp->umem_mutex); return -EFAULT; @@ -335,8 +340,9 @@ int rxe_odp_flush_pmem_iova(struct rxe_mr *mr, u64 iova, int err; u8 *va; + /* A flush never modifies memory; read-only access suffices. */ err = rxe_odp_map_range_and_lock(mr, iova, length, - RXE_PAGEFAULT_DEFAULT); + RXE_PAGEFAULT_RDONLY); if (err) return err; diff --git a/drivers/infiniband/sw/rxe/rxe_verbs.c b/drivers/infiniband/sw/rxe/rxe_verbs.c index 96c7716057fe..3864284522eb 100644 --- a/drivers/infiniband/sw/rxe/rxe_verbs.c +++ b/drivers/infiniband/sw/rxe/rxe_verbs.c @@ -1331,19 +1331,20 @@ static struct ib_mr *rxe_rereg_user_mr(struct ib_mr *ibmr, int flags, if (err) return ERR_PTR(err); + if ((flags & IB_MR_REREG_ACCESS) && + (access & ~RXE_ACCESS_SUPPORTED_MR)) { + rxe_err_mr(mr, "access = %#x not supported\n", access); + return ERR_PTR(-EOPNOTSUPP); + } + if (flags & IB_MR_REREG_PD) { rxe_put(old_pd); rxe_get(pd); mr->ibmr.pd = ibpd; } - if (flags & IB_MR_REREG_ACCESS) { - if (access & ~RXE_ACCESS_SUPPORTED_MR) { - rxe_err_mr(mr, "access = %#x not supported\n", access); - return ERR_PTR(-EOPNOTSUPP); - } + if (flags & IB_MR_REREG_ACCESS) mr->access = access; - } return NULL; } diff --git a/drivers/infiniband/sw/siw/siw_cm.c b/drivers/infiniband/sw/siw/siw_cm.c index 0245b25e7271..ed49818793dd 100644 --- a/drivers/infiniband/sw/siw/siw_cm.c +++ b/drivers/infiniband/sw/siw/siw_cm.c @@ -1719,9 +1719,12 @@ int siw_accept(struct iw_cm_id *id, struct iw_cm_conn_param *params) SIW_QP_ATTR_STATE | SIW_QP_ATTR_LLP_HANDLE | SIW_QP_ATTR_ORD | SIW_QP_ATTR_IRD | SIW_QP_ATTR_MPA); + if (rv) { + qp->cep = NULL; + siw_cep_put(cep); + goto error_unlock; + } up_write(&qp->state_lock); - if (rv) - goto error; siw_dbg_cep(cep, "[QP %u]: send mpa reply, %d byte pdata\n", qp_id(qp), params->private_data_len); diff --git a/drivers/infiniband/sw/siw/siw_qp_rx.c b/drivers/infiniband/sw/siw/siw_qp_rx.c index b566d163c5aa..e5b641c8d694 100644 --- a/drivers/infiniband/sw/siw/siw_qp_rx.c +++ b/drivers/infiniband/sw/siw/siw_qp_rx.c @@ -1079,7 +1079,7 @@ static int siw_get_hdr(struct siw_rx_stream *srx) if (iwarp_pktinfo[opcode].hdr_len > sizeof(struct iwarp_ctrl_tagged)) { int hdrlen = iwarp_pktinfo[opcode].hdr_len; - bytes = min_t(int, hdrlen - MIN_DDP_HDR, srx->skb_new); + bytes = min_t(int, hdrlen - srx->fpdu_part_rcvd, srx->skb_new); skb_copy_bits(skb, srx->skb_offset, (char *)c_hdr + srx->fpdu_part_rcvd, bytes); diff --git a/drivers/infiniband/ulp/ipoib/ipoib.h b/drivers/infiniband/ulp/ipoib/ipoib.h index 91f866e3fb8b..143e03b64902 100644 --- a/drivers/infiniband/ulp/ipoib/ipoib.h +++ b/drivers/infiniband/ulp/ipoib/ipoib.h @@ -87,6 +87,7 @@ enum { IPOIB_FLAG_INITIALIZED = 1, IPOIB_FLAG_ADMIN_UP = 2, IPOIB_PKEY_ASSIGNED = 3, + IPOIB_FLAG_MCAST_FLUSH = 4, IPOIB_FLAG_SUBINTERFACE = 5, IPOIB_STOP_REAPER = 7, IPOIB_FLAG_ADMIN_CM = 9, @@ -414,6 +415,12 @@ struct ipoib_dev_priv { const struct net_device_ops *rn_ops; }; +static inline bool ipoib_mcast_allowed(struct ipoib_dev_priv *priv) +{ + return test_bit(IPOIB_FLAG_OPER_UP, &priv->flags) && + !test_bit(IPOIB_FLAG_MCAST_FLUSH, &priv->flags); +} + struct ipoib_ah { struct net_device *dev; struct ib_ah *ah; diff --git a/drivers/infiniband/ulp/ipoib/ipoib_ib.c b/drivers/infiniband/ulp/ipoib/ipoib_ib.c index 5061d52a7b12..81bbb3f7c113 100644 --- a/drivers/infiniband/ulp/ipoib/ipoib_ib.c +++ b/drivers/infiniband/ulp/ipoib/ipoib_ib.c @@ -1227,17 +1227,19 @@ static void __ipoib_ib_dev_flush(struct ipoib_dev_priv *priv, } if (level == IPOIB_FLUSH_LIGHT) { - int oper_up; ipoib_mark_paths_invalid(dev); - /* Set IPoIB operation as down to prevent races between: + /* Set MCAST_FLUSH to prevent races between: * the flush flow which leaves MCG and on the fly joins * which can happen during that time. mcast restart task * should deal with join requests we missed. + * + * Do not clear OPER_UP for this; restoring it races with + * ipoib_ib_dev_down() and can leave OPER_UP set after the + * device is down. */ - oper_up = test_and_clear_bit(IPOIB_FLAG_OPER_UP, &priv->flags); + set_bit(IPOIB_FLAG_MCAST_FLUSH, &priv->flags); ipoib_mcast_dev_flush(dev); - if (oper_up) - set_bit(IPOIB_FLAG_OPER_UP, &priv->flags); + clear_bit(IPOIB_FLAG_MCAST_FLUSH, &priv->flags); ipoib_reap_dead_ahs(priv); } diff --git a/drivers/infiniband/ulp/ipoib/ipoib_multicast.c b/drivers/infiniband/ulp/ipoib/ipoib_multicast.c index 6401af2fd548..379b78374e21 100644 --- a/drivers/infiniband/ulp/ipoib/ipoib_multicast.c +++ b/drivers/infiniband/ulp/ipoib/ipoib_multicast.c @@ -74,7 +74,7 @@ static void __ipoib_mcast_schedule_join_thread(struct ipoib_dev_priv *priv, struct ipoib_mcast *mcast, bool delay) { - if (!test_bit(IPOIB_FLAG_OPER_UP, &priv->flags)) + if (!ipoib_mcast_allowed(priv)) return; /* @@ -469,7 +469,7 @@ static int ipoib_mcast_join(struct net_device *dev, struct ipoib_mcast *mcast) int ret = 0; if (!priv->broadcast || - !test_bit(IPOIB_FLAG_OPER_UP, &priv->flags)) + !ipoib_mcast_allowed(priv)) return -EINVAL; init_completion(&mcast->done); @@ -555,7 +555,7 @@ void ipoib_mcast_join_task(struct work_struct *work) unsigned long delay_until = 0; struct ipoib_mcast *mcast = NULL; - if (!test_bit(IPOIB_FLAG_OPER_UP, &priv->flags)) + if (!ipoib_mcast_allowed(priv)) return; if (ib_query_port(priv->ca, priv->port, &port_attr)) { @@ -577,7 +577,7 @@ void ipoib_mcast_join_task(struct work_struct *work) netif_addr_unlock_bh(dev); spin_lock_irq(&priv->lock); - if (!test_bit(IPOIB_FLAG_OPER_UP, &priv->flags)) + if (!ipoib_mcast_allowed(priv)) goto out; if (!priv->broadcast) { @@ -749,7 +749,7 @@ void ipoib_mcast_send(struct net_device *dev, u8 *daddr, struct sk_buff *skb) spin_lock_irqsave(&priv->lock, flags); - if (!test_bit(IPOIB_FLAG_OPER_UP, &priv->flags) || + if (!ipoib_mcast_allowed(priv) || !priv->broadcast || !test_bit(IPOIB_MCAST_FLAG_ATTACHED, &priv->broadcast->flags)) { ++dev->stats.tx_dropped; @@ -871,7 +871,7 @@ void ipoib_mcast_restart_task(struct work_struct *work) LIST_HEAD(remove_list); struct ib_sa_mcmember_rec rec; - if (!test_bit(IPOIB_FLAG_OPER_UP, &priv->flags)) + if (!ipoib_mcast_allowed(priv)) /* * shortcut...on shutdown flush is called next, just * let it do all the work @@ -965,9 +965,9 @@ void ipoib_mcast_restart_task(struct work_struct *work) ipoib_mcast_remove_list(&remove_list); /* - * Double check that we are still up + * Double check that we are still up and not flushing */ - if (test_bit(IPOIB_FLAG_OPER_UP, &priv->flags)) { + if (ipoib_mcast_allowed(priv)) { spin_lock_irq(&priv->lock); __ipoib_mcast_schedule_join_thread(priv, NULL, 0); spin_unlock_irq(&priv->lock); diff --git a/drivers/infiniband/ulp/iser/iser_initiator.c b/drivers/infiniband/ulp/iser/iser_initiator.c index 12a2d12fef07..7ea6888b479c 100644 --- a/drivers/infiniband/ulp/iser/iser_initiator.c +++ b/drivers/infiniband/ulp/iser/iser_initiator.c @@ -598,11 +598,8 @@ static int iser_check_remote_inv(struct iser_conn *iser_conn, struct ib_wc *wc, iser_dbg("conn %p: remote invalidation for rkey %#x\n", iser_conn, rkey); - if (unlikely(!iser_conn->snd_w_inv)) { - iser_err("conn %p: unexpected remote invalidation, terminating connection\n", - iser_conn); - return -EPROTO; - } + if (unlikely(!iser_conn->snd_w_inv)) + goto bad_inv; task = iscsi_itt_to_ctask(iser_conn->iscsi_conn, hdr->itt); if (likely(task)) { @@ -611,12 +608,16 @@ static int iser_check_remote_inv(struct iser_conn *iser_conn, struct ib_wc *wc, if (iser_task->dir[ISER_DIR_IN]) { desc = iser_task->rdma_reg[ISER_DIR_IN].desc; + if (unlikely(!desc)) + goto bad_inv; if (unlikely(iser_inv_desc(desc, rkey))) return -EINVAL; } if (iser_task->dir[ISER_DIR_OUT]) { desc = iser_task->rdma_reg[ISER_DIR_OUT].desc; + if (unlikely(!desc)) + goto bad_inv; if (unlikely(iser_inv_desc(desc, rkey))) return -EINVAL; } @@ -627,6 +628,11 @@ static int iser_check_remote_inv(struct iser_conn *iser_conn, struct ib_wc *wc, } return 0; + +bad_inv: + iser_err("conn %p: unexpected remote invalidation, terminating connection\n", + iser_conn); + return -EPROTO; } diff --git a/drivers/infiniband/ulp/isert/ib_isert.c b/drivers/infiniband/ulp/isert/ib_isert.c index 5087ea983071..e69db43370ff 100644 --- a/drivers/infiniband/ulp/isert/ib_isert.c +++ b/drivers/infiniband/ulp/isert/ib_isert.c @@ -21,6 +21,7 @@ #include <target/target_core_fabric.h> #include <target/iscsi/iscsi_transport.h> #include <linux/semaphore.h> +#include <linux/wait_bit.h> #include "ib_isert.h" @@ -310,6 +311,7 @@ isert_init_conn(struct isert_conn *isert_conn) init_completion(&isert_conn->login_req_comp); init_waitqueue_head(&isert_conn->rem_wait); kref_init(&isert_conn->kref); + atomic_set(&isert_conn->ctrl_comp_cnt, 0); mutex_init(&isert_conn->mutex); INIT_WORK(&isert_conn->release_work, isert_release_work); } @@ -1694,6 +1696,8 @@ isert_do_control_comp(struct work_struct *work) struct isert_conn *isert_conn = isert_cmd->conn; struct ib_device *ib_dev = isert_conn->cm_id->device; struct iscsit_cmd *cmd = isert_cmd->iscsit_cmd; + /* The switch below may free isert_cmd. */ + bool counted = isert_cmd->ctrl_counted; isert_dbg("Cmd %p i_state %d\n", isert_cmd, cmd->i_state); @@ -1715,6 +1719,14 @@ isert_do_control_comp(struct work_struct *work) dump_stack(); break; } + + /* + * The count is what keeps isert_conn alive, so drop it last. The wait + * queue lives in the global hash table, not in isert_conn, so this is + * safe even if the waiter has already freed the connection. + */ + if (counted && atomic_dec_and_test(&isert_conn->ctrl_comp_cnt)) + wake_up_var(&isert_conn->ctrl_comp_cnt); } static void @@ -1758,6 +1770,12 @@ isert_send_done(struct ib_cq *cq, struct ib_wc *wc) case ISTATE_SEND_TEXTRSP: isert_unmap_tx_desc(tx_desc, ib_dev); + /* Paired with the wait in isert_wait_conn(). */ + isert_cmd->ctrl_counted = + isert_cmd->iscsit_cmd->i_state != ISTATE_SEND_LOGOUTRSP; + if (isert_cmd->ctrl_counted) + atomic_inc(&isert_conn->ctrl_comp_cnt); + INIT_WORK(&isert_cmd->comp_work, isert_do_control_comp); queue_work(isert_comp_wq, &isert_cmd->comp_work); return; @@ -2602,6 +2620,10 @@ static void isert_wait_conn(struct iscsit_conn *conn) isert_wait4cmds(conn); isert_wait4logout(isert_conn); + /* Paired with the count taken in isert_send_done(). */ + wait_var_event(&isert_conn->ctrl_comp_cnt, + !atomic_read(&isert_conn->ctrl_comp_cnt)); + queue_work(isert_release_wq, &isert_conn->release_work); } diff --git a/drivers/infiniband/ulp/isert/ib_isert.h b/drivers/infiniband/ulp/isert/ib_isert.h index 0bac5aa66c80..519b17e54bd3 100644 --- a/drivers/infiniband/ulp/isert/ib_isert.h +++ b/drivers/infiniband/ulp/isert/ib_isert.h @@ -153,6 +153,7 @@ struct isert_cmd { struct work_struct comp_work; struct scatterlist sg; bool ctx_init_done; + bool ctrl_counted; }; static inline struct isert_cmd *tx_desc_to_cmd(struct iser_tx_desc *desc) @@ -187,6 +188,7 @@ struct isert_conn { struct mutex mutex; struct kref kref; struct work_struct release_work; + atomic_t ctrl_comp_cnt; bool logout_posted; bool snd_w_inv; wait_queue_head_t rem_wait; diff --git a/drivers/infiniband/ulp/rtrs/rtrs-clt-trace.h b/drivers/infiniband/ulp/rtrs/rtrs-clt-trace.h index 7738e2676855..29e23404bb7b 100644 --- a/drivers/infiniband/ulp/rtrs/rtrs-clt-trace.h +++ b/drivers/infiniband/ulp/rtrs/rtrs-clt-trace.h @@ -55,7 +55,7 @@ DECLARE_EVENT_CLASS(rtrs_clt_conn_class, __entry->max_reconnect_attempts = clt->max_reconnect_attempts; __entry->fail_cnt = clt_path->stats->reconnects.fail_cnt; __entry->success_cnt = clt_path->stats->reconnects.successful_cnt; - memcpy(__entry->sessname, kobject_name(&clt_path->kobj), NAME_MAX); + strscpy(__entry->sessname, kobject_name(&clt_path->kobj) ?: "", NAME_MAX); ), TP_printk("RTRS-CLT: sess='%s' state=%s attempts='%d' max-attempts='%d' fail='%d' success='%d'", diff --git a/drivers/infiniband/ulp/rtrs/rtrs-clt.c b/drivers/infiniband/ulp/rtrs/rtrs-clt.c index 7b2c51ae614f..eac38b57b00d 100644 --- a/drivers/infiniband/ulp/rtrs/rtrs-clt.c +++ b/drivers/infiniband/ulp/rtrs/rtrs-clt.c @@ -1732,6 +1732,8 @@ static void destroy_con_cq_qp(struct rtrs_clt_con *con) /* * Be careful here: destroy_con_cq_qp() can be called even * create_con_cq_qp() failed, see comments there. + * Caller must set con->destroyed under this lock first so a + * racing ADDR_RESOLVED cannot ib_cq_pool_get() after we PUT/SKIP. */ lockdep_assert_held(&con->con_mutex); rtrs_cq_qp_destroy(&con->c); @@ -1766,6 +1768,10 @@ static int rtrs_rdma_addr_resolved(struct rtrs_clt_con *con) int err; mutex_lock(&con->con_mutex); + if (con->destroyed) { + mutex_unlock(&con->con_mutex); + return -ECONNABORTED; + } err = create_con_cq_qp(con); mutex_unlock(&con->con_mutex); if (err) { @@ -2221,6 +2227,7 @@ static void rtrs_clt_stop_and_destroy_conns(struct rtrs_clt_path *clt_path) break; con = to_clt_con(clt_path->s.con[cid]); mutex_lock(&con->con_mutex); + con->destroyed = true; destroy_con_cq_qp(con); mutex_unlock(&con->con_mutex); destroy_cm(con); @@ -2387,6 +2394,7 @@ destroy: if (con->c.cm_id) { stop_cm(con); mutex_lock(&con->con_mutex); + con->destroyed = true; destroy_con_cq_qp(con); mutex_unlock(&con->con_mutex); destroy_cm(con); diff --git a/drivers/infiniband/ulp/rtrs/rtrs-clt.h b/drivers/infiniband/ulp/rtrs/rtrs-clt.h index 1305601a6251..ad64f4517c4b 100644 --- a/drivers/infiniband/ulp/rtrs/rtrs-clt.h +++ b/drivers/infiniband/ulp/rtrs/rtrs-clt.h @@ -75,6 +75,8 @@ struct rtrs_clt_con { unsigned int cpu; struct mutex con_mutex; int cm_err; + /* Set under con_mutex before CQ/QP teardown. */ + bool destroyed; }; /** diff --git a/drivers/infiniband/ulp/rtrs/rtrs-srv-trace.h b/drivers/infiniband/ulp/rtrs/rtrs-srv-trace.h index 587d3e033081..a7d7b971e6c8 100644 --- a/drivers/infiniband/ulp/rtrs/rtrs-srv-trace.h +++ b/drivers/infiniband/ulp/rtrs/rtrs-srv-trace.h @@ -61,7 +61,7 @@ TRACE_EVENT(send_io_resp_imm, __entry->msg_id = id->msg_id; __entry->wr_cnt = atomic_read(&con->c.wr_cnt); __entry->signal_interval = s->signal_interval; - memcpy(__entry->sessname, kobject_name(&srv_path->kobj), NAME_MAX); + strscpy(__entry->sessname, kobject_name(&srv_path->kobj) ?: "", NAME_MAX); ), TP_printk("sess='%s' state='%s' dir=%s err='%d' inval='%d' glob-inval='%d' msgid='%u' wrcnt='%d' sig-interval='%u'", diff --git a/drivers/infiniband/ulp/srp/ib_srp.c b/drivers/infiniband/ulp/srp/ib_srp.c index 6b429ef63f8f..955f36efeebd 100644 --- a/drivers/infiniband/ulp/srp/ib_srp.c +++ b/drivers/infiniband/ulp/srp/ib_srp.c @@ -1038,15 +1038,20 @@ static void srp_del_scsi_host_attr(struct Scsi_Host *shost) static void srp_remove_target(struct srp_target_port *target) { + struct scsi_device *sdev; struct srp_rdma_ch *ch; int i; WARN_ON_ONCE(target->state != SRP_TARGET_REMOVED); srp_del_scsi_host_attr(target->scsi_host); - srp_rport_get(target->rport); - srp_remove_host(target->scsi_host); - scsi_remove_host(target->scsi_host); + /* + * Remove all logical units. This must happen before the + * srp_disconnect_target() call because scsi_remove_device() may trigger + * submission of SCSI commands. See also sd_shutdown(). + */ + shost_for_each_device(sdev, target->scsi_host) + scsi_remove_device(sdev); srp_stop_rport_timers(target->rport); srp_disconnect_target(target); kobj_ns_drop(KOBJ_NS_TYPE_NET, to_ns_common(target->net)); @@ -1055,7 +1060,8 @@ static void srp_remove_target(struct srp_target_port *target) srp_free_ch_ib(target, ch); } cancel_work_sync(&target->tl_err_work); - srp_rport_put(target->rport); + srp_remove_host(target->scsi_host); + scsi_remove_host(target->scsi_host); kfree(target->ch); target->ch = NULL; diff --git a/drivers/input/evdev.c b/drivers/input/evdev.c index 3a718d600006..8bfaaa45e0b9 100644 --- a/drivers/input/evdev.c +++ b/drivers/input/evdev.c @@ -1229,6 +1229,8 @@ static long evdev_do_ioctl(struct file *file, unsigned int cmd, t = _IOC_NR(cmd) & ABS_MAX; + memset(&abs, 0, sizeof(abs)); + if (copy_from_user(&abs, p, min_t(size_t, size, sizeof(struct input_absinfo)))) return -EFAULT; diff --git a/drivers/input/input-compat.c b/drivers/input/input-compat.c index a5043193ead8..8860a0730294 100644 --- a/drivers/input/input-compat.c +++ b/drivers/input/input-compat.c @@ -76,6 +76,8 @@ int input_ff_effect_from_user(const char __user *buffer, size_t size, */ compat_effect = (struct ff_effect_compat *)effect; + memset(effect, 0, sizeof(*effect)); + if (copy_from_user(compat_effect, buffer, sizeof(struct ff_effect_compat))) return -EFAULT; diff --git a/drivers/input/joystick/xpad.c b/drivers/input/joystick/xpad.c index 2da0b7f1722a..6406ceab988a 100644 --- a/drivers/input/joystick/xpad.c +++ b/drivers/input/joystick/xpad.c @@ -215,7 +215,7 @@ static const struct xpad_device { { 0x0e6f, 0x0139, "Afterglow Prismatic Wired Controller", 0, XTYPE_XBOXONE }, { 0x0e6f, 0x013a, "PDP Xbox One Controller", 0, XTYPE_XBOXONE }, { 0x0e6f, 0x0146, "Rock Candy Wired Controller for Xbox One", 0, XTYPE_XBOXONE }, - { 0x0e6f, 0x0147, "PDP Marvel Xbox One Controller", 0, XTYPE_XBOXONE }, + { 0x0e6f, 0x0147, "PDP Marvel Xbox 360 Controller", 0, XTYPE_XBOX360 }, { 0x0e6f, 0x015c, "PDP Xbox One Arcade Stick", MAP_TRIGGERS_TO_BUTTONS, XTYPE_XBOXONE }, { 0x0e6f, 0x015d, "PDP Mirror's Edge Official Wired Controller for Xbox One", 0, XTYPE_XBOXONE }, { 0x0e6f, 0x0161, "PDP Xbox One Controller", 0, XTYPE_XBOXONE }, @@ -227,6 +227,7 @@ static const struct xpad_device { { 0x0e6f, 0x0213, "Afterglow Gamepad for Xbox 360", 0, XTYPE_XBOX360 }, { 0x0e6f, 0x021f, "Rock Candy Gamepad for Xbox 360", 0, XTYPE_XBOX360 }, { 0x0e6f, 0x0246, "Rock Candy Gamepad for Xbox One 2015", 0, XTYPE_XBOXONE }, + { 0x0e6f, 0x024c, "PDP Victrix Pro BFG Wired Controller for Xbox", 0, XTYPE_XBOXONE }, { 0x0e6f, 0x02a0, "PDP Xbox One Controller", 0, XTYPE_XBOXONE }, { 0x0e6f, 0x02a1, "PDP Xbox One Controller", 0, XTYPE_XBOXONE }, { 0x0e6f, 0x02a2, "PDP Wired Controller for Xbox One - Crimson Red", 0, XTYPE_XBOXONE }, @@ -292,6 +293,12 @@ static const struct xpad_device { { 0x1689, 0xfd00, "Razer Onza Tournament Edition", 0, XTYPE_XBOX360 }, { 0x1689, 0xfd01, "Razer Onza Classic Edition", 0, XTYPE_XBOX360 }, { 0x1689, 0xfe00, "Razer Sabertooth", 0, XTYPE_XBOX360 }, + { 0x16d0, 0x1103, "Azeron Cyro", 0, XTYPE_XBOX360 }, + { 0x16d0, 0x113c, "Azeron Cyborg", 0, XTYPE_XBOX360 }, + { 0x16d0, 0x1192, "Azeron Classic/Compact", 0, XTYPE_XBOX360 }, + { 0x16d0, 0x1212, "Azeron Cyro Lefty", 0, XTYPE_XBOX360 }, + { 0x16d0, 0x12f7, "Azeron Cyborg II", 0, XTYPE_XBOX360 }, + { 0x16d0, 0x13ea, "Azeron Keyzen", 0, XTYPE_XBOX360 }, { 0x17ef, 0x6182, "Lenovo Legion Controller for Windows", 0, XTYPE_XBOX360 }, { 0x1949, 0x041a, "Amazon Game Controller", 0, XTYPE_XBOX360 }, { 0x1a86, 0xe310, "Legion Go S", 0, XTYPE_XBOX360 }, @@ -534,6 +541,7 @@ static const struct usb_device_id xpad_table[] = { XPAD_XBOX360_VENDOR(0x15e4), /* Numark Xbox 360 controllers */ XPAD_XBOX360_VENDOR(0x162e), /* Joytech Xbox 360 controllers */ XPAD_XBOX360_VENDOR(0x1689), /* Razer Onza */ + XPAD_XBOX360_VENDOR(0x16d0), /* Azeron controllers */ XPAD_XBOX360_VENDOR(0x17ef), /* Lenovo */ XPAD_XBOX360_VENDOR(0x1949), /* Amazon controllers */ XPAD_XBOX360_VENDOR(0x1a86), /* Nanjing Qinheng Microelectronics (WCH) */ diff --git a/drivers/input/keyboard/adp5588-keys.c b/drivers/input/keyboard/adp5588-keys.c index 40371f5bd9ba..4f0ddff5baba 100644 --- a/drivers/input/keyboard/adp5588-keys.c +++ b/drivers/input/keyboard/adp5588-keys.c @@ -446,12 +446,6 @@ static int adp5588_gpio_add(struct adp5588_kpad *kpad) mutex_init(&kpad->gpio_lock); - error = devm_gpiochip_add_data(dev, &kpad->gc, kpad); - if (error) { - dev_err(dev, "gpiochip_add failed: %d\n", error); - return error; - } - for (i = 0; i <= ADP5588_BANK(ADP5588_MAXGPIO); i++) { kpad->dat_out[i] = adp5588_read(kpad->client, GPIO_DAT_OUT1 + i); @@ -459,6 +453,12 @@ static int adp5588_gpio_add(struct adp5588_kpad *kpad) kpad->pull_dis[i] = adp5588_read(kpad->client, GPIO_PULL1 + i); } + error = devm_gpiochip_add_data(dev, &kpad->gc, kpad); + if (error) { + dev_err(dev, "gpiochip_add failed: %d\n", error); + return error; + } + return 0; } diff --git a/drivers/input/keyboard/atkbd.c b/drivers/input/keyboard/atkbd.c index b9ad2381f885..93650c98206d 100644 --- a/drivers/input/keyboard/atkbd.c +++ b/drivers/input/keyboard/atkbd.c @@ -1946,6 +1946,14 @@ static const struct dmi_system_id atkbd_dmi_quirk_table[] __initconst = { }, .callback = atkbd_deactivate_fixup, }, + { + /* Xiaomi Redmi Book Pro 16 2026 (TM2425) */ + .matches = { + DMI_MATCH(DMI_SYS_VENDOR, "XIAOMI"), + DMI_MATCH(DMI_PRODUCT_NAME, "REDMI Book Pro 16 2026"), + }, + .callback = atkbd_deactivate_fixup, + }, { } }; diff --git a/drivers/input/misc/soc_button_array.c b/drivers/input/misc/soc_button_array.c index a6c984205123..064243677fcf 100644 --- a/drivers/input/misc/soc_button_array.c +++ b/drivers/input/misc/soc_button_array.c @@ -15,6 +15,7 @@ #include <linux/dmi.h> #include <linux/gpio/consumer.h> #include <linux/gpio_keys.h> +#include <linux/platform_data/x86/soc.h> #include <linux/platform_device.h> static bool use_low_level_irq; @@ -159,7 +160,7 @@ soc_button_device_create(struct platform_device *pdev, struct gpio_keys_platform_data *gpio_keys_pdata; const struct dmi_system_id *dmi_id; int invalid_acpi_index = -1; - int error, gpio, irq; + int error, gpio, irq = 0; int n_buttons = 0; for (info = button_info; info->name; info++) @@ -190,8 +191,9 @@ soc_button_device_create(struct platform_device *pdev, error = soc_button_lookup_gpio(&pdev->dev, info->acpi_index, &gpio, &irq); if (error || irq < 0) { /* - * Skip GPIO if not present. Note we deliberately - * ignore -EPROBE_DEFER errors here. On some devices + * Propagate -EPROBE_DEFER, skip button on other errors. + * + * -EPROBE_DEFER is ignored on Bay & Cherry Trail. Here * Intel is using so called virtual GPIOs which are not * GPIOs at all but some way for AML code to check some * random status bits without need a custom opregion. @@ -200,6 +202,12 @@ soc_button_device_create(struct platform_device *pdev, * we do not have a driver for these so they will never * show up, therefore we ignore -EPROBE_DEFER. */ + if ((error == -EPROBE_DEFER || irq == -EPROBE_DEFER) && + !(soc_intel_is_byt() || soc_intel_is_cht())) { + error = -EPROBE_DEFER; + goto err_free_mem; + } + continue; } @@ -368,7 +376,7 @@ static struct soc_button_info *soc_button_get_button_info(struct device *dev) } } - if (!btns_desc) { + if (!btns_desc || !btns_desc->package.count) { dev_err(dev, "ACPI Button Descriptors not found\n"); button_info = ERR_PTR(-ENODEV); goto out; diff --git a/drivers/input/mouse/synaptics.c b/drivers/input/mouse/synaptics.c index 2170bbe4c589..a8ce89d41faa 100644 --- a/drivers/input/mouse/synaptics.c +++ b/drivers/input/mouse/synaptics.c @@ -1838,6 +1838,14 @@ static int synaptics_setup_intertouch(struct psmouse *psmouse, return -ENXIO; } + + /* Disable intertouch on known-broken board revisions */ + if (info->board_id == 2722) { + psmouse_info(psmouse, + "Disabling intertouch for board id %u\n", + info->board_id); + return -ENXIO; + } } psmouse_info(psmouse, "Trying to set up SMBus access\n"); diff --git a/drivers/input/rmi4/rmi_driver.c b/drivers/input/rmi4/rmi_driver.c index 5d49a9021c7d..a349dfd17519 100644 --- a/drivers/input/rmi4/rmi_driver.c +++ b/drivers/input/rmi4/rmi_driver.c @@ -991,6 +991,15 @@ int rmi_driver_suspend(struct rmi_device *rmi_dev, bool enable_wake) { int retval; + /* + * Transport driver will try to suspend RMI device even if physical + * driver did not bind to the RMI device, because transport device + * (I2C, SPI) is fully registered and operational. Exit early if + * there is no driver data attached to the RMI device. + */ + if (!dev_get_drvdata(&rmi_dev->dev)) + return 0; + retval = rmi_suspend_functions(rmi_dev); if (retval) dev_warn(&rmi_dev->dev, "Failed to suspend functions: %d\n", @@ -1005,6 +1014,10 @@ int rmi_driver_resume(struct rmi_device *rmi_dev, bool clear_wake) { int retval; + /* Skip if not fully bound to RMI driver */ + if (!dev_get_drvdata(&rmi_dev->dev)) + return 0; + rmi_enable_irq(rmi_dev, clear_wake); retval = rmi_resume_functions(rmi_dev); diff --git a/drivers/input/rmi4/rmi_smbus.c b/drivers/input/rmi4/rmi_smbus.c index 3160714a514a..8b61c2382297 100644 --- a/drivers/input/rmi4/rmi_smbus.c +++ b/drivers/input/rmi4/rmi_smbus.c @@ -140,7 +140,7 @@ static int rmi_smb_write_block(struct rmi_transport_dev *xport, u16 rmiaddr, u8 commandcode; struct rmi_smb_xport *rmi_smb = container_of(xport, struct rmi_smb_xport, xport); - int cur_len = (int)len; + size_t cur_len = len; mutex_lock(&rmi_smb->page_mutex); @@ -148,7 +148,7 @@ static int rmi_smb_write_block(struct rmi_transport_dev *xport, u16 rmiaddr, /* * break into 32 bytes chunks to write get command code */ - int block_len = min_t(int, len, SMB_MAX_COUNT); + int block_len = min_t(size_t, cur_len, SMB_MAX_COUNT); retval = rmi_smb_get_command_code(xport, rmiaddr, block_len, false, &commandcode); @@ -161,9 +161,9 @@ static int rmi_smb_write_block(struct rmi_transport_dev *xport, u16 rmiaddr, goto exit; /* prepare to write next block of bytes */ - cur_len -= SMB_MAX_COUNT; - databuff += SMB_MAX_COUNT; - rmiaddr += SMB_MAX_COUNT; + cur_len -= block_len; + databuff += block_len; + rmiaddr += block_len; } exit: mutex_unlock(&rmi_smb->page_mutex); diff --git a/drivers/input/serio/hp_sdc.c b/drivers/input/serio/hp_sdc.c index 1461ef319f92..ecf5347b63ab 100644 --- a/drivers/input/serio/hp_sdc.c +++ b/drivers/input/serio/hp_sdc.c @@ -981,7 +981,7 @@ static void hp_sdc_exit(void) free_irq(hp_sdc.irq, &hp_sdc); write_unlock_irq(&hp_sdc.lock); - timer_delete_sync(&hp_sdc.kicker); + timer_shutdown_sync(&hp_sdc.kicker); tasklet_kill(&hp_sdc.task); diff --git a/drivers/input/serio/i8042-acpipnpio.h b/drivers/input/serio/i8042-acpipnpio.h index 9ecb0eed48c4..54eb3687000a 100644 --- a/drivers/input/serio/i8042-acpipnpio.h +++ b/drivers/input/serio/i8042-acpipnpio.h @@ -262,6 +262,13 @@ static const struct dmi_system_id i8042_dmi_quirk_table[] __initconst = { { .matches = { DMI_MATCH(DMI_SYS_VENDOR, "Acer"), + DMI_MATCH(DMI_PRODUCT_NAME, "Aspire AG15-42P"), + }, + .driver_data = (void *)(SERIO_QUIRK_RESET_ALWAYS) + }, + { + .matches = { + DMI_MATCH(DMI_SYS_VENDOR, "Acer"), DMI_MATCH(DMI_PRODUCT_NAME, "Aspire ES1-132"), }, .driver_data = (void *)(SERIO_QUIRK_RESET_ALWAYS) diff --git a/drivers/input/touchscreen/cyttsp5.c b/drivers/input/touchscreen/cyttsp5.c index 9266c07314be..e878a02dc9b7 100644 --- a/drivers/input/touchscreen/cyttsp5.c +++ b/drivers/input/touchscreen/cyttsp5.c @@ -710,6 +710,7 @@ static irqreturn_t cyttsp5_handle_irq(int irq, void *handle) size = 2; } else { report_id = ts->input_buf[2]; + size = min(size, CY_MAX_INPUT); } switch (report_id) { diff --git a/drivers/input/touchscreen/eeti_ts.c b/drivers/input/touchscreen/eeti_ts.c index b12602bc368e..b870a939508b 100644 --- a/drivers/input/touchscreen/eeti_ts.c +++ b/drivers/input/touchscreen/eeti_ts.c @@ -276,6 +276,7 @@ static const struct of_device_id of_eeti_ts_match[] = { { .compatible = "eeti,exc3000-i2c", }, { } }; +MODULE_DEVICE_TABLE(of, of_eeti_ts_match); #endif static struct i2c_driver eeti_ts_driver = { diff --git a/drivers/input/touchscreen/tsc2007_core.c b/drivers/input/touchscreen/tsc2007_core.c index e4d7da0f4434..e2f49b37e18c 100644 --- a/drivers/input/touchscreen/tsc2007_core.c +++ b/drivers/input/touchscreen/tsc2007_core.c @@ -221,7 +221,6 @@ static int tsc2007_get_pendown_state_gpio(struct device *dev) static int tsc2007_probe_properties(struct device *dev, struct tsc2007 *ts) { u32 val32; - u64 val64; if (!device_property_read_u32(dev, "ti,max-rt", &val32)) ts->max_rt = val32; @@ -237,8 +236,8 @@ static int tsc2007_probe_properties(struct device *dev, struct tsc2007 *ts) if (!device_property_read_u32(dev, "ti,fuzzz", &val32)) ts->fuzzz = val32; - if (!device_property_read_u64(dev, "ti,poll-period", &val64)) - ts->poll_period = msecs_to_jiffies(val64); + if (!device_property_read_u32(dev, "ti,poll-period", &val32)) + ts->poll_period = msecs_to_jiffies(val32); else ts->poll_period = msecs_to_jiffies(1); diff --git a/drivers/memstick/core/ms_block.c b/drivers/memstick/core/ms_block.c index ce33907bfc24..65154e569a9e 100644 --- a/drivers/memstick/core/ms_block.c +++ b/drivers/memstick/core/ms_block.c @@ -2204,6 +2204,8 @@ static void msb_remove(struct memstick_dev *card) msb_data_clear(msb); mutex_unlock(&msb_disk_lock); + destroy_workqueue(msb->io_queue); + put_disk(msb->disk); memstick_set_drvdata(card, NULL); } diff --git a/drivers/mmc/core/bus.c b/drivers/mmc/core/bus.c index be5cf338bdeb..92ad39c0cabc 100644 --- a/drivers/mmc/core/bus.c +++ b/drivers/mmc/core/bus.c @@ -417,8 +417,8 @@ void mmc_remove_card(struct mmc_card *card) mmc_hostname(card->host), card->rca); } device_del(&card->dev); - of_node_put(card->dev.of_node); } + of_node_put(card->dev.of_node); if (host->cqe_enabled) { host->cqe_ops->cqe_disable(host); diff --git a/drivers/mmc/core/host.c b/drivers/mmc/core/host.c index b7ce3137d452..828078b3f962 100644 --- a/drivers/mmc/core/host.c +++ b/drivers/mmc/core/host.c @@ -698,6 +698,7 @@ EXPORT_SYMBOL(mmc_remove_host); void mmc_free_host(struct mmc_host *host) { cancel_delayed_work_sync(&host->detect); + cancel_work_sync(&host->sdio_irq_work); mmc_pwrseq_free(host); put_device(&host->class_dev); } diff --git a/drivers/mmc/core/sdio_uart.c b/drivers/mmc/core/sdio_uart.c index 7fd5dedf3ac0..705163c42972 100644 --- a/drivers/mmc/core/sdio_uart.c +++ b/drivers/mmc/core/sdio_uart.c @@ -104,6 +104,9 @@ static int sdio_uart_add_port(struct sdio_uart_port *port) } spin_unlock(&sdio_uart_table_lock); + if (ret) + kfifo_free(&port->xmit_fifo); + return ret; } diff --git a/drivers/mmc/host/mmc_hsq.c b/drivers/mmc/host/mmc_hsq.c index 79836705c176..57e172bd3487 100644 --- a/drivers/mmc/host/mmc_hsq.c +++ b/drivers/mmc/host/mmc_hsq.c @@ -7,6 +7,7 @@ * Author: Baolin Wang <baolin.wang@linaro.org> */ +#include <linux/devm-helpers.h> #include <linux/mmc/card.h> #include <linux/mmc/host.h> #include <linux/module.h> @@ -345,6 +346,7 @@ static const struct mmc_cqe_ops mmc_hsq_ops = { int mmc_hsq_init(struct mmc_hsq *hsq, struct mmc_host *mmc) { + int ret; int i; hsq->num_slots = HSQ_NUM_SLOTS; hsq->next_tag = HSQ_INVALID_TAG; @@ -363,7 +365,11 @@ int mmc_hsq_init(struct mmc_hsq *hsq, struct mmc_host *mmc) for (i = 0; i < HSQ_NUM_SLOTS; i++) hsq->tag_slot[i] = HSQ_INVALID_TAG; - INIT_WORK(&hsq->retry_work, mmc_hsq_retry_handler); + ret = devm_work_autocancel(mmc_dev(mmc), &hsq->retry_work, + mmc_hsq_retry_handler); + if (ret) + return ret; + spin_lock_init(&hsq->lock); init_waitqueue_head(&hsq->wait_queue); diff --git a/drivers/mmc/host/mmc_spi.c b/drivers/mmc/host/mmc_spi.c index b471a7795b4d..5217d713c826 100644 --- a/drivers/mmc/host/mmc_spi.c +++ b/drivers/mmc/host/mmc_spi.c @@ -952,6 +952,7 @@ crc_recover: status = mmc_spi_command_send(host, mrq, &stop, 0); crc_retry--; mrq->data->error = 0; + mrq->data->bytes_xfered = 0; goto crc_recover; } diff --git a/drivers/mmc/host/mmci.c b/drivers/mmc/host/mmci.c index e500051bd572..416bdb184ed4 100644 --- a/drivers/mmc/host/mmci.c +++ b/drivers/mmc/host/mmci.c @@ -2511,6 +2511,9 @@ static void mmci_remove(struct amba_device *dev) writel(0, host->base + MMCICOMMAND); writel(0, host->base + MMCIDATACTRL); + if (variant->busy_detect) + disable_delayed_work_sync(&host->ux500_busy_timeout_work); + mmci_dma_release(host); clk_disable_unprepare(host->clk); } diff --git a/drivers/mmc/host/mxcmmc.c b/drivers/mmc/host/mxcmmc.c index c405cfb8b269..097498a3f8ff 100644 --- a/drivers/mmc/host/mxcmmc.c +++ b/drivers/mmc/host/mxcmmc.c @@ -1173,6 +1173,10 @@ static void mxcmci_remove(struct platform_device *pdev) mmc_remove_host(mmc); + devm_free_irq(&pdev->dev, platform_get_irq(pdev, 0), host); + cancel_work_sync(&host->datawork); + timer_delete_sync(&host->watchdog); + if (host->pdata && host->pdata->exit) host->pdata->exit(&pdev->dev, mmc); diff --git a/drivers/mmc/host/rtsx_pci_sdmmc.c b/drivers/mmc/host/rtsx_pci_sdmmc.c index 8dfbc62f165b..d25d4cb59bc9 100644 --- a/drivers/mmc/host/rtsx_pci_sdmmc.c +++ b/drivers/mmc/host/rtsx_pci_sdmmc.c @@ -1425,6 +1425,11 @@ static void realtek_init_host(struct realtek_pci_sdmmc *host) mmc->caps = mmc->caps | MMC_CAP_AGGRESSIVE_PM; mmc->caps2 = MMC_CAP2_NO_PRESCAN_POWERUP | MMC_CAP2_FULL_PWR_CYCLE | MMC_CAP2_NO_SDIO; + + if (pcr->pci->device == 0x522a && + pcr->pci->subsystem_vendor == PCI_VENDOR_ID_LENOVO && + pcr->pci->subsystem_device == 0x504a) + mmc->caps2 |= MMC_CAP2_NO_WRITE_PROTECT; mmc->max_current_330 = 400; mmc->max_current_180 = 800; mmc->ops = &realtek_pci_sdmmc_ops; diff --git a/drivers/mmc/host/sdhci-of-aspeed.c b/drivers/mmc/host/sdhci-of-aspeed.c index f5d973783cbe..d317626feb4a 100644 --- a/drivers/mmc/host/sdhci-of-aspeed.c +++ b/drivers/mmc/host/sdhci-of-aspeed.c @@ -560,12 +560,14 @@ static int aspeed_sdc_probe(struct platform_device *pdev) cpdev = of_platform_device_create(child, NULL, &pdev->dev); if (!cpdev) { ret = -ENODEV; - goto err_clk; + goto err_children; } } return 0; +err_children: + device_for_each_child_reverse(&pdev->dev, NULL, of_platform_device_destroy); err_clk: clk_disable_unprepare(sdc->clk); return ret; @@ -575,6 +577,7 @@ static void aspeed_sdc_remove(struct platform_device *pdev) { struct aspeed_sdc *sdc = dev_get_drvdata(&pdev->dev); + device_for_each_child_reverse(&pdev->dev, NULL, of_platform_device_destroy); clk_disable_unprepare(sdc->clk); } diff --git a/drivers/mmc/host/sdhci_am654.c b/drivers/mmc/host/sdhci_am654.c index 2a27db2f558b..2e332f52c301 100644 --- a/drivers/mmc/host/sdhci_am654.c +++ b/drivers/mmc/host/sdhci_am654.c @@ -126,7 +126,7 @@ static const struct timing_data td[] = { NULL, MMC_CAP_UHS_SDR104}, [MMC_TIMING_UHS_DDR50] = {"ti,otap-del-sel-ddr50", - NULL, + "ti,itap-del-sel-ddr50", MMC_CAP_UHS_DDR50}, [MMC_TIMING_MMC_DDR52] = {"ti,otap-del-sel-ddr52", "ti,itap-del-sel-ddr52", @@ -144,6 +144,8 @@ struct sdhci_am654_data { u32 otap_del_sel[ARRAY_SIZE(td)]; u32 itap_del_sel[ARRAY_SIZE(td)]; u32 itap_del_ena[ARRAY_SIZE(td)]; + u32 itap_del_sel_dt_ddr50; + u32 itap_del_ena_dt_ddr50; int clkbuf_sel; int trm_icp; int drv_strength; @@ -151,7 +153,6 @@ struct sdhci_am654_data { u32 flags; u32 quirks; bool dll_enable; - u32 tuning_loop; #define SDHCI_AM654_QUIRK_FORCE_CDTEST BIT(0) #define SDHCI_AM654_QUIRK_SUPPRESS_V1P8_ENA BIT(1) @@ -443,15 +444,13 @@ static int sdhci_am654_execute_tuning(struct mmc_host *mmc, u32 opcode) struct sdhci_host *host = mmc_priv(mmc); int err = sdhci_execute_tuning(mmc, opcode); - if (err) - return err; /* * Tuning data remains in the buffer after tuning. * Do a command and data reset to get rid of it */ sdhci_reset(host, SDHCI_RESET_CMD | SDHCI_RESET_DATA); - return 0; + return err; } static u32 sdhci_am654_cqhci_irq(struct sdhci_host *host, u32 intmask) @@ -530,7 +529,6 @@ static int sdhci_am654_do_tuning(struct sdhci_host *host, { struct sdhci_pltfm_host *pltfm_host = sdhci_priv(host); struct sdhci_am654_data *sdhci_am654 = sdhci_pltfm_priv(pltfm_host); - unsigned char timing = host->mmc->ios.timing; struct window fail_window[ITAPDLY_LENGTH]; struct device *dev = mmc_dev(host->mmc); u8 curr_pass, itap; @@ -539,11 +537,8 @@ static int sdhci_am654_do_tuning(struct sdhci_host *host, memset(fail_window, 0, sizeof(fail_window)); - /* Enable ITAPDLY */ - sdhci_am654->itap_del_ena[timing] = 0x1; - for (itap = 0; itap < ITAPDLY_LENGTH; itap++) { - sdhci_am654_write_itapdly(sdhci_am654, itap, sdhci_am654->itap_del_ena[timing]); + sdhci_am654_write_itapdly(sdhci_am654, itap, 0x1); curr_pass = !mmc_send_tuning(host->mmc, opcode, NULL); @@ -576,20 +571,36 @@ static int sdhci_am654_platform_execute_tuning(struct sdhci_host *host, struct sdhci_am654_data *sdhci_am654 = sdhci_pltfm_priv(pltfm_host); unsigned char timing = host->mmc->ios.timing; struct device *dev = mmc_dev(host->mmc); + unsigned int tuning_loop = 0; int itapdly; do { itapdly = sdhci_am654_do_tuning(host, opcode); if (itapdly >= 0) break; - } while (++sdhci_am654->tuning_loop < RETRY_TUNING_MAX); + } while (++tuning_loop < RETRY_TUNING_MAX); if (itapdly < 0) { - dev_err(dev, "Failed to find itapdly, fail tuning\n"); + if (timing == MMC_TIMING_UHS_DDR50) { + dev_dbg(dev, "Failed DDR50 tuning, fallback to DT ITAP\n"); + sdhci_am654->itap_del_sel[timing] = sdhci_am654->itap_del_sel_dt_ddr50; + sdhci_am654->itap_del_ena[timing] = sdhci_am654->itap_del_ena_dt_ddr50; + } else { + dev_err(dev, "Failed to find itapdly, fail tuning\n"); + sdhci_am654->itap_del_ena[timing] = 0; + sdhci_am654->itap_del_sel[timing] = 0; + } + + sdhci_am654_write_itapdly(sdhci_am654, + sdhci_am654->itap_del_sel[timing], + sdhci_am654->itap_del_ena[timing]); return -1; } dev_dbg(dev, "Passed tuning, final itapdly=%d\n", itapdly); + + /* Enable ITAPDLY */ + sdhci_am654->itap_del_ena[timing] = 0x1; sdhci_am654_write_itapdly(sdhci_am654, itapdly, sdhci_am654->itap_del_ena[timing]); /* Save ITAPDLY */ sdhci_am654->itap_del_sel[timing] = itapdly; @@ -758,6 +769,11 @@ static int sdhci_am654_get_otap_delay(struct sdhci_host *host, } } + sdhci_am654->itap_del_sel_dt_ddr50 = + sdhci_am654->itap_del_sel[MMC_TIMING_UHS_DDR50]; + sdhci_am654->itap_del_ena_dt_ddr50 = + sdhci_am654->itap_del_ena[MMC_TIMING_UHS_DDR50]; + return 0; } @@ -806,9 +822,6 @@ static int sdhci_am654_init(struct sdhci_host *host) regmap_update_bits(sdhci_am654->base, CTL_CFG_3, TUNINGFORSDR50_MASK, TUNINGFORSDR50_MASK); - /* Use to re-execute tuning */ - sdhci_am654->tuning_loop = 0; - ret = sdhci_setup_host(host); if (ret) return ret; diff --git a/drivers/mmc/host/sh_mmcif.c b/drivers/mmc/host/sh_mmcif.c index 7706d01b149e..8bd9cdedafa4 100644 --- a/drivers/mmc/host/sh_mmcif.c +++ b/drivers/mmc/host/sh_mmcif.c @@ -1461,6 +1461,7 @@ static int sh_mmcif_probe(struct platform_device *pdev) host->pd = pdev; spin_lock_init(&host->lock); + mutex_init(&host->thread_lock); mmc->ops = &sh_mmcif_ops; sh_mmcif_init_ocr(host); @@ -1515,8 +1516,6 @@ static int sh_mmcif_probe(struct platform_device *pdev) goto err_clk; } - mutex_init(&host->thread_lock); - ret = mmc_add_host(mmc); if (ret < 0) goto err_clk; diff --git a/drivers/net/dsa/mxl862xx/mxl862xx.c b/drivers/net/dsa/mxl862xx/mxl862xx.c index cfa7e3e269a2..e05ad52cd297 100644 --- a/drivers/net/dsa/mxl862xx/mxl862xx.c +++ b/drivers/net/dsa/mxl862xx/mxl862xx.c @@ -685,10 +685,22 @@ static int mxl862xx_setup(struct dsa_switch *ds) if (ret) return ret; + ret = mxl862xx_setup_mdio(ds); + if (ret) + return ret; + schedule_delayed_work(&priv->stats_work, MXL862XX_STATS_POLL_INTERVAL); - return mxl862xx_setup_mdio(ds); + return 0; +} + +static void mxl862xx_teardown(struct dsa_switch *ds) +{ + struct mxl862xx_priv *priv = ds->priv; + + set_bit(MXL862XX_FLAG_WORK_STOPPED, &priv->flags); + disable_delayed_work_sync(&priv->stats_work); } static int mxl862xx_port_state(struct dsa_switch *ds, int port, bool enable) @@ -2047,9 +2059,7 @@ static void mxl862xx_get_stats64(struct dsa_switch *ds, int port, spin_unlock_bh(&priv->ports[port].stats_lock); - /* Trigger a fresh poll so the next read sees up-to-date counters. - * No-op if the work is already pending, running, or teardown started. - */ + /* Trigger a fresh poll so the next read sees up-to-date counters. */ if (!test_bit(MXL862XX_FLAG_WORK_STOPPED, &priv->flags)) schedule_delayed_work(&priv->stats_work, 0); } @@ -2057,6 +2067,7 @@ static void mxl862xx_get_stats64(struct dsa_switch *ds, int port, static const struct dsa_switch_ops mxl862xx_switch_ops = { .get_tag_protocol = mxl862xx_get_tag_protocol, .setup = mxl862xx_setup, + .teardown = mxl862xx_teardown, .port_setup = mxl862xx_port_setup, .port_teardown = mxl862xx_port_teardown, .phylink_get_caps = mxl862xx_phylink_get_caps, @@ -2131,7 +2142,6 @@ static int mxl862xx_probe(struct mdio_device *mdiodev) err = dsa_register_switch(ds); if (err) { set_bit(MXL862XX_FLAG_WORK_STOPPED, &priv->flags); - cancel_delayed_work_sync(&priv->stats_work); mxl862xx_host_shutdown(priv); for (i = 0; i < MXL862XX_MAX_PORTS; i++) cancel_work_sync(&priv->ports[i].host_flood_work); @@ -2152,7 +2162,6 @@ static void mxl862xx_remove(struct mdio_device *mdiodev) priv = ds->priv; set_bit(MXL862XX_FLAG_WORK_STOPPED, &priv->flags); - cancel_delayed_work_sync(&priv->stats_work); dsa_unregister_switch(ds); @@ -2181,7 +2190,7 @@ static void mxl862xx_shutdown(struct mdio_device *mdiodev) dsa_switch_shutdown(ds); set_bit(MXL862XX_FLAG_WORK_STOPPED, &priv->flags); - cancel_delayed_work_sync(&priv->stats_work); + disable_delayed_work_sync(&priv->stats_work); mxl862xx_host_shutdown(priv); diff --git a/drivers/net/ethernet/broadcom/genet/bcmgenet.c b/drivers/net/ethernet/broadcom/genet/bcmgenet.c index a2305e6428d1..b916080f4ff1 100644 --- a/drivers/net/ethernet/broadcom/genet/bcmgenet.c +++ b/drivers/net/ethernet/broadcom/genet/bcmgenet.c @@ -749,8 +749,17 @@ static void bcmgenet_hfb_init(struct bcmgenet_priv *priv) INIT_LIST_HEAD(&priv->rxnfc_rules[i].list); priv->rxnfc_rules[i].state = BCMGENET_RXNFC_STATE_UNUSED; } +} + +static void bcmgenet_hfb_restore(struct bcmgenet_priv *priv) +{ + struct bcmgenet_rxnfc_rule *rule; bcmgenet_hfb_clear(priv); + + list_for_each_entry(rule, &priv->rxnfc_list, list) + if (rule->state != BCMGENET_RXNFC_STATE_UNUSED) + bcmgenet_hfb_create_rxnfc_filter(priv, rule); } static int bcmgenet_begin(struct net_device *dev) @@ -3376,8 +3385,8 @@ static int bcmgenet_open(struct net_device *dev) bcmgenet_set_hw_addr(priv, dev->dev_addr); - /* HFB init */ - bcmgenet_hfb_init(priv); + /* Restore the filters, the MAC was reset above */ + bcmgenet_hfb_restore(priv); /* Reinitialize TDMA and RDMA and SW housekeeping */ ret = bcmgenet_init_dma(priv, true); @@ -4075,6 +4084,7 @@ static int bcmgenet_probe(struct platform_device *pdev) /* Mii wait queue */ init_waitqueue_head(&priv->wq); + bcmgenet_hfb_init(priv); INIT_WORK(&priv->bcmgenet_irq_work, bcmgenet_irq_task); priv->clk_wol = devm_clk_get_optional(&priv->pdev->dev, "enet-wol"); @@ -4272,10 +4282,7 @@ static int bcmgenet_resume(struct device *d) bcmgenet_set_hw_addr(priv, dev->dev_addr); /* Restore hardware filters */ - bcmgenet_hfb_clear(priv); - list_for_each_entry(rule, &priv->rxnfc_list, list) - if (rule->state != BCMGENET_RXNFC_STATE_UNUSED) - bcmgenet_hfb_create_rxnfc_filter(priv, rule); + bcmgenet_hfb_restore(priv); /* Reinitialize TDMA and RDMA and SW housekeeping */ ret = bcmgenet_init_dma(priv, false); diff --git a/drivers/net/ethernet/cadence/macb_ptp.c b/drivers/net/ethernet/cadence/macb_ptp.c index 6d9166389988..14ae57fa00cb 100644 --- a/drivers/net/ethernet/cadence/macb_ptp.c +++ b/drivers/net/ethernet/cadence/macb_ptp.c @@ -50,7 +50,12 @@ static int gem_tsu_get_time(struct ptp_clock_info *ptp, struct timespec64 *ts, spin_lock_irqsave(&bp->tsu_clk_lock, flags); ptp_read_system_prets(sts); + /* explicit barriers are needed because gem_readl() is relaxed */ + if (sts) + rmb(); first = gem_readl(bp, TN); + if (sts) + rmb(); ptp_read_system_postts(sts); secl = gem_readl(bp, TSL); sech = gem_readl(bp, TSH); @@ -62,7 +67,11 @@ static int gem_tsu_get_time(struct ptp_clock_info *ptp, struct timespec64 *ts, * (assume all done within 1s) */ ptp_read_system_prets(sts); + if (sts) + rmb(); ts->tv_nsec = gem_readl(bp, TN); + if (sts) + rmb(); ptp_read_system_postts(sts); secl = gem_readl(bp, TSL); sech = gem_readl(bp, TSH); diff --git a/drivers/net/ethernet/cortina/gemini.c b/drivers/net/ethernet/cortina/gemini.c index f08de623e6f7..2dd2fa801829 100644 --- a/drivers/net/ethernet/cortina/gemini.c +++ b/drivers/net/ethernet/cortina/gemini.c @@ -1798,7 +1798,7 @@ static irqreturn_t gmac_irq(int irq, void *data) if (val & (GMAC0_RX_OVERRUN_INT_BIT << (netdev->dev_id * 8))) { spin_lock(&geth->irq_lock); - writel(GMAC0_RXDERR_INT_BIT << (netdev->dev_id * 8), + writel(GMAC0_RX_OVERRUN_INT_BIT << (netdev->dev_id * 8), geth->base + GLOBAL_INTERRUPT_STATUS_4_REG); u64_stats_update_begin(&port->ir_stats_syncp); ++port->stats.rx_fifo_errors; diff --git a/drivers/net/ethernet/marvell/mvpp2/mvpp2_main.c b/drivers/net/ethernet/marvell/mvpp2/mvpp2_main.c index ccc24a1301f2..848ee655c6ea 100644 --- a/drivers/net/ethernet/marvell/mvpp2/mvpp2_main.c +++ b/drivers/net/ethernet/marvell/mvpp2/mvpp2_main.c @@ -5086,7 +5086,8 @@ static int mvpp2_change_mtu(struct net_device *dev, int mtu) netdev_warn(dev, "mtu %d too high, switching to shared buffers", mtu); mvpp2_bm_switch_buffers(priv, false); } - } else { + } else if (priv->hw_version >= MVPP22 && + mvpp2_get_nrxqs(priv) * 2 <= MVPP2_BM_MAX_POOLS) { bool jumbo = false; int i; diff --git a/drivers/net/ethernet/meta/fbnic/fbnic_txrx.c b/drivers/net/ethernet/meta/fbnic/fbnic_txrx.c index 401f8b8ae1ca..e7918d3f6aba 100644 --- a/drivers/net/ethernet/meta/fbnic/fbnic_txrx.c +++ b/drivers/net/ethernet/meta/fbnic/fbnic_txrx.c @@ -311,6 +311,29 @@ fbnic_rx_csum(u64 rcd, struct sk_buff *skb, struct fbnic_ring *rcq, } } +static void fbnic_tx_doorbell(struct fbnic_ring *ring, __le64 *meta) +{ + *meta |= cpu_to_le64(FBNIC_TWD_FLAG_REQ_COMPLETION); + ring->deferred_meta = -1; + + /* Force DMA writes to flush before writing to tail */ + dma_wmb(); + + writel(ring->tail, ring->doorbell); +} + +/* Packets handed to us with xmit_more set are left in the ring without a + * doorbell, and without a completion request, in the expectation that the + * packet ending the burst will ring for all of them. If that packet gets + * dropped instead we have to ring here, otherwise the descriptors sit in + * the ring until the next transmit, which may never come. + */ +static void fbnic_tx_flush_doorbell(struct fbnic_ring *ring) +{ + if (ring->deferred_meta >= 0) + fbnic_tx_doorbell(ring, &ring->desc[ring->deferred_meta]); +} + static bool fbnic_tx_map(struct fbnic_ring *ring, struct sk_buff *skb, __le64 *meta) { @@ -378,14 +401,10 @@ fbnic_tx_map(struct fbnic_ring *ring, struct sk_buff *skb, __le64 *meta) /* Verify there is room for another packet */ fbnic_maybe_stop_tx(skb->dev, ring, FBNIC_MAX_SKB_DESC); - if (fbnic_tx_sent_queue(skb, ring)) { - *meta |= cpu_to_le64(FBNIC_TWD_FLAG_REQ_COMPLETION); - - /* Force DMA writes to flush before writing to tail */ - dma_wmb(); - - writel(tail, ring->doorbell); - } + if (fbnic_tx_sent_queue(skb, ring)) + fbnic_tx_doorbell(ring, meta); + else + ring->deferred_meta = meta - ring->desc; return false; dma_error: @@ -425,8 +444,10 @@ fbnic_xmit_frame_ring(struct sk_buff *skb, struct fbnic_ring *ring) * otherwise try next time */ desc_needed = skb_shinfo(skb)->nr_frags + 10; - if (fbnic_maybe_stop_tx(skb->dev, ring, desc_needed)) + if (fbnic_maybe_stop_tx(skb->dev, ring, desc_needed)) { + fbnic_tx_flush_doorbell(ring); return NETDEV_TX_BUSY; + } *meta = cpu_to_le64(FBNIC_TWD_FLAG_DEST_MAC); @@ -447,6 +468,8 @@ fbnic_xmit_frame_ring(struct sk_buff *skb, struct fbnic_ring *ring) err_free: dev_kfree_skb_any(skb); err_count: + fbnic_tx_flush_doorbell(ring); + u64_stats_update_begin(&ring->stats.syncp); ring->stats.dropped++; u64_stats_update_end(&ring->stats.syncp); @@ -2491,6 +2514,7 @@ static void fbnic_enable_twq0(struct fbnic_ring *twq) fbnic_ring_wr32(twq, FBNIC_QUEUE_TWQ0_CTL, FBNIC_QUEUE_TWQ_CTL_RESET); twq->tail = 0; twq->head = 0; + twq->deferred_meta = -1; /* Store descriptor ring address and size */ fbnic_ring_wr32(twq, FBNIC_QUEUE_TWQ0_BAL, lower_32_bits(twq->dma)); diff --git a/drivers/net/ethernet/meta/fbnic/fbnic_txrx.h b/drivers/net/ethernet/meta/fbnic/fbnic_txrx.h index e03c9d2c38dc..f5899446dcc5 100644 --- a/drivers/net/ethernet/meta/fbnic/fbnic_txrx.h +++ b/drivers/net/ethernet/meta/fbnic/fbnic_txrx.h @@ -128,9 +128,14 @@ struct fbnic_ring { /* Rx BDQs only */ struct page_pool *page_pool; - /* Deferred_head is used to cache the head for TWQ1 if + /* TWQ0 only, index of the meta descriptor of the last packet + * placed in the ring without ringing the doorbell, -1 if the + * doorbell is in sync with the tail. + */ + s32 deferred_meta; + + /* TCQ only, used to cache the head for TWQ1 if * an attempt is made to clean TWQ1 with zero napi_budget. - * We do not use it for any other ring. */ s32 deferred_head; }; diff --git a/drivers/net/ethernet/microchip/lan743x_main.c b/drivers/net/ethernet/microchip/lan743x_main.c index 24ae56a3c9ed..82d3ec20ba25 100644 --- a/drivers/net/ethernet/microchip/lan743x_main.c +++ b/drivers/net/ethernet/microchip/lan743x_main.c @@ -2604,7 +2604,7 @@ process_extension: rx->adapter->netdev); if (rx->adapter->netdev->features & NETIF_F_RXCSUM) { if (!is_ice && !is_tce && !is_icsm) - skb->ip_summed = CHECKSUM_UNNECESSARY; + rx->skb_head->ip_summed = CHECKSUM_UNNECESSARY; } netdev_dbg(netdev, "sending %d byte frame to OS", rx->skb_head->len); diff --git a/drivers/net/ethernet/socionext/netsec.c b/drivers/net/ethernet/socionext/netsec.c index d14a6584473c..79a0a324c921 100644 --- a/drivers/net/ethernet/socionext/netsec.c +++ b/drivers/net/ethernet/socionext/netsec.c @@ -2149,6 +2149,7 @@ pm_disable: pm_runtime_put_sync(&pdev->dev); pm_runtime_disable(&pdev->dev); free_ndev: + of_node_put(priv->phy_np); free_netdev(ndev); dev_err(&pdev->dev, "init failed\n"); @@ -2166,6 +2167,7 @@ static void netsec_remove(struct platform_device *pdev) netif_napi_del(&priv->napi); pm_runtime_disable(&pdev->dev); + of_node_put(priv->phy_np); free_netdev(priv->ndev); } diff --git a/drivers/net/ethernet/stmicro/stmmac/hwif.h b/drivers/net/ethernet/stmicro/stmmac/hwif.h index 04dafec021b4..9314bcb85c22 100644 --- a/drivers/net/ethernet/stmicro/stmmac/hwif.h +++ b/drivers/net/ethernet/stmicro/stmmac/hwif.h @@ -494,7 +494,7 @@ struct stmmac_ops { #define stmmac_set_arp_offload(__priv, __args...) \ stmmac_do_void_callback(__priv, mac, set_arp_offload, __args) #define stmmac_fpe_map_preemption_class(__priv, __args...) \ - stmmac_do_void_callback(__priv, mac, fpe_map_preemption_class, __args) + stmmac_do_callback(__priv, mac, fpe_map_preemption_class, __args) /* PTP and HW Timer helpers */ struct stmmac_hwtimestamp { diff --git a/drivers/net/ethernet/stmicro/stmmac/stmmac_ethtool.c b/drivers/net/ethernet/stmicro/stmmac/stmmac_ethtool.c index 154cc0c7623d..1be5310ca766 100644 --- a/drivers/net/ethernet/stmicro/stmmac/stmmac_ethtool.c +++ b/drivers/net/ethernet/stmicro/stmmac/stmmac_ethtool.c @@ -1016,8 +1016,6 @@ static int stmmac_get_ts_info(struct net_device *dev, if (priv->ptp_clock) info->phc_index = ptp_clock_index(priv->ptp_clock); - else - info->phc_index = 0; info->tx_types = (1 << HWTSTAMP_TX_OFF) | (1 << HWTSTAMP_TX_ON); diff --git a/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c b/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c index 62c3441911e7..1fb5f804ea23 100644 --- a/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c +++ b/drivers/net/ethernet/stmicro/stmmac/stmmac_main.c @@ -4513,16 +4513,16 @@ static int stmmac_tso_get_num_desc(struct stmmac_tx_queue *tx_q, */ static netdev_tx_t stmmac_tso_xmit(struct sk_buff *skb, struct net_device *dev) { + unsigned int first_entry, entry, tx_packets, proto_hdr_len; struct dma_desc *desc, *first, *mss_desc = NULL; struct stmmac_priv *priv = netdev_priv(dev); - unsigned int first_entry, entry, tx_packets; struct stmmac_txq_stats *txq_stats; int i, first_tx, nfrags, ndesc; struct stmmac_tx_queue *tx_q; bool set_ic, is_last_segment; u32 pay_len, mss, queue; - u8 proto_hdr_len, hdr; dma_addr_t des; + u8 hdr; nfrags = skb_shinfo(skb)->nr_frags; queue = skb_get_queue_mapping(skb); @@ -4570,7 +4570,7 @@ static netdev_tx_t stmmac_tso_xmit(struct sk_buff *skb, struct net_device *dev) } if (netif_msg_tx_queued(priv)) { - pr_info("%s: hdrlen %d, hdr_len %d, pay_len %d, mss %d\n", + pr_info("%s: hdrlen %d, hdr_len %u, pay_len %d, mss %d\n", __func__, hdr, proto_hdr_len, pay_len, mss); pr_info("\tskb->len %d, skb->data_len %d\n", skb->len, skb->data_len); diff --git a/drivers/net/ethernet/stmicro/stmmac/stmmac_tc.c b/drivers/net/ethernet/stmicro/stmmac/stmmac_tc.c index 14cabe76e53e..42a00446e9b4 100644 --- a/drivers/net/ethernet/stmicro/stmmac/stmmac_tc.c +++ b/drivers/net/ethernet/stmicro/stmmac/stmmac_tc.c @@ -970,7 +970,7 @@ static int tc_taprio_configure(struct stmmac_priv *priv, struct netlink_ext_ack *extack = qopt->mqprio.extack; struct timespec64 time, current_time, qopt_time; ktime_t current_time_ns; - int i, ret = 0; + int err, i, ret = 0; u64 ctr; if (qopt->base_time < 0) @@ -1120,9 +1120,9 @@ disable: mutex_unlock(&priv->est_lock); } - stmmac_fpe_map_preemption_class(priv, priv->dev, extack, 0); + err = stmmac_fpe_map_preemption_class(priv, priv->dev, extack, 0); - return ret; + return qopt->cmd == TAPRIO_CMD_DESTROY ? err : ret; } static void tc_taprio_stats(struct stmmac_priv *priv, @@ -1237,58 +1237,99 @@ static int tc_query_caps(struct stmmac_priv *priv, } } -static void stmmac_reset_tc_mqprio(struct net_device *ndev, - struct netlink_ext_ack *extack) +static int stmmac_set_ndev_tcs(struct net_device *ndev, u8 ntc, + struct netdev_tc_txq *tc_to_txq) +{ + int i, err; + + netdev_reset_tc(ndev); + if (!ntc) + return 0; + + err = netdev_set_num_tc(ndev, ntc); + if (err) + return err; + + for (i = 0; i < ntc; i++) { + u16 count, offset; + + count = tc_to_txq[i].count; + offset = tc_to_txq[i].offset; + netdev_set_tc_queue(ndev, i, count, offset); + } + + return 0; +} + +static int stmmac_reset_tc_mqprio(struct net_device *ndev, + struct netlink_ext_ack *extack) { struct stmmac_priv *priv = netdev_priv(ndev); netdev_reset_tc(ndev); netif_set_real_num_tx_queues(ndev, priv->plat->tx_queues_to_use); - stmmac_fpe_map_preemption_class(priv, ndev, extack, 0); + + return stmmac_fpe_map_preemption_class(priv, ndev, extack, 0); } static int tc_setup_dwmac510_mqprio(struct stmmac_priv *priv, struct tc_mqprio_qopt_offload *mqprio) { + unsigned int ndev_num_tx_queues, num_tx_queues = 0; + struct netdev_tc_txq ndev_tc_to_txq[TC_MAX_QUEUE]; + struct netdev_tc_txq tc_to_txq[TC_MAX_QUEUE] = {}; struct netlink_ext_ack *extack = mqprio->extack; struct tc_mqprio_qopt *qopt = &mqprio->qopt; - u32 offset, count, num_stack_tx_queues = 0; struct net_device *ndev = priv->dev; - u32 num_tc = qopt->num_tc; - int err; - - if (!num_tc) { - stmmac_reset_tc_mqprio(ndev, extack); - return 0; - } + u8 ndev_prio_tc_map[TC_BITMASK + 1]; + int i, err, ndev_ntc; - err = netdev_set_num_tc(ndev, num_tc); - if (err) - return err; + if (!qopt->num_tc) + return stmmac_reset_tc_mqprio(ndev, extack); - for (u32 tc = 0; tc < num_tc; tc++) { - offset = qopt->offset[tc]; - count = qopt->count[tc]; - num_stack_tx_queues += count; + if (qopt->num_tc > ARRAY_SIZE(tc_to_txq)) + return -EINVAL; - err = netdev_set_tc_queue(ndev, tc, count, offset); - if (err) - goto err_reset_tc; + /* save current tc values for reset */ + ndev_ntc = netdev_get_num_tc(ndev); + for (i = 0; i < ARRAY_SIZE(ndev->tc_to_txq); i++) + ndev_tc_to_txq[i].combined = + READ_ONCE(ndev->tc_to_txq[i].combined); + for (i = 0; i < ARRAY_SIZE(ndev_prio_tc_map); i++) + ndev_prio_tc_map[i] = READ_ONCE(ndev->prio_tc_map[i]); + + for (i = 0; i < qopt->num_tc; i++) { + tc_to_txq[i] = (struct netdev_tc_txq) { + .count = qopt->count[i], + .offset = qopt->offset[i], + }; + num_tx_queues += qopt->count[i]; } - err = netif_set_real_num_tx_queues(ndev, num_stack_tx_queues); + err = stmmac_set_ndev_tcs(ndev, qopt->num_tc, tc_to_txq); + if (err) + goto error_reset_tc; + + ndev_num_tx_queues = ndev->real_num_tx_queues; + err = netif_set_real_num_tx_queues(ndev, num_tx_queues); if (err) - goto err_reset_tc; + goto error_reset_tc; err = stmmac_fpe_map_preemption_class(priv, ndev, extack, mqprio->preemptible_tcs); if (err) - goto err_reset_tc; + goto error_reset_num_tx_queues; return 0; -err_reset_tc: - stmmac_reset_tc_mqprio(ndev, extack); +error_reset_num_tx_queues: + if (netif_set_real_num_tx_queues(ndev, ndev_num_tx_queues)) + netdev_warn(ndev, "Failed to restore %u TX queues\n", + ndev_num_tx_queues); +error_reset_tc: + stmmac_set_ndev_tcs(ndev, ndev_ntc, ndev_tc_to_txq); + for (i = 0; i < ARRAY_SIZE(ndev_prio_tc_map); i++) + netdev_set_prio_tc_map(ndev, i, ndev_prio_tc_map[i]); return err; } diff --git a/drivers/net/fddi/skfp/skfddi.c b/drivers/net/fddi/skfp/skfddi.c index a273362c9e70..feea7baa4816 100644 --- a/drivers/net/fddi/skfp/skfddi.c +++ b/drivers/net/fddi/skfp/skfddi.c @@ -928,7 +928,8 @@ static int skfp_ctl_set_mac_address(struct net_device *dev, void *addr) dev_addr_set(dev, p_sockaddr->sa_data); spin_lock_irqsave(&bp->DriverLock, Flags); - ResetAdapter(smc); + if (netif_running(dev)) + ResetAdapter(smc); spin_unlock_irqrestore(&bp->DriverLock, Flags); return 0; /* always return zero */ diff --git a/drivers/net/phy/mediatek/mtk-phy-lib.c b/drivers/net/phy/mediatek/mtk-phy-lib.c index dfd0f4e439a2..608072fbfde9 100644 --- a/drivers/net/phy/mediatek/mtk-phy-lib.c +++ b/drivers/net/phy/mediatek/mtk-phy-lib.c @@ -156,20 +156,27 @@ int mtk_phy_led_hw_ctrl_get(struct phy_device *phydev, u8 index, if (!rules) return 0; - if (on & on_set) + /* TRIGGER_NETDEV_LINK must not be reported together with any of the + * per-speed rules, the netdev trigger rejects that combination. + * on_set holds every speed this LED can indicate and is what + * mtk_phy_led_hw_ctrl_set() programs for TRIGGER_NETDEV_LINK, so + * report the speed independent rule only when they are all on. + */ + if ((on & on_set) == on_set) { *rules |= BIT(TRIGGER_NETDEV_LINK); + } else { + if (on & MTK_PHY_LED_ON_LINK10) + *rules |= BIT(TRIGGER_NETDEV_LINK_10); - if (on & MTK_PHY_LED_ON_LINK10) - *rules |= BIT(TRIGGER_NETDEV_LINK_10); + if (on & MTK_PHY_LED_ON_LINK100) + *rules |= BIT(TRIGGER_NETDEV_LINK_100); - if (on & MTK_PHY_LED_ON_LINK100) - *rules |= BIT(TRIGGER_NETDEV_LINK_100); + if (on & MTK_PHY_LED_ON_LINK1000) + *rules |= BIT(TRIGGER_NETDEV_LINK_1000); - if (on & MTK_PHY_LED_ON_LINK1000) - *rules |= BIT(TRIGGER_NETDEV_LINK_1000); - - if (on & MTK_PHY_LED_ON_LINK2500) - *rules |= BIT(TRIGGER_NETDEV_LINK_2500); + if (on & MTK_PHY_LED_ON_LINK2500) + *rules |= BIT(TRIGGER_NETDEV_LINK_2500); + } if (on & MTK_PHY_LED_ON_FDX) *rules |= BIT(TRIGGER_NETDEV_FULL_DUPLEX); diff --git a/drivers/net/wireless/ath/ath11k/mac.c b/drivers/net/wireless/ath/ath11k/mac.c index 2d55cdc4d165..ae91b57c8422 100644 --- a/drivers/net/wireless/ath/ath11k/mac.c +++ b/drivers/net/wireless/ath/ath11k/mac.c @@ -873,6 +873,22 @@ static int ath11k_mac_set_kickout(struct ath11k_vif *arvif) return 0; } +static void ath11k_mac_station_cleanup(struct ieee80211_sta *sta) +{ + struct ath11k_sta *arsta; + + if (!sta) + return; + + arsta = ath11k_sta_to_arsta(sta); + + kfree(arsta->tx_stats); + arsta->tx_stats = NULL; + + kfree(arsta->rx_stats); + arsta->rx_stats = NULL; +} + void ath11k_mac_peer_cleanup_all(struct ath11k *ar) { struct ath11k_peer *peer, *tmp; @@ -885,6 +901,7 @@ void ath11k_mac_peer_cleanup_all(struct ath11k *ar) list_for_each_entry_safe(peer, tmp, &ab->peers, list) { ath11k_peer_rx_tid_cleanup(ar, peer); ath11k_peer_rhash_delete(ab, peer); + ath11k_mac_station_cleanup(peer->sta); list_del(&peer->list); kfree(peer); } @@ -9892,7 +9909,6 @@ static int ath11k_mac_station_remove(struct ath11k *ar, { struct ath11k_base *ab = ar->ab; struct ath11k_vif *arvif = ath11k_vif_to_arvif(vif); - struct ath11k_sta *arsta = ath11k_sta_to_arsta(sta); int ret; if (ab->hw_params.vdev_start_delay && @@ -9916,12 +9932,7 @@ static int ath11k_mac_station_remove(struct ath11k *ar, sta->addr, arvif->vdev_id); ath11k_mac_dec_num_stations(arvif, sta); - - kfree(arsta->tx_stats); - arsta->tx_stats = NULL; - - kfree(arsta->rx_stats); - arsta->rx_stats = NULL; + ath11k_mac_station_cleanup(sta); return ret; } diff --git a/drivers/net/wireless/ath/ath12k/wifi7/ahb.c b/drivers/net/wireless/ath/ath12k/wifi7/ahb.c index 98a6606ffd76..6e9e9034cba1 100644 --- a/drivers/net/wireless/ath/ath12k/wifi7/ahb.c +++ b/drivers/net/wireless/ath/ath12k/wifi7/ahb.c @@ -15,21 +15,6 @@ #include "dp.h" #include "core.h" -/* - * Node name to UserPD ID mapping - * - * The io_start field is used for additional validation when the reg - * property is present in the device tree. If io_start is 0, only - * node_name matching is performed. - * - * For platforms where not all WiFi nodes have a 'reg' property, set - * io_start to 0 for those entries. The driver will match purely by - * node name in such cases. - */ -static const struct ath12k_ahb_userpd_map ath12k_wifi7_ahb_userpd_map[] = { - { .io_start = 0x0c000000, .node_name = "wifi", .upd_id = ATH12K_AHB_USERPD_ID_0 }, -}; - static const struct ath12k_ahb_desc ath12k_wifi7_ahb_desc[] = { [ATH12K_HW_IPQ5332_HW10] = { .hw_rev = ATH12K_HW_IPQ5332_HW10, @@ -55,40 +40,6 @@ static const struct of_device_id ath12k_wifi7_ahb_of_match[] = { MODULE_DEVICE_TABLE(of, ath12k_wifi7_ahb_of_match); -/* - * ath12k_wifi7_ahb_get_userpd_id - Resolve UserPD ID from DT properties - * @ab: ath12k base structure - * - * Returns: UserPD ID (1-based) on success, 0 on failure - * - * Resolution logic: - * 1. If reg property exist in DT, get userpd_id from io_start - * 2. If reg property is absent, get userpd_id from DT node name - * 3. Return 0 if no match found (probe will fail) - */ -static u32 ath12k_wifi7_ahb_get_userpd_id(struct ath12k_base *ab) -{ - const struct ath12k_ahb_userpd_map *map; - struct resource *res; - size_t i; - - res = platform_get_resource(ab->pdev, IORESOURCE_MEM, 0); - - for (i = 0; i < ARRAY_SIZE(ath12k_wifi7_ahb_userpd_map); i++) { - map = &ath12k_wifi7_ahb_userpd_map[i]; - - if (res) { - if (map->io_start && map->io_start == res->start) - return map->upd_id; - } else if (map->node_name && - of_node_name_eq(ab->dev->of_node, map->node_name)) { - return map->upd_id; - } - } - - return 0; -} - static int ath12k_wifi7_ahb_probe(struct platform_device *pdev) { const struct ath12k_ahb_desc *desc; @@ -106,7 +57,7 @@ static int ath12k_wifi7_ahb_probe(struct platform_device *pdev) ab->hw_rev = desc->hw_rev; ab->hif.ops = desc->ops; ab_ahb->scm_auth_enabled = desc->auth_enabled; - ab_ahb->userpd_id = ath12k_wifi7_ahb_get_userpd_id(ab); + ab_ahb->userpd_id = ATH12K_AHB_USERPD_ID_0; if (!ab_ahb->userpd_id) return -EOPNOTSUPP; diff --git a/drivers/net/wireless/ath/wcn36xx/dxe.c b/drivers/net/wireless/ath/wcn36xx/dxe.c index 44020ec265fb..801f1218ef89 100644 --- a/drivers/net/wireless/ath/wcn36xx/dxe.c +++ b/drivers/net/wireless/ath/wcn36xx/dxe.c @@ -1055,7 +1055,7 @@ void wcn36xx_dxe_deinit(struct wcn36xx *wcn) free_irq(wcn->tx_irq, wcn); free_irq(wcn->rx_irq, wcn); - timer_delete(&wcn->tx_ack_timer); + timer_shutdown_sync(&wcn->tx_ack_timer); if (wcn->tx_ack_skb) { ieee80211_tx_status_irqsafe(wcn->hw, wcn->tx_ack_skb); diff --git a/drivers/net/wireless/broadcom/brcm80211/brcmfmac/core.c b/drivers/net/wireless/broadcom/brcm80211/brcmfmac/core.c index dad6f4563d14..d2ae67985606 100644 --- a/drivers/net/wireless/broadcom/brcm80211/brcmfmac/core.c +++ b/drivers/net/wireless/broadcom/brcm80211/brcmfmac/core.c @@ -555,6 +555,8 @@ void brcmf_txfinalize(struct brcmf_if *ifp, struct sk_buff *txp, bool success) if (type == ETH_P_PAE) { atomic_dec(&ifp->pend_8021x_cnt); + /* Order the decrement before waitqueue_active() */ + smp_mb__after_atomic(); if (waitqueue_active(&ifp->pend_8021x_wait)) wake_up(&ifp->pend_8021x_wait); } diff --git a/drivers/net/wireless/broadcom/brcm80211/brcmfmac/cyw/core.c b/drivers/net/wireless/broadcom/brcm80211/brcmfmac/cyw/core.c index 545eb9aae966..6d5098f6f00f 100644 --- a/drivers/net/wireless/broadcom/brcm80211/brcmfmac/cyw/core.c +++ b/drivers/net/wireless/broadcom/brcm80211/brcmfmac/cyw/core.c @@ -198,7 +198,7 @@ brcmf_cyw_external_auth(struct wiphy *wiphy, struct net_device *dev, { struct brcmf_if *ifp; struct brcmf_pub *drvr; - struct brcmf_auth_req_status_le auth_status; + struct brcmf_auth_req_status_le auth_status = {}; int ret = 0; brcmf_dbg(TRACE, "Enter\n"); @@ -206,6 +206,9 @@ brcmf_cyw_external_auth(struct wiphy *wiphy, struct net_device *dev, ifp = netdev_priv(dev); drvr = ifp->drvr; if (params->status == WLAN_STATUS_SUCCESS) { + if (params->pmkid) + memcpy(auth_status.pmkid, params->pmkid, + WLAN_PMKID_LEN); auth_status.flags = cpu_to_le16(BRCMF_EXTAUTH_SUCCESS); } else { bphy_err(drvr, "External authentication failed: status=%d\n", diff --git a/drivers/net/wireless/broadcom/brcm80211/brcmsmac/mac80211_if.c b/drivers/net/wireless/broadcom/brcm80211/brcmsmac/mac80211_if.c index 6255d673d2d3..c1a2318d7ea6 100644 --- a/drivers/net/wireless/broadcom/brcm80211/brcmsmac/mac80211_if.c +++ b/drivers/net/wireless/broadcom/brcm80211/brcmsmac/mac80211_if.c @@ -1571,6 +1571,10 @@ void brcms_free_timer(struct brcms_timer *t) /* delete the timer in case it is active */ brcms_del_timer(t); + /* Ensure the callback has finished before freeing the timer + * structure, since brcms_del_timer() uses non-synchronous cancel. + */ + cancel_delayed_work_sync(&t->dly_wrk); if (wl->timers == t) { wl->timers = wl->timers->next; diff --git a/drivers/net/wireless/intel/ipw2x00/ipw2100.c b/drivers/net/wireless/intel/ipw2x00/ipw2100.c index 2b8a23865bfb..43b4e432956b 100644 --- a/drivers/net/wireless/intel/ipw2x00/ipw2100.c +++ b/drivers/net/wireless/intel/ipw2x00/ipw2100.c @@ -2712,7 +2712,9 @@ static void __ipw2100_rx_process(struct ipw2100_priv *priv) break; } #endif - if (stats.len < sizeof(struct libipw_hdr_3addr)) + if (sq->drv[i].frame_size < + sizeof(struct libipw_hdr_3addr) || + sq->drv[i].frame_size > IPW_RX_NIC_BUFFER_LENGTH) break; switch (WLAN_FC_GET_TYPE(le16_to_cpu(u->rx_data.header.frame_ctl))) { case IEEE80211_FTYPE_MGMT: diff --git a/drivers/net/wireless/intel/ipw2x00/ipw2200.c b/drivers/net/wireless/intel/ipw2x00/ipw2200.c index 4bc9bb406e8e..8249d493ee22 100644 --- a/drivers/net/wireless/intel/ipw2x00/ipw2200.c +++ b/drivers/net/wireless/intel/ipw2x00/ipw2200.c @@ -8322,6 +8322,15 @@ static void ipw_rx(struct ipw_priv *priv) break; } + if (unlikely(le16_to_cpu(pkt->u.frame.length) > + IPW_RX_BUF_SIZE - + IPW_RX_FRAME_SIZE)) { + IPW_DEBUG_DROP("Received oversized packet. Dropping.\n"); + priv->net_dev->stats.rx_errors++; + priv->wstats.discard.misc++; + break; + } + switch (WLAN_FC_GET_TYPE (le16_to_cpu(header->frame_ctl))) { diff --git a/drivers/net/wireless/intel/ipw2x00/libipw_crypto_tkip.c b/drivers/net/wireless/intel/ipw2x00/libipw_crypto_tkip.c index 24bb28ab7a49..2b0cf0ec496a 100644 --- a/drivers/net/wireless/intel/ipw2x00/libipw_crypto_tkip.c +++ b/drivers/net/wireless/intel/ipw2x00/libipw_crypto_tkip.c @@ -474,14 +474,16 @@ static int libipw_michael_mic_verify(struct sk_buff *skb, int keyidx, int hdr_len, void *priv) { struct libipw_tkip_data *tkey = priv; - u8 mic[8]; + u8 mic[MICHAEL_MIC_LEN]; - if (!tkey->key_set) + if (!tkey->key_set || skb->len < hdr_len + MICHAEL_MIC_LEN) return -1; michael_mic(&tkey->key[24], (struct ieee80211_hdr *)skb->data, - skb->data + hdr_len, skb->len - 8 - hdr_len, mic); - if (memcmp(mic, skb->data + skb->len - 8, 8) != 0) { + skb->data + hdr_len, + skb->len - MICHAEL_MIC_LEN - hdr_len, mic); + if (memcmp(mic, skb->data + skb->len - MICHAEL_MIC_LEN, + MICHAEL_MIC_LEN) != 0) { struct ieee80211_hdr *hdr; hdr = (struct ieee80211_hdr *)skb->data; printk(KERN_DEBUG "%s: Michael MIC verification failed for " @@ -499,7 +501,7 @@ static int libipw_michael_mic_verify(struct sk_buff *skb, int keyidx, tkey->rx_iv32 = tkey->rx_iv32_new; tkey->rx_iv16 = tkey->rx_iv16_new; - skb_trim(skb, skb->len - 8); + skb_trim(skb, skb->len - MICHAEL_MIC_LEN); return 0; } diff --git a/drivers/net/wireless/intel/ipw2x00/libipw_rx.c b/drivers/net/wireless/intel/ipw2x00/libipw_rx.c index c8841f9b9ad9..424349a6935e 100644 --- a/drivers/net/wireless/intel/ipw2x00/libipw_rx.c +++ b/drivers/net/wireless/intel/ipw2x00/libipw_rx.c @@ -1209,6 +1209,9 @@ static int libipw_handle_assoc_resp(struct libipw_device *ieee, struct libipw_as struct libipw_network *network = &network_resp; struct net_device *dev = ieee->dev; + if (stats->len < sizeof(*frame)) + return 1; + network->flags = 0; network->qos_data.active = 0; network->qos_data.supported = 0; @@ -1421,6 +1424,9 @@ static void libipw_process_probe_response(struct libipw_device #endif unsigned long flags; + if (stats->len < sizeof(*beacon)) + return; + LIBIPW_DEBUG_SCAN("'%*pE' (%pM): %c%c%c%c %c%c%c%c-%c%c%c%c %c%c%c%c\n", info_element->len, info_element->data, beacon->header.addr3, diff --git a/drivers/net/wireless/intel/iwlegacy/common.c b/drivers/net/wireless/intel/iwlegacy/common.c index 0bb807ff8edf..e5113c6b2d5c 100644 --- a/drivers/net/wireless/intel/iwlegacy/common.c +++ b/drivers/net/wireless/intel/iwlegacy/common.c @@ -2326,7 +2326,7 @@ il_dealloc_bcast_stations(struct il_priv *il) if (!(il->stations[i].used & IL_STA_BCAST)) continue; - il->stations[i].used &= ~IL_STA_UCODE_ACTIVE; + il->stations[i].used = 0; il->num_stations--; if (WARN_ON(il->num_stations < 0)) il->num_stations = 0; diff --git a/drivers/net/wireless/intersil/p54/eeprom.c b/drivers/net/wireless/intersil/p54/eeprom.c index 95580921d933..0475222d54fc 100644 --- a/drivers/net/wireless/intersil/p54/eeprom.c +++ b/drivers/net/wireless/intersil/p54/eeprom.c @@ -414,17 +414,22 @@ free: } static int p54_convert_rev0(struct ieee80211_hw *dev, - struct pda_pa_curve_data *curve_data) + struct pda_pa_curve_data *curve_data, size_t len) { struct p54_common *priv = dev->priv; struct p54_pa_curve_data_sample *dst; struct pda_pa_curve_data_sample_rev0 *src; + size_t needed = curve_data->channels * + (sizeof(*src) * curve_data->points_per_channel + 2); size_t cd_len = sizeof(*curve_data) + (curve_data->points_per_channel*sizeof(*dst) + 2) * curve_data->channels; unsigned int i, j; void *source, *target; + if (len < sizeof(*curve_data) + needed) + return -EINVAL; + priv->curve_data = kmalloc(sizeof(*priv->curve_data) + cd_len, GFP_KERNEL); if (!priv->curve_data) @@ -466,17 +471,22 @@ static int p54_convert_rev0(struct ieee80211_hw *dev, } static int p54_convert_rev1(struct ieee80211_hw *dev, - struct pda_pa_curve_data *curve_data) + struct pda_pa_curve_data *curve_data, size_t len) { struct p54_common *priv = dev->priv; struct p54_pa_curve_data_sample *dst; struct pda_pa_curve_data_sample_rev1 *src; + size_t needed = curve_data->channels * + (sizeof(*src) * curve_data->points_per_channel + 3); size_t cd_len = sizeof(*curve_data) + (curve_data->points_per_channel*sizeof(*dst) + 2) * curve_data->channels; unsigned int i, j; void *source, *target; + if (len < sizeof(*curve_data) + needed) + return -EINVAL; + priv->curve_data = kzalloc(cd_len + sizeof(*priv->curve_data), GFP_KERNEL); if (!priv->curve_data) @@ -763,6 +773,7 @@ int p54_parse_eeprom(struct ieee80211_hw *dev, void *eeprom, int len) case PDR_PRISM_PA_CAL_CURVE_DATA: { struct pda_pa_curve_data *curve_data = (struct pda_pa_curve_data *)entry->data; + if (data_len < sizeof(*curve_data)) { err = -EINVAL; goto err; @@ -770,10 +781,10 @@ int p54_parse_eeprom(struct ieee80211_hw *dev, void *eeprom, int len) switch (curve_data->cal_method_rev) { case 0: - err = p54_convert_rev0(dev, curve_data); + err = p54_convert_rev0(dev, curve_data, data_len); break; case 1: - err = p54_convert_rev1(dev, curve_data); + err = p54_convert_rev1(dev, curve_data, data_len); break; default: wiphy_err(dev->wiphy, @@ -801,7 +812,8 @@ int p54_parse_eeprom(struct ieee80211_hw *dev, void *eeprom, int len) break; case PDR_INTERFACE_LIST: tmp = entry->data; - while ((u8 *)tmp < entry->data + data_len) { + while ((u8 *)tmp + sizeof(struct exp_if) <= + entry->data + data_len) { struct exp_if *exp_if = tmp; if (exp_if->if_id == cpu_to_le16(IF_ID_ISL39000)) synth = le16_to_cpu(exp_if->variant); diff --git a/drivers/net/wireless/marvell/libertas_tf/main.c b/drivers/net/wireless/marvell/libertas_tf/main.c index 42be6fa22f9c..411f075b6186 100644 --- a/drivers/net/wireless/marvell/libertas_tf/main.c +++ b/drivers/net/wireless/marvell/libertas_tf/main.c @@ -173,8 +173,8 @@ static int lbtf_init_adapter(struct lbtf_private *priv) static void lbtf_free_adapter(struct lbtf_private *priv) { lbtf_deb_enter(LBTF_DEB_MAIN); - lbtf_free_cmd_buffer(priv); timer_delete_sync(&priv->command_timer); + lbtf_free_cmd_buffer(priv); lbtf_deb_leave(LBTF_DEB_MAIN); } diff --git a/drivers/net/wireless/marvell/mwifiex/cfg80211.c b/drivers/net/wireless/marvell/mwifiex/cfg80211.c index 7a1ba32f1fb3..936939ea9c47 100644 --- a/drivers/net/wireless/marvell/mwifiex/cfg80211.c +++ b/drivers/net/wireless/marvell/mwifiex/cfg80211.c @@ -4277,6 +4277,7 @@ mwifiex_cfg80211_authenticate(struct wiphy *wiphy, struct mwifiex_adapter *adapter = priv->adapter; struct sk_buff *skb; u16 pkt_len, auth_alg; + size_t frame_len; int ret; struct mwifiex_ieee80211_mgmt *mgmt; struct mwifiex_txinfo *tx_info; @@ -4349,10 +4350,17 @@ mwifiex_cfg80211_authenticate(struct wiphy *wiphy, mwifiex_cancel_scan(adapter); - pkt_len = (u16)req->ie_len + req->auth_data_len + + frame_len = req->ie_len + req->auth_data_len + MWIFIEX_MGMT_HEADER_LEN + MWIFIEX_AUTH_BODY_LEN; if (req->auth_data_len >= 4) - pkt_len -= 4; + frame_len -= 4; + + if (frame_len > U16_MAX) { + mwifiex_dbg(priv->adapter, ERROR, + "auth frame too long: %zu bytes\n", frame_len); + return -EINVAL; + } + pkt_len = frame_len; skb = dev_alloc_skb(MWIFIEX_MIN_DATA_HEADER_LEN + MWIFIEX_MGMT_FRAME_HEADER_SIZE + diff --git a/drivers/net/wireless/marvell/mwifiex/pcie.c b/drivers/net/wireless/marvell/mwifiex/pcie.c index a760de191fce..a9425e9a94f4 100644 --- a/drivers/net/wireless/marvell/mwifiex/pcie.c +++ b/drivers/net/wireless/marvell/mwifiex/pcie.c @@ -3068,7 +3068,7 @@ static int mwifiex_pcie_request_irq(struct mwifiex_adapter *adapter) ret); for (j = 0; j < i; j++) free_irq(card->msix_entries[j].vector, - &card->msix_ctx[i]); + &card->msix_ctx[j]); pci_disable_msix(pdev); } else { mwifiex_dbg(adapter, MSG, "MSIx enabled!"); diff --git a/drivers/net/wireless/marvell/mwifiex/scan.c b/drivers/net/wireless/marvell/mwifiex/scan.c index 97c0ec3b822e..bdd4b8465863 100644 --- a/drivers/net/wireless/marvell/mwifiex/scan.c +++ b/drivers/net/wireless/marvell/mwifiex/scan.c @@ -104,12 +104,24 @@ has_vendor_hdr(struct ieee_types_vendor_specific *ie, u8 key) * a given oui in PTK. */ static u8 -mwifiex_search_oui_in_ie(struct ie_body *iebody, u8 *oui) +mwifiex_search_oui_in_ie(struct ie_body *iebody, u8 *oui, int ie_len) { + const size_t ptk_body_offset = offsetof(struct ie_body, ptk_body); u8 count; + /* ie_len is the number of bytes available at iebody. Keep it signed + * and reject a negative (underflowed) length before the unsigned + * comparisons below, so a small or zero IE length cannot wrap. + */ + if (ie_len < 0 || (size_t)ie_len < ptk_body_offset) + return MWIFIEX_OUI_NOT_PRESENT; + count = iebody->ptk_cnt[0]; + /* Reject an OUI count whose list would run past the element. */ + if (ptk_body_offset + count * sizeof(iebody->ptk_body) > (size_t)ie_len) + return MWIFIEX_OUI_NOT_PRESENT; + /* There could be multiple OUIs for PTK hence 1) Take the length. 2) Check all the OUIs for AES. @@ -143,11 +155,14 @@ mwifiex_is_rsn_oui_present(struct mwifiex_bssdescriptor *bss_desc, u32 cipher) u8 ret = MWIFIEX_OUI_NOT_PRESENT; if (has_ieee_hdr(bss_desc->bcn_rsn_ie, WLAN_EID_RSN)) { + int ie_len = (int)bss_desc->bcn_rsn_ie->ieee_hdr.len - + RSN_GTK_OUI_OFFSET; + iebody = (struct ie_body *) (((u8 *) bss_desc->bcn_rsn_ie->data) + RSN_GTK_OUI_OFFSET); oui = &mwifiex_rsn_oui[cipher][0]; - ret = mwifiex_search_oui_in_ie(iebody, oui); + ret = mwifiex_search_oui_in_ie(iebody, oui, ie_len); if (ret) return ret; } @@ -169,10 +184,14 @@ mwifiex_is_wpa_oui_present(struct mwifiex_bssdescriptor *bss_desc, u32 cipher) u8 ret = MWIFIEX_OUI_NOT_PRESENT; if (has_vendor_hdr(bss_desc->bcn_wpa_ie, WLAN_EID_VENDOR_SPECIFIC)) { + int ie_len = (int)bss_desc->bcn_wpa_ie->vend_hdr.len - + (int)sizeof(bss_desc->bcn_wpa_ie->vend_hdr.oui) - + WPA_GTK_OUI_OFFSET; + iebody = (struct ie_body *)((u8 *)bss_desc->bcn_wpa_ie->data + WPA_GTK_OUI_OFFSET); oui = &mwifiex_wpa_oui[cipher][0]; - ret = mwifiex_search_oui_in_ie(iebody, oui); + ret = mwifiex_search_oui_in_ie(iebody, oui, ie_len); if (ret) return ret; } @@ -2096,6 +2115,7 @@ int mwifiex_ret_802_11_scan(struct mwifiex_private *priv, u32 bytes_left; u32 idx; u32 tlv_buf_size; + size_t fixed_size; struct mwifiex_ie_types_chan_band_list_param_set *chan_band_tlv; struct chan_band_param_set *chan_band; u8 is_bgscan_resp; @@ -2111,6 +2131,14 @@ int mwifiex_ret_802_11_scan(struct mwifiex_private *priv, else scan_rsp = &resp->params.scan_resp; + scan_resp_size = le16_to_cpu(resp->size); + fixed_size = scan_rsp->bss_desc_and_tlv_buffer - (u8 *)resp; + if (scan_resp_size < fixed_size) { + mwifiex_dbg(adapter, ERROR, + "SCAN_RESP: response is too short\n"); + ret = -1; + goto check_next_scan; + } if (scan_rsp->number_of_sets > MWIFIEX_MAX_AP) { mwifiex_dbg(adapter, ERROR, @@ -2128,8 +2156,6 @@ int mwifiex_ret_802_11_scan(struct mwifiex_private *priv, "info: SCAN_RESP: bss_descript_size %d\n", bytes_left); - scan_resp_size = le16_to_cpu(resp->size); - mwifiex_dbg(adapter, INFO, "info: SCAN_RESP: returned %d APs before parsing\n", scan_rsp->number_of_sets); @@ -2137,15 +2163,17 @@ int mwifiex_ret_802_11_scan(struct mwifiex_private *priv, bss_info = scan_rsp->bss_desc_and_tlv_buffer; /* - * The size of the TLV buffer is equal to the entire command response - * size (scan_resp_size) minus the fixed fields (sizeof()'s), the - * BSS Descriptions (bss_descript_size as bytesLef) and the command - * response header (S_DS_GEN) + * The TLV buffer follows the command-specific fixed fields and the BSS + * descriptions. Background-scan responses have an additional fixed + * field before scan_rsp, which is included in fixed_size. */ - tlv_buf_size = scan_resp_size - (bytes_left - + sizeof(scan_rsp->bss_descript_size) - + sizeof(scan_rsp->number_of_sets) - + S_DS_GEN); + if (bytes_left > scan_resp_size - fixed_size) { + mwifiex_dbg(adapter, ERROR, + "SCAN_RESP: BSS data exceeds response\n"); + ret = -1; + goto check_next_scan; + } + tlv_buf_size = scan_resp_size - fixed_size - bytes_left; tlv_data = (struct mwifiex_ie_types_data *) (scan_rsp-> bss_desc_and_tlv_buffer + diff --git a/drivers/net/wireless/marvell/mwifiex/util.c b/drivers/net/wireless/marvell/mwifiex/util.c index 7d3631d21223..71305efb77ac 100644 --- a/drivers/net/wireless/marvell/mwifiex/util.c +++ b/drivers/net/wireless/marvell/mwifiex/util.c @@ -317,10 +317,16 @@ mwifiex_parse_mgmt_packet(struct mwifiex_private *priv, u8 *payload, u16 len, switch (stype) { case IEEE80211_STYPE_ACTION: - category = *(payload + sizeof(struct ieee80211_hdr)); + if (len < sizeof(*ieee_hdr) + 1) + return -1; + + category = *(payload + sizeof(*ieee_hdr)); switch (category) { case WLAN_CATEGORY_PUBLIC: - action_code = *(payload + sizeof(struct ieee80211_hdr) + if (len < sizeof(*ieee_hdr) + 2) + return -1; + + action_code = *(payload + sizeof(*ieee_hdr) + 1); if (action_code == WLAN_PUB_ACTION_TDLS_DISCOVER_RES) { addr2 = ieee_hdr->addr2; diff --git a/drivers/net/wireless/microchip/wilc1000/cfg80211.c b/drivers/net/wireless/microchip/wilc1000/cfg80211.c index bb2748a19329..9ad21d41e999 100644 --- a/drivers/net/wireless/microchip/wilc1000/cfg80211.c +++ b/drivers/net/wireless/microchip/wilc1000/cfg80211.c @@ -1058,6 +1058,13 @@ void wilc_wfi_p2p_rx(struct wilc_vif *vif, u8 *buff, u32 size) if (!ieee80211_is_public_action((struct ieee80211_hdr *)buff, size)) goto out_rx_mgmt; + /* ieee80211_is_public_action() only validates up to the category + * byte, so reject frames too short for the P2P public action header + * before dereferencing it or computing size - ie_offset. + */ + if (size < ie_offset) + goto out_rx_mgmt; + d = (struct wilc_p2p_pub_act_frame *)(&mgmt->u.action); if (d->oui_subtype != GO_NEG_REQ && d->oui_subtype != GO_NEG_RSP && d->oui_subtype != P2P_INV_REQ && d->oui_subtype != P2P_INV_RSP) @@ -1200,6 +1207,13 @@ static int mgmt_tx(struct wiphy *wiphy, goto out_set_timeout; } + /* ieee80211_is_public_action() only validates up to the category + * byte, so reject frames too short for the P2P public action header + * before dereferencing it or computing len - ie_offset. + */ + if (len < ie_offset) + goto out_set_timeout; + d = (struct wilc_p2p_pub_act_frame *)(&mgmt->u.action); if (d->oui_type != WLAN_OUI_TYPE_WFA_P2P || d->oui_subtype != GO_NEG_CONF) { diff --git a/drivers/net/wireless/microchip/wilc1000/wlan.c b/drivers/net/wireless/microchip/wilc1000/wlan.c index 4b116fe6f9ea..55a77a2e3288 100644 --- a/drivers/net/wireless/microchip/wilc1000/wlan.c +++ b/drivers/net/wireless/microchip/wilc1000/wlan.c @@ -1197,6 +1197,15 @@ static void wilc_wlan_handle_isr_ext(struct wilc *wilc, u32 int_status) if (size <= 0) return; + /* A size exceeding the RX buffer is bogus; drop the transfer + * instead of overflowing the buffer. + */ + if (size > WILC_RX_BUFF_SIZE) { + wilc->hif_func->hif_clear_int_ext(wilc, + DATA_INT_CLR | ENABLE_RX_VMM); + return; + } + if (WILC_RX_BUFF_SIZE - offset < size) offset = 0; diff --git a/drivers/net/wireless/rsi/rsi_91x_mgmt.c b/drivers/net/wireless/rsi/rsi_91x_mgmt.c index bb167f03367b..d9dcbb255317 100644 --- a/drivers/net/wireless/rsi/rsi_91x_mgmt.c +++ b/drivers/net/wireless/rsi/rsi_91x_mgmt.c @@ -852,8 +852,6 @@ int rsi_hal_load_key(struct rsi_common *common, memcpy(set_key->tx_mic_key, &data[16], 8); memcpy(set_key->rx_mic_key, &data[24], 8); } - } else { - memset(&set_key[FRAME_DESC_SZ], 0, frame_len - FRAME_DESC_SZ); } skb_put(skb, frame_len); diff --git a/drivers/net/wireless/ti/wlcore/main.c b/drivers/net/wireless/ti/wlcore/main.c index 5595f7a1fc0c..edf6ca23c6c3 100644 --- a/drivers/net/wireless/ti/wlcore/main.c +++ b/drivers/net/wireless/ti/wlcore/main.c @@ -3724,10 +3724,8 @@ void wlcore_regdomain_config(struct wl1271 *wl) goto out; ret = wlcore_cmd_regdomain_config_locked(wl); - if (ret < 0) { + if (ret < 0) wl12xx_queue_recovery_work(wl); - goto out; - } pm_runtime_put_autosuspend(wl->dev); out: diff --git a/drivers/net/wireless/virtual/mac80211_hwsim_main.c b/drivers/net/wireless/virtual/mac80211_hwsim_main.c index 02b6d81cccd1..b9446577ff49 100644 --- a/drivers/net/wireless/virtual/mac80211_hwsim_main.c +++ b/drivers/net/wireless/virtual/mac80211_hwsim_main.c @@ -2327,7 +2327,12 @@ static void mac80211_hwsim_stop(struct ieee80211_hw *hw, bool suspend) struct sk_buff *skb; int i; - data->started = false; + /* + * Serialise against wmediumd userspace, so no more frames + * can be handed to mac80211 after this returns. + */ + scoped_guard(mutex, &data->mutex) + data->started = false; for (i = 0; i < ARRAY_SIZE(data->link_data); i++) hrtimer_cancel(&data->link_data[i].beacon_timer); @@ -6505,12 +6510,12 @@ static int hwsim_cloned_frame_received_nl(struct sk_buff *skb_2, if (frame_data_len < sizeof(struct ieee80211_hdr_3addr) || frame_data_len > IEEE80211_MAX_DATA_LEN) - goto err; + goto out; /* Allocate new skb here */ skb = alloc_skb(frame_data_len, GFP_KERNEL); if (skb == NULL) - goto err; + goto out; /* Copy the data */ skb_put_data(skb, frame_data, frame_data_len); @@ -6535,10 +6540,17 @@ static int hwsim_cloned_frame_received_nl(struct sk_buff *skb_2, goto out; } + /* + * Serialise against mac80211_hwsim_stop() - mac80211 doesn't allow + * frames reported while the HW is down, hence the ->started check + * must be under mutex. + */ + mutex_lock(&data2->mutex); + /* check if radio is configured properly */ if ((data2->idle && !data2->tmp_chan) || !data2->started) - goto out; + goto out_unlock; /* A frame is received from user space */ memset(&rx_status, 0, sizeof(rx_status)); @@ -6557,22 +6569,18 @@ static int hwsim_cloned_frame_received_nl(struct sk_buff *skb_2, iter_data.channel = ieee80211_get_channel(data2->hw->wiphy, rx_status.freq); if (!iter_data.channel) - goto out; + goto out_unlock; rx_status.band = iter_data.channel->band; - mutex_lock(&data2->mutex); if (!hwsim_chans_compat(iter_data.channel, channel)) { ieee80211_iterate_active_interfaces_atomic( data2->hw, IEEE80211_IFACE_ITER_NORMAL, mac80211_hwsim_tx_iter, &iter_data); - if (!iter_data.receive) { - mutex_unlock(&data2->mutex); - goto out; - } + if (!iter_data.receive) + goto out_unlock; } - mutex_unlock(&data2->mutex); } else if (!channel) { - goto out; + goto out_unlock; } else { rx_status.freq = channel->center_freq; rx_status.band = channel->band; @@ -6580,7 +6588,7 @@ static int hwsim_cloned_frame_received_nl(struct sk_buff *skb_2, rx_status.rate_idx = nla_get_u32(info->attrs[HWSIM_ATTR_RX_RATE]); if (rx_status.rate_idx >= data2->hw->wiphy->bands[rx_status.band]->n_bitrates) - goto out; + goto out_unlock; rx_status.signal = nla_get_u32(info->attrs[HWSIM_ATTR_SIGNAL]); hdr = (void *)skb->data; @@ -6590,10 +6598,11 @@ static int hwsim_cloned_frame_received_nl(struct sk_buff *skb_2, rx_status.boottime_ns = ktime_get_boottime_ns(); mac80211_hwsim_rx(data2, &rx_status, skb); + mutex_unlock(&data2->mutex); return 0; -err: - pr_debug("mac80211_hwsim: error occurred in %s\n", __func__); +out_unlock: + mutex_unlock(&data2->mutex); out: dev_kfree_skb(skb); return -EINVAL; diff --git a/drivers/net/wireless/virtual/virt_wifi.c b/drivers/net/wireless/virtual/virt_wifi.c index 2335e45db8b8..48afc2432f93 100644 --- a/drivers/net/wireless/virtual/virt_wifi.c +++ b/drivers/net/wireless/virtual/virt_wifi.c @@ -434,6 +434,7 @@ static netdev_tx_t virt_wifi_start_xmit(struct sk_buff *skb, priv->tx_packets++; if (!priv->is_connected) { priv->tx_failed++; + dev_kfree_skb_any(skb); return NET_XMIT_DROP; } @@ -557,7 +558,6 @@ static int virt_wifi_newlink(struct net_device *dev, } eth_hw_addr_inherit(dev, priv->lowerdev); - netif_stacked_transfer_operstate(priv->lowerdev, dev); dev->ieee80211_ptr = kzalloc_obj(*dev->ieee80211_ptr); @@ -583,6 +583,8 @@ static int virt_wifi_newlink(struct net_device *dev, goto unregister_netdev; } + netif_stacked_transfer_operstate(priv->lowerdev, dev); + dev->priv_destructor = virt_wifi_net_device_destructor; priv->being_deleted = false; priv->is_connected = false; diff --git a/drivers/net/wwan/mhi_wwan_mbim.c b/drivers/net/wwan/mhi_wwan_mbim.c index a94998712597..336f89756f10 100644 --- a/drivers/net/wwan/mhi_wwan_mbim.c +++ b/drivers/net/wwan/mhi_wwan_mbim.c @@ -251,6 +251,14 @@ static int mbim_rx_verify_ndp16(struct sk_buff *skb, struct usb_cdc_ncm_ndp16 *n return ret; } +static void mhi_mbim_rx_drop(struct mhi_mbim_link *link, struct sk_buff *skb) +{ + dev_kfree_skb_any(skb); + u64_stats_update_begin(&link->rx_syncp); + u64_stats_inc(&link->rx_errors); + u64_stats_update_end(&link->rx_syncp); +} + static void mhi_mbim_rx(struct mhi_mbim_context *mbim, struct sk_buff *skb) { int ndpoffset; @@ -320,7 +328,10 @@ static void mhi_mbim_rx(struct mhi_mbim_context *mbim, struct sk_buff *skb) continue; skb_put(skbn, dgram_len); - skb_copy_bits(skb, dgram_offset, skbn->data, dgram_len); + if (skb_copy_bits(skb, dgram_offset, skbn->data, dgram_len)) { + mhi_mbim_rx_drop(link, skbn); + continue; + } switch (skbn->data[0] & 0xf0) { case 0x40: @@ -332,10 +343,7 @@ static void mhi_mbim_rx(struct mhi_mbim_context *mbim, struct sk_buff *skb) default: net_err_ratelimited("%s: unknown protocol\n", link->ndev->name); - dev_kfree_skb_any(skbn); - u64_stats_update_begin(&link->rx_syncp); - u64_stats_inc(&link->rx_errors); - u64_stats_update_end(&link->rx_syncp); + mhi_mbim_rx_drop(link, skbn); continue; } @@ -349,9 +357,13 @@ static void mhi_mbim_rx(struct mhi_mbim_context *mbim, struct sk_buff *skb) unlock: rcu_read_unlock(); next_ndp: - /* Other NDP to process? */ - ndpoffset = (int)le16_to_cpu(ndp16.wNextNdpIndex); - if (!ndpoffset) + /* Other NDP to process? The offsets must advance, or a + * self-referencing NDP keeps the loop spinning forever. + */ + n = (int)le16_to_cpu(ndp16.wNextNdpIndex); + if (n > ndpoffset) + ndpoffset = n; + else break; } diff --git a/drivers/net/wwan/t7xx/t7xx_netdev.c b/drivers/net/wwan/t7xx/t7xx_netdev.c index fc0a7cb181df..8f32c2d26931 100644 --- a/drivers/net/wwan/t7xx/t7xx_netdev.c +++ b/drivers/net/wwan/t7xx/t7xx_netdev.c @@ -420,6 +420,10 @@ static void t7xx_ccmni_recv_skb(struct t7xx_ccmni_ctrl *ccmni_ctlb, struct sk_bu skb_cb = T7XX_SKB_CB(skb); netif_id = skb_cb->netif_idx; + if (netif_id >= NIC_DEV_MAX) { + dev_kfree_skb(skb); + return; + } ccmni = READ_ONCE(ccmni_ctlb->ccmni_inst[netif_id]); if (!ccmni) { dev_kfree_skb(skb); diff --git a/drivers/pci/controller/dwc/pci-imx6.c b/drivers/pci/controller/dwc/pci-imx6.c index 39790e66b98d..f7a2eb257c16 100644 --- a/drivers/pci/controller/dwc/pci-imx6.c +++ b/drivers/pci/controller/dwc/pci-imx6.c @@ -1394,12 +1394,6 @@ static int imx_pcie_host_init(struct dw_pcie_rp *pp) } } - ret = imx_pcie_clk_enable(imx_pcie); - if (ret) { - dev_err(dev, "unable to enable pcie clocks: %d\n", ret); - goto err_pwrctrl_power_off; - } - if (pp->bridge && imx_check_flag(imx_pcie, IMX_PCIE_FLAG_HAS_LUT)) { pp->bridge->enable_device = imx_pcie_enable_device; pp->bridge->disable_device = imx_pcie_disable_device; @@ -1415,6 +1409,12 @@ static int imx_pcie_host_init(struct dw_pcie_rp *pp) imx_pcie_configure_type(imx_pcie); + ret = imx_pcie_clk_enable(imx_pcie); + if (ret) { + dev_err(dev, "unable to enable pcie clocks: %d\n", ret); + goto err_pwrctrl_power_off; + } + if (imx_pcie->phy) { ret = phy_init(imx_pcie->phy); if (ret) { diff --git a/drivers/perf/arm-cmn.c b/drivers/perf/arm-cmn.c index b162de3d9d16..33ee2be9b386 100644 --- a/drivers/perf/arm-cmn.c +++ b/drivers/perf/arm-cmn.c @@ -1582,13 +1582,14 @@ static void arm_cmn_claim_wp_idx(struct arm_cmn_dtm *dtm, static u32 arm_cmn_wp_config(struct perf_event *event, int wp_idx) { + struct arm_cmn *cmn = to_cmn(event->pmu); u32 config; u32 dev = CMN_EVENT_WP_DEV_SEL(event); u32 chn = CMN_EVENT_WP_CHN_SEL(event); u32 grp = CMN_EVENT_WP_GRP(event); u32 exc = CMN_EVENT_WP_EXCLUSIVE(event); u32 combine = CMN_EVENT_WP_COMBINE(event); - bool is_cmn600 = to_cmn(event->pmu)->part == PART_CMN600; + bool is_cmn600 = cmn->part == PART_CMN600; /* CMN-600 supports only primary and secondary matching groups */ if (is_cmn600) @@ -1596,8 +1597,11 @@ static u32 arm_cmn_wp_config(struct perf_event *event, int wp_idx) config = FIELD_PREP(CMN_DTM_WPn_CONFIG_WP_DEV_SEL, dev) | FIELD_PREP(CMN_DTM_WPn_CONFIG_WP_CHN_SEL, chn) | - FIELD_PREP(CMN_DTM_WPn_CONFIG_WP_GRP, grp) | - FIELD_PREP(CMN_DTM_WPn_CONFIG_WP_DEV_SEL2, dev >> 1); + FIELD_PREP(CMN_DTM_WPn_CONFIG_WP_GRP, grp); + + if (!cmn->multi_dtm) + config |= FIELD_PREP(CMN_DTM_WPn_CONFIG_WP_DEV_SEL2, dev >> 1); + if (exc) config |= is_cmn600 ? CMN600_WPn_CONFIG_WP_EXCLUSIVE : CMN_DTM_WPn_CONFIG_WP_EXCLUSIVE; diff --git a/drivers/phy/mediatek/phy-mtk-hdmi-mt8195.c b/drivers/phy/mediatek/phy-mtk-hdmi-mt8195.c index 1426a2db984d..a4bc1268946d 100644 --- a/drivers/phy/mediatek/phy-mtk-hdmi-mt8195.c +++ b/drivers/phy/mediatek/phy-mtk-hdmi-mt8195.c @@ -36,7 +36,7 @@ mtk_phy_tmds_clk_ratio(struct mtk_hdmi_phy *hdmi_phy, bool enable) * clock bit ratio 1:40, under 3.4Gbps, clock bit ratio 1:10 */ if (enable) - mtk_phy_update_field(regs + HDMI20_CLK_CFG, REG_TXC_DIV, 3); + mtk_phy_update_field(regs + HDMI20_CLK_CFG, REG_TXC_DIV, VAL_TXC_DIV4); else mtk_phy_clear_bits(regs + HDMI20_CLK_CFG, REG_TXC_DIV); } @@ -290,7 +290,7 @@ static int mtk_hdmi_pll_calc(struct mtk_hdmi_phy *hdmi_phy, struct clk_hw *hw, posdiv2 = 1; /* Digital clk divider, max /32 */ - digital_div = div_u64(ns_hdmipll_ck, posdiv1 * posdiv2 * pixel_clk); + digital_div = div64_u64(ns_hdmipll_ck, posdiv1 * posdiv2 * pixel_clk); if (!(digital_div <= 32 && digital_div >= 1)) return -EINVAL; diff --git a/drivers/phy/mediatek/phy-mtk-hdmi-mt8195.h b/drivers/phy/mediatek/phy-mtk-hdmi-mt8195.h index e26caaf4d104..58800d7659ca 100644 --- a/drivers/phy/mediatek/phy-mtk-hdmi-mt8195.h +++ b/drivers/phy/mediatek/phy-mtk-hdmi-mt8195.h @@ -17,6 +17,9 @@ #define HDMI20_CLK_CFG 0x70 #define REG_TXC_DIV GENMASK(31, 30) +#define VAL_TXC_DIV2 1 +#define VAL_TXC_DIV4 2 +#define VAL_TXC_DIV8 3 #define HDMI_1_CFG_0 0x00 #define RG_HDMITX21_DRV_IBIAS_CLK GENMASK(10, 5) diff --git a/drivers/phy/renesas/phy-rcar-gen3-usb2.c b/drivers/phy/renesas/phy-rcar-gen3-usb2.c index 9ae9975d3255..b5ba751805aa 100644 --- a/drivers/phy/renesas/phy-rcar-gen3-usb2.c +++ b/drivers/phy/renesas/phy-rcar-gen3-usb2.c @@ -27,6 +27,7 @@ #include <linux/reset.h> #include <linux/string.h> #include <linux/usb/of.h> +#include <linux/wait.h> #include <linux/workqueue.h> /******* USB2.0 Host registers (original offset is +0x200) *******/ @@ -106,6 +107,13 @@ /* RZ/G2L specific */ #define USB2_LINECTRL1_USB2_IDMON BIT(0) +/* + * The OTG initialization is expected to finish in 20ms. Choose a large enough + * timeout to avoid waiters exit prematurely the waiting section under heavy + * CPU load. + */ +#define USB2_OTG_INIT_TIMEOUT msecs_to_jiffies(120) + #define NUM_OF_PHYS 4 enum rcar_gen3_phy_index { PHY_INDEX_BOTH_HC, @@ -138,12 +146,20 @@ struct rcar_gen3_chan { struct rcar_gen3_phy rphys[NUM_OF_PHYS]; struct regulator *vbus; struct work_struct work; + wait_queue_head_t otg_init_done; spinlock_t lock; /* protects access to hardware and driver data structure. */ enum usb_dr_mode dr_mode; bool extcon_host; bool is_otg_channel; bool uses_otg_pins; bool otg_internal_reg; + /* + * The OTG can be initialized only once and needs to release the spinlock + * and wait for 20 ms due to hardware constraints. If a thread executes + * PHY configuration code while the OTG PHY is waiting for the 20 ms, the + * thread will have to wait for the OTG PHY initialization to complete. + */ + bool otg_initializing; }; struct rcar_gen3_phy_drv_data { @@ -392,26 +408,58 @@ static ssize_t role_store(struct device *dev, struct device_attribute *attr, struct rcar_gen3_chan *ch = dev_get_drvdata(dev); bool is_b_device; enum phy_mode cur_mode, new_mode; + int retries = NUM_OF_PHYS; + unsigned long flags; + int ret = -EIO; - guard(spinlock_irqsave)(&ch->lock); + spin_lock_irqsave(&ch->lock, flags); - if (!ch->is_otg_channel || !rcar_gen3_is_any_otg_rphy_initialized(ch)) - return -EIO; + if (!ch->is_otg_channel) + goto unlock; + + while (retries-- && ch->otg_initializing) { + spin_unlock_irqrestore(&ch->lock, flags); + + ret = wait_event_timeout(ch->otg_init_done, !ch->otg_initializing, + USB2_OTG_INIT_TIMEOUT); + ret = ret ? 0 : -ETIMEDOUT; + if (ret && !retries) + goto exit; + + spin_lock_irqsave(&ch->lock, flags); + } + + /* If another thread started a new initialization just return -EBUSY. */ + if (ch->otg_initializing) { + ret = -EBUSY; + goto unlock; + } else { + ret = 0; + } + + if (!rcar_gen3_is_any_otg_rphy_initialized(ch)) { + ret = -EIO; + goto unlock; + } - if (sysfs_streq(buf, "host")) + if (sysfs_streq(buf, "host")) { new_mode = PHY_MODE_USB_HOST; - else if (sysfs_streq(buf, "peripheral")) + } else if (sysfs_streq(buf, "peripheral")) { new_mode = PHY_MODE_USB_DEVICE; - else - return -EINVAL; + } else { + ret = -EINVAL; + goto unlock; + } /* is_b_device: true is B-Device. false is A-Device. */ is_b_device = rcar_gen3_check_id(ch); cur_mode = rcar_gen3_get_phy_mode(ch); /* If current and new mode is the same, this returns the error */ - if (cur_mode == new_mode) - return -EINVAL; + if (cur_mode == new_mode) { + ret = -EINVAL; + goto unlock; + } if (new_mode == PHY_MODE_USB_HOST) { /* And is_host must be false */ if (!is_b_device) /* A-Peripheral */ @@ -425,7 +473,10 @@ static ssize_t role_store(struct device *dev, struct device_attribute *attr, rcar_gen3_init_for_peri(ch); } - return count; +unlock: + spin_unlock_irqrestore(&ch->lock, flags); +exit: + return ret ?: count; } static ssize_t role_show(struct device *dev, struct device_attribute *attr, @@ -441,14 +492,11 @@ static ssize_t role_show(struct device *dev, struct device_attribute *attr, } static DEVICE_ATTR_RW(role); -static void rcar_gen3_init_otg(struct rcar_gen3_chan *ch) +static void rcar_gen3_init_otg_phase0(struct rcar_gen3_chan *ch) { void __iomem *usb2_base = ch->base; u32 val; - if (!ch->is_otg_channel || rcar_gen3_is_any_otg_rphy_initialized(ch)) - return; - /* Should not use functions of read-modify-write a register */ val = readl(usb2_base + USB2_LINECTRL1); val = (val & ~USB2_LINECTRL1_DP_RPD) | USB2_LINECTRL1_DPRPD_EN | @@ -471,7 +519,11 @@ static void rcar_gen3_init_otg(struct rcar_gen3_chan *ch) writel(val | USB2_ADPCTRL_IDPULLUP, usb2_base + USB2_ADPCTRL); } } - mdelay(20); +} + +static void rcar_gen3_init_otg_phase1(struct rcar_gen3_chan *ch) +{ + void __iomem *usb2_base = ch->base; writel(0xffffffff, usb2_base + USB2_OBINTSTA); writel(ch->phy_data->obint_enable_bits, usb2_base + USB2_OBINTEN); @@ -502,6 +554,7 @@ static irqreturn_t rcar_gen3_phy_usb2_irq(int irq, void *_ch) void __iomem *usb2_base = ch->base; struct device *dev = ch->dev; irqreturn_t ret = IRQ_NONE; + unsigned long flags; u32 status; pm_runtime_get_noresume(dev); @@ -509,33 +562,102 @@ static irqreturn_t rcar_gen3_phy_usb2_irq(int irq, void *_ch) if (pm_runtime_suspended(dev)) goto rpm_put; - scoped_guard(spinlock, &ch->lock) { - status = readl(usb2_base + USB2_OBINTSTA); - if (status & ch->phy_data->obint_enable_bits) { - dev_vdbg(dev, "%s: %08x\n", __func__, status); - if (ch->phy_data->vblvl_ctrl) - writel(USB2_OBINTSTA_CLEAR, usb2_base + USB2_OBINTSTA); - else - writel(ch->phy_data->obint_enable_bits, usb2_base + USB2_OBINTSTA); - rcar_gen3_device_recognition(ch); - rcar_gen3_configure_vblvl_ctrl(ch); - ret = IRQ_HANDLED; - } + spin_lock_irqsave(&ch->lock, flags); + + status = readl(usb2_base + USB2_OBINTSTA); + if (status & ch->phy_data->obint_enable_bits) { + dev_vdbg(dev, "%s: %08x\n", __func__, status); + if (ch->phy_data->vblvl_ctrl) + writel(USB2_OBINTSTA_CLEAR, usb2_base + USB2_OBINTSTA); + else + writel(ch->phy_data->obint_enable_bits, usb2_base + USB2_OBINTSTA); + + ret = IRQ_HANDLED; + + /* This should not happen! */ + if (ch->otg_initializing) + goto unlock; + + rcar_gen3_device_recognition(ch); + rcar_gen3_configure_vblvl_ctrl(ch); } +unlock: + spin_unlock_irqrestore(&ch->lock, flags); rpm_put: pm_runtime_put_noidle(dev); return ret; } +static void rcar_gen3_phy_usb2_irqs_mask_all(struct rcar_gen3_chan *channel, + u32 *masked_irqs_bits) +{ + u32 val, bitmask = USB2_INT_ENABLE_UCOM_INTEN; + void __iomem *usb2_base = channel->base; + + for (unsigned int i = 0; i < NUM_OF_PHYS; i++) + bitmask |= channel->rphys[i].int_enable_bits; + + val = readl(usb2_base + USB2_INT_ENABLE); + *masked_irqs_bits = val & bitmask; + val &= ~bitmask; + writel(val, usb2_base + USB2_INT_ENABLE); + + /* + * Don't report channel->phy_data->obint_enable_bits IRQs. These are + * unmasked anyway in rcar_gen3_init_otg_phase1(). + */ + val = readl(usb2_base + USB2_OBINTEN); + val &= ~channel->phy_data->obint_enable_bits; + writel(val, usb2_base + USB2_OBINTEN); +} + +static void rcar_gen3_phy_usb2_irqs_unmask(struct rcar_gen3_chan *channel, + u32 irqs_bits) +{ + u32 val, bitmask = USB2_INT_ENABLE_UCOM_INTEN; + void __iomem *usb2_base = channel->base; + + for (unsigned int i = 0; i < NUM_OF_PHYS; i++) + bitmask |= channel->rphys[i].int_enable_bits; + + val = readl(usb2_base + USB2_INT_ENABLE); + val &= ~bitmask; + val |= irqs_bits; + writel(val, usb2_base + USB2_INT_ENABLE); +} + static int rcar_gen3_phy_usb2_init(struct phy *p) { struct rcar_gen3_phy *rphy = phy_get_drvdata(p); struct rcar_gen3_chan *channel = rphy->ch; void __iomem *usb2_base = channel->base; + int retries = NUM_OF_PHYS; + unsigned long flags; u32 val; + int ret; + + spin_lock_irqsave(&channel->lock, flags); - guard(spinlock_irqsave)(&channel->lock); + while (retries-- && channel->otg_initializing) { + spin_unlock_irqrestore(&channel->lock, flags); + + ret = wait_event_timeout(channel->otg_init_done, !channel->otg_initializing, + USB2_OTG_INIT_TIMEOUT); + ret = ret ? 0 : -ETIMEDOUT; + if (ret && !retries) + return ret; + + spin_lock_irqsave(&channel->lock, flags); + } + + /* If another thread started a new initialization just return -EBUSY. */ + if (channel->otg_initializing) { + ret = -EBUSY; + goto unlock; + } else { + ret = 0; + } /* Initialize USB2 part */ val = readl(usb2_base + USB2_INT_ENABLE); @@ -548,8 +670,23 @@ static int rcar_gen3_phy_usb2_init(struct phy *p) } /* Initialize otg part (only if we initialize a PHY with IRQs). */ - if (rphy->int_enable_bits) - rcar_gen3_init_otg(channel); + if (rphy->int_enable_bits && channel->is_otg_channel && + !rcar_gen3_is_any_otg_rphy_initialized(channel)) { + u32 masked_irq_bits = 0; + + rcar_gen3_init_otg_phase0(channel); + rcar_gen3_phy_usb2_irqs_mask_all(channel, &masked_irq_bits); + channel->otg_initializing = true; + spin_unlock_irqrestore(&channel->lock, flags); + + fsleep(20000); + + spin_lock_irqsave(&channel->lock, flags); + rcar_gen3_phy_usb2_irqs_unmask(channel, masked_irq_bits); + rcar_gen3_init_otg_phase1(channel); + channel->otg_initializing = false; + wake_up_all(&channel->otg_init_done); + } if (channel->phy_data->vblvl_ctrl) { /* SIDDQ mode release */ @@ -568,7 +705,10 @@ static int rcar_gen3_phy_usb2_init(struct phy *p) rphy->initialized = true; - return 0; +unlock: + spin_unlock_irqrestore(&channel->lock, flags); + + return ret; } static int rcar_gen3_phy_usb2_exit(struct phy *p) @@ -576,9 +716,32 @@ static int rcar_gen3_phy_usb2_exit(struct phy *p) struct rcar_gen3_phy *rphy = phy_get_drvdata(p); struct rcar_gen3_chan *channel = rphy->ch; void __iomem *usb2_base = channel->base; + int retries = NUM_OF_PHYS; + unsigned long flags; u32 val; + int ret; + + spin_lock_irqsave(&channel->lock, flags); + + while (retries-- && channel->otg_initializing) { + spin_unlock_irqrestore(&channel->lock, flags); + + ret = wait_event_timeout(channel->otg_init_done, !channel->otg_initializing, + USB2_OTG_INIT_TIMEOUT); + ret = ret ? 0 : -ETIMEDOUT; + if (ret && !retries) + return ret; + + spin_lock_irqsave(&channel->lock, flags); + } - guard(spinlock_irqsave)(&channel->lock); + /* If another thread started a new initialization just return -EBUSY. */ + if (channel->otg_initializing) { + ret = -EBUSY; + goto unlock; + } else { + ret = 0; + } rphy->initialized = false; @@ -588,7 +751,9 @@ static int rcar_gen3_phy_usb2_exit(struct phy *p) val &= ~USB2_INT_ENABLE_UCOM_INTEN; writel(val, usb2_base + USB2_INT_ENABLE); - return 0; +unlock: + spin_unlock_irqrestore(&channel->lock, flags); + return ret; } static int rcar_gen3_phy_usb2_power_on(struct phy *p) @@ -596,8 +761,10 @@ static int rcar_gen3_phy_usb2_power_on(struct phy *p) struct rcar_gen3_phy *rphy = phy_get_drvdata(p); struct rcar_gen3_chan *channel = rphy->ch; void __iomem *usb2_base = channel->base; + int retries = NUM_OF_PHYS; + unsigned long flags; u32 val; - int ret = 0; + int ret; if (channel->vbus && !channel->otg_internal_reg) { ret = regulator_enable(channel->vbus); @@ -605,7 +772,27 @@ static int rcar_gen3_phy_usb2_power_on(struct phy *p) return ret; } - guard(spinlock_irqsave)(&channel->lock); + spin_lock_irqsave(&channel->lock, flags); + + while (retries-- && channel->otg_initializing) { + spin_unlock_irqrestore(&channel->lock, flags); + + ret = wait_event_timeout(channel->otg_init_done, !channel->otg_initializing, + USB2_OTG_INIT_TIMEOUT); + ret = ret ? 0 : -ETIMEDOUT; + if (ret && !retries) + goto disable_regulator; + + spin_lock_irqsave(&channel->lock, flags); + } + + /* If another thread started a new initialization just return -EBUSY. */ + if (channel->otg_initializing) { + ret = -EBUSY; + goto unlock; + } else { + ret = 0; + } if (!rcar_gen3_are_all_rphys_power_off(channel)) goto out; @@ -620,27 +807,59 @@ out: /* The powered flag should be set for any other phys anyway */ rphy->powered = true; - return 0; +unlock: + spin_unlock_irqrestore(&channel->lock, flags); + +disable_regulator: + if (ret && channel->vbus && !channel->otg_internal_reg) + regulator_disable(channel->vbus); + + return ret; } static int rcar_gen3_phy_usb2_power_off(struct phy *p) { struct rcar_gen3_phy *rphy = phy_get_drvdata(p); struct rcar_gen3_chan *channel = rphy->ch; - int ret = 0; + int retries = NUM_OF_PHYS; + unsigned long flags; + int ret; - scoped_guard(spinlock_irqsave, &channel->lock) { - rphy->powered = false; + spin_lock_irqsave(&channel->lock, flags); - if (rcar_gen3_are_all_rphys_power_off(channel)) { - u32 val = readl(channel->base + USB2_USBCTR); + while (retries-- && channel->otg_initializing) { + spin_unlock_irqrestore(&channel->lock, flags); - val |= USB2_USBCTR_PLL_RST; - writel(val, channel->base + USB2_USBCTR); - } + ret = wait_event_timeout(channel->otg_init_done, !channel->otg_initializing, + USB2_OTG_INIT_TIMEOUT); + ret = ret ? 0 : -ETIMEDOUT; + if (ret && !retries) + return ret; + + spin_lock_irqsave(&channel->lock, flags); + } + + /* If another thread started a new initialization just return -EBUSY. */ + if (channel->otg_initializing) { + ret = -EBUSY; + goto unlock; + } else { + ret = 0; + } + + rphy->powered = false; + + if (rcar_gen3_are_all_rphys_power_off(channel)) { + u32 val = readl(channel->base + USB2_USBCTR); + + val |= USB2_USBCTR_PLL_RST; + writel(val, channel->base + USB2_USBCTR); } - if (channel->vbus && !channel->otg_internal_reg) +unlock: + spin_unlock_irqrestore(&channel->lock, flags); + + if (!ret && channel->vbus && !channel->otg_internal_reg) ret = regulator_disable(channel->vbus); return ret; @@ -1022,6 +1241,7 @@ static int rcar_gen3_phy_usb2_probe(struct platform_device *pdev) return ret; spin_lock_init(&channel->lock); + init_waitqueue_head(&channel->otg_init_done); for (i = 0; i < NUM_OF_PHYS; i++) { channel->rphys[i].phy = devm_phy_create(dev, NULL, channel->phy_data->phy_usb2_ops); diff --git a/drivers/power/sequencing/Kconfig b/drivers/power/sequencing/Kconfig index 1c5f5820f5b7..226c62704d9b 100644 --- a/drivers/power/sequencing/Kconfig +++ b/drivers/power/sequencing/Kconfig @@ -29,7 +29,8 @@ config POWER_SEQUENCING_QCOM_WCN config POWER_SEQUENCING_TH1520_GPU tristate "T-HEAD TH1520 GPU power sequencing driver" - depends on (ARCH_THEAD && AUXILIARY_BUS) || COMPILE_TEST + depends on ARCH_THEAD || COMPILE_TEST + select AUXILIARY_BUS help Say Y here to enable the power sequencing driver for the TH1520 SoC GPU. This driver handles the complex clock and reset sequence diff --git a/drivers/power/sequencing/core.c b/drivers/power/sequencing/core.c index 0cb71efbb268..3076b3879af9 100644 --- a/drivers/power/sequencing/core.c +++ b/drivers/power/sequencing/core.c @@ -101,6 +101,7 @@ static struct pwrseq_unit *pwrseq_unit_new(const struct pwrseq_unit_data *data) } kref_init(&unit->ref); + INIT_LIST_HEAD(&unit->list); INIT_LIST_HEAD(&unit->deps); unit->enable = data->enable; unit->disable = data->disable; @@ -504,10 +505,6 @@ pwrseq_device_register(const struct pwrseq_config *config) */ device_initialize(&pwrseq->dev); - ret = dev_set_name(&pwrseq->dev, "pwrseq.%d", pwrseq->id); - if (ret) - goto err_put_pwrseq; - pwrseq->owner = config->owner ?: THIS_MODULE; pwrseq->match = config->match; @@ -516,6 +513,10 @@ pwrseq_device_register(const struct pwrseq_config *config) INIT_LIST_HEAD(&pwrseq->targets); INIT_LIST_HEAD(&pwrseq->units); + ret = dev_set_name(&pwrseq->dev, "pwrseq.%d", pwrseq->id); + if (ret) + goto err_put_pwrseq; + ret = pwrseq_setup_targets(config->targets, pwrseq); if (ret) goto err_put_pwrseq; @@ -912,6 +913,8 @@ int pwrseq_enable(struct pwrseq_desc *desc) if (!ret) desc->powered_on = true; } + if (ret) + return ret; if (target->post_enable) { ret = target->post_enable(pwrseq); diff --git a/drivers/scsi/Kconfig b/drivers/scsi/Kconfig index 4a2af0f702e1..1eec66195cf4 100644 --- a/drivers/scsi/Kconfig +++ b/drivers/scsi/Kconfig @@ -753,6 +753,7 @@ config SCSI_IBMVFC tristate "IBM Virtual FC support" depends on PPC_PSERIES && SCSI depends on SCSI_FC_ATTRS + depends on NVME_FC || NVME_FC=n help This is the IBM POWER Virtual FC Client diff --git a/drivers/scsi/fnic/fnic.h b/drivers/scsi/fnic/fnic.h index c576a7f5083e..3ba1592940ca 100644 --- a/drivers/scsi/fnic/fnic.h +++ b/drivers/scsi/fnic/fnic.h @@ -31,7 +31,7 @@ #define DRV_NAME "fnic" #define DRV_DESCRIPTION "Cisco FCoE HBA Driver" -#define DRV_VERSION "1.9.0.0" +#define DRV_VERSION "1.9.0.1" #define PFX DRV_NAME ": " #define DFX DRV_NAME "%d: " diff --git a/drivers/scsi/fnic/fnic_isr.c b/drivers/scsi/fnic/fnic_isr.c index 02856745580f..43149d8312ec 100644 --- a/drivers/scsi/fnic/fnic_isr.c +++ b/drivers/scsi/fnic/fnic_isr.c @@ -245,7 +245,14 @@ int fnic_set_intr_mode_msix(struct fnic *fnic) unsigned int m = ARRAY_SIZE(fnic->wq); unsigned int o = ARRAY_SIZE(fnic->hw_copy_wq); unsigned int min_irqs = n + m + 1 + 1; /*rq, raw wq, wq, err*/ - + /* + * Make driver critical vectors unmanaged, or else it can get tied + * to an offline CPU. This can happen when hyper-threading is off. + */ + struct irq_affinity affd = { + .pre_vectors = n + m + 1, /* rq, raw wq, 1 ioq */ + .post_vectors = 1, /* err */ + }; /* * We need n RQs, m WQs, o Copy WQs, n+m+o CQs, and n+m+o+1 INTRs * (last INTR is used for WQ/RQ errors and notification area) @@ -263,8 +270,8 @@ int fnic_set_intr_mode_msix(struct fnic *fnic) int vec_count = 0; int vecs = fnic->rq_count + fnic->raw_wq_count + fnic->wq_copy_count + 1; - vec_count = pci_alloc_irq_vectors(fnic->pdev, min_irqs, vecs, - PCI_IRQ_MSIX | PCI_IRQ_AFFINITY); + vec_count = pci_alloc_irq_vectors_affinity(fnic->pdev, min_irqs, + vecs, PCI_IRQ_MSIX|PCI_IRQ_AFFINITY, &affd); FNIC_ISR_DBG(KERN_INFO, fnic, "allocated %d MSI-X vectors\n", vec_count); diff --git a/drivers/scsi/fnic/fnic_main.c b/drivers/scsi/fnic/fnic_main.c index 9b3025007075..f13c381a66d7 100644 --- a/drivers/scsi/fnic/fnic_main.c +++ b/drivers/scsi/fnic/fnic_main.c @@ -744,8 +744,19 @@ out_free_ioreq_tables: return ret; } +static void fnic_mq_init_queue_map(struct fnic *fnic, + struct blk_mq_queue_map *qmap) +{ + unsigned int cpu; + + for_each_possible_cpu(cpu) + qmap->mq_map[cpu] = 0; +} + void fnic_mq_map_queues_cpus(struct Scsi_Host *host) { + const struct cpumask *mask; + unsigned int queue, cpu; struct fnic *fnic = *((struct fnic **) shost_priv(host)); struct pci_dev *l_pdev = fnic->pdev; int intr_mode = fnic->config.intr_mode; @@ -766,7 +777,35 @@ void fnic_mq_map_queues_cpus(struct Scsi_Host *host) return; } - blk_mq_map_hw_queues(qmap, &l_pdev->dev, FNIC_PCI_OFFSET); + fnic_mq_init_queue_map(fnic, qmap); + + /* + * Setup CPU to Queue mapping for all managed MSI-X IRQs. + * Q0 is driver critical and non-managed, hence start from Q1. + */ + for (queue = 1; queue < qmap->nr_queues; queue++) { + int irq_num = pci_irq_vector(fnic->pdev, + queue + FNIC_PCI_OFFSET); + + if (irq_num < 0) + continue; + + mask = pci_irq_get_affinity(fnic->pdev, + queue + FNIC_PCI_OFFSET); + if (!mask) { + shost_printk(KERN_ERR, host, + "failed to get irq_affinity map for queue:%d\n", irq_num); + continue; + } + FNIC_MAIN_DBG(KERN_INFO, fnic, + "got irq_affinity map for %d:\n", irq_num); + for_each_cpu(cpu, mask) { + qmap->mq_map[cpu] = qmap->queue_offset + queue; + FNIC_MAIN_DBG(KERN_INFO, fnic, + "[Q%d] cpu:%d <=> irq:%d\n", + queue, cpu, irq_num); + } + } } static int fnic_probe(struct pci_dev *pdev, const struct pci_device_id *ent) diff --git a/drivers/scsi/pm8001/pm8001_init.c b/drivers/scsi/pm8001/pm8001_init.c index 54b35893261a..5af81c73a8f8 100644 --- a/drivers/scsi/pm8001/pm8001_init.c +++ b/drivers/scsi/pm8001/pm8001_init.c @@ -58,15 +58,15 @@ MODULE_PARM_DESC(link_rate, "Enable link rate.\n" bool pm8001_use_msix = true; module_param_named(use_msix, pm8001_use_msix, bool, 0444); -MODULE_PARM_DESC(zoned, "Use MSIX interrupts. Default: true"); +MODULE_PARM_DESC(use_msix, "Use MSIX interrupts. Default: true"); static bool pm8001_use_tasklet = true; module_param_named(use_tasklet, pm8001_use_tasklet, bool, 0444); -MODULE_PARM_DESC(zoned, "Use MSIX interrupts. Default: true"); +MODULE_PARM_DESC(use_tasklet, "Use tasklets for interrupt handling. Default: true"); static bool pm8001_read_wwn = true; module_param_named(read_wwn, pm8001_read_wwn, bool, 0444); -MODULE_PARM_DESC(zoned, "Get WWN from the controller. Default: true"); +MODULE_PARM_DESC(read_wwn, "Get WWN from the controller. Default: true"); uint pcs_event_log_severity = 0x03; module_param(pcs_event_log_severity, int, 0644); diff --git a/drivers/scsi/qla2xxx/qla_os.c b/drivers/scsi/qla2xxx/qla_os.c index c0efdbff5da7..3c412c7fb6fe 100644 --- a/drivers/scsi/qla2xxx/qla_os.c +++ b/drivers/scsi/qla2xxx/qla_os.c @@ -352,7 +352,7 @@ MODULE_PARM_DESC(ql2xnvme_queues, int ql2xfc2target = 1; module_param(ql2xfc2target, int, 0444); -MODULE_PARM_DESC(qla2xfc2target, +MODULE_PARM_DESC(ql2xfc2target, "Enables FC2 Target support. " "0 - FC2 Target support is disabled. " "1 - FC2 Target support is enabled (default)."); diff --git a/drivers/scsi/scsi.c b/drivers/scsi/scsi.c index 76cdad063f7b..f285521d9de6 100644 --- a/drivers/scsi/scsi.c +++ b/drivers/scsi/scsi.c @@ -727,6 +727,7 @@ int scsi_cdl_enable(struct scsi_device *sdev, bool enable) struct scsi_mode_data data; struct scsi_sense_hdr sshdr; char *buf_data; + size_t avail, offset; int len; ret = scsi_mode_sense(sdev, 0x08, 0x0a, 0xf2, buf, sizeof(buf), @@ -735,11 +736,24 @@ int scsi_cdl_enable(struct scsi_device *sdev, bool enable) return -EINVAL; /* Enable or disable CDL using the ATA feature page */ - len = min_t(size_t, sizeof(buf), - data.length - data.header_length - - data.block_descriptor_length); - buf_data = buf + data.header_length + - data.block_descriptor_length; + avail = min_t(size_t, data.length, sizeof(buf)); + if (data.header_length > avail) + return -EINVAL; + + offset = data.header_length; + avail -= data.header_length; + + if (data.block_descriptor_length > avail) + return -EINVAL; + + offset += data.block_descriptor_length; + avail -= data.block_descriptor_length; + + if (avail < 5) + return -EINVAL; + + buf_data = buf + offset; + len = avail; /* * If we want to enable CDL and CDL is already enabled on the diff --git a/drivers/soc/samsung/exynos-pmu.c b/drivers/soc/samsung/exynos-pmu.c index f5fcdde9750e..efccdd63e40e 100644 --- a/drivers/soc/samsung/exynos-pmu.c +++ b/drivers/soc/samsung/exynos-pmu.c @@ -409,13 +409,12 @@ static struct notifier_block exynos_cpupm_reboot_nb = { static int setup_cpuhp_and_cpuidle(struct device *dev) { - struct device_node *intr_gen_node; + struct device_node *intr_gen_node __free(device_node) = + of_parse_phandle(dev->of_node, "google,pmu-intr-gen-syscon", 0); struct resource intrgen_res; void __iomem *virt_addr; int ret, cpu; - intr_gen_node = of_parse_phandle(dev->of_node, - "google,pmu-intr-gen-syscon", 0); if (!intr_gen_node) { /* * To maintain support for older DTs that didn't specify syscon @@ -431,8 +430,6 @@ static int setup_cpuhp_and_cpuidle(struct device *dev) * syscon provided regmap. */ ret = of_address_to_resource(intr_gen_node, 0, &intrgen_res); - of_node_put(intr_gen_node); - virt_addr = devm_ioremap(dev, intrgen_res.start, resource_size(&intrgen_res)); if (!virt_addr) diff --git a/drivers/soundwire/cadence_master.c b/drivers/soundwire/cadence_master.c index e690237fe981..a0a63b76793e 100644 --- a/drivers/soundwire/cadence_master.c +++ b/drivers/soundwire/cadence_master.c @@ -1702,6 +1702,13 @@ int sdw_cdns_clock_stop(struct sdw_cdns *cdns, bool block_wake) } /* + * wait for any in-flight peripheral event handling to complete before stopping the clock. + * No need to disable peripheral interrupts before canceling the work, as the peripheral + * interrupts are already masked before the work is scheduled. + */ + cancel_work_sync(&cdns->work); + + /* * Before entering clock stop we mask the Slave * interrupts. This helps avoid having to deal with e.g. a * Slave becoming UNATTACHED while the clock is being stopped diff --git a/drivers/soundwire/dmi-quirks.c b/drivers/soundwire/dmi-quirks.c index 768255dd12db..62fa64b22412 100644 --- a/drivers/soundwire/dmi-quirks.c +++ b/drivers/soundwire/dmi-quirks.c @@ -203,6 +203,13 @@ static const struct dmi_system_id adr_remap_quirk_table[] = { { .matches = { DMI_MATCH(DMI_SYS_VENDOR, "ASUS"), + DMI_MATCH(DMI_BOARD_NAME, "GX651AX"), + }, + .driver_data = (void *)ghost_realtek, + }, + { + .matches = { + DMI_MATCH(DMI_SYS_VENDOR, "ASUS"), DMI_MATCH(DMI_BOARD_NAME, "UX5406AA"), }, .driver_data = (void *)ghost_realtek, diff --git a/drivers/watchdog/da9062_wdt.c b/drivers/watchdog/da9062_wdt.c index 426962547df1..4d558652e9e7 100644 --- a/drivers/watchdog/da9062_wdt.c +++ b/drivers/watchdog/da9062_wdt.c @@ -256,7 +256,7 @@ static int __maybe_unused da9062_wdt_suspend(struct device *dev) if (!wdt->use_sw_pm) return 0; - if (watchdog_active(wdd)) + if (watchdog_active(wdd) || watchdog_hw_running(wdd)) return da9062_wdt_stop(wdd); return 0; @@ -270,7 +270,7 @@ static int __maybe_unused da9062_wdt_resume(struct device *dev) if (!wdt->use_sw_pm) return 0; - if (watchdog_active(wdd)) + if (watchdog_active(wdd) || watchdog_hw_running(wdd)) return da9062_wdt_start(wdd); return 0; diff --git a/drivers/watchdog/da9063_wdt.c b/drivers/watchdog/da9063_wdt.c index 92e1b78ff481..3703110e82fc 100644 --- a/drivers/watchdog/da9063_wdt.c +++ b/drivers/watchdog/da9063_wdt.c @@ -271,7 +271,7 @@ static int da9063_wdt_suspend(struct device *dev) if (!da9063->use_sw_pm) return 0; - if (watchdog_active(wdd)) + if (watchdog_active(wdd) || watchdog_hw_running(wdd)) return da9063_wdt_stop(wdd); return 0; @@ -285,7 +285,7 @@ static int da9063_wdt_resume(struct device *dev) if (!da9063->use_sw_pm) return 0; - if (watchdog_active(wdd)) + if (watchdog_active(wdd) || watchdog_hw_running(wdd)) return da9063_wdt_start(wdd); return 0; diff --git a/drivers/watchdog/digicolor_wdt.c b/drivers/watchdog/digicolor_wdt.c index 073d37867f47..de1a3267a972 100644 --- a/drivers/watchdog/digicolor_wdt.c +++ b/drivers/watchdog/digicolor_wdt.c @@ -25,6 +25,7 @@ struct dc_wdt { void __iomem *base; struct clk *clk; spinlock_t lock; + unsigned long rate; }; static unsigned timeout; @@ -61,7 +62,7 @@ static int dc_wdt_start(struct watchdog_device *wdog) { struct dc_wdt *wdt = watchdog_get_drvdata(wdog); - dc_wdt_set(wdt, wdog->timeout * clk_get_rate(wdt->clk)); + dc_wdt_set(wdt, wdog->timeout * wdt->rate); return 0; } @@ -79,7 +80,7 @@ static int dc_wdt_set_timeout(struct watchdog_device *wdog, unsigned int t) { struct dc_wdt *wdt = watchdog_get_drvdata(wdog); - dc_wdt_set(wdt, t * clk_get_rate(wdt->clk)); + dc_wdt_set(wdt, t * wdt->rate); wdog->timeout = t; return 0; @@ -90,7 +91,7 @@ static unsigned int dc_wdt_get_timeleft(struct watchdog_device *wdog) struct dc_wdt *wdt = watchdog_get_drvdata(wdog); uint32_t count = readl_relaxed(wdt->base + TIMER_A_COUNT); - return count / clk_get_rate(wdt->clk); + return count / wdt->rate; } static const struct watchdog_ops dc_wdt_ops = { @@ -130,7 +131,11 @@ static int dc_wdt_probe(struct platform_device *pdev) wdt->clk = devm_clk_get(dev, NULL); if (IS_ERR(wdt->clk)) return PTR_ERR(wdt->clk); - dc_wdt_wdd.max_timeout = U32_MAX / clk_get_rate(wdt->clk); + + wdt->rate = clk_get_rate(wdt->clk); + if (!wdt->rate) + return -EINVAL; + dc_wdt_wdd.max_timeout = U32_MAX / wdt->rate; dc_wdt_wdd.timeout = dc_wdt_wdd.max_timeout; dc_wdt_wdd.parent = dev; diff --git a/drivers/watchdog/msc313e_wdt.c b/drivers/watchdog/msc313e_wdt.c index 4a5cce2a16b1..d62586f6e09d 100644 --- a/drivers/watchdog/msc313e_wdt.c +++ b/drivers/watchdog/msc313e_wdt.c @@ -46,6 +46,9 @@ static void msc313e_wdt_set_hw_timeout(struct msc313e_wdt_priv *priv, { u32 t = timeout * clk_get_rate(priv->clk); + /* Clear before to prevent premature reset during non-atomic updates. */ + writew(1, priv->base + REG_WDT_CLR); + writew(t & 0xffff, priv->base + REG_WDT_MAX_PRD_L); writew((t >> 16) & 0xffff, priv->base + REG_WDT_MAX_PRD_H); writew(1, priv->base + REG_WDT_CLR); @@ -76,6 +79,9 @@ static int msc313e_wdt_stop(struct watchdog_device *wdev) { struct msc313e_wdt_priv *priv = watchdog_get_drvdata(wdev); + /* Clear before to prevent premature reset during non-atomic updates. */ + writew(1, priv->base + REG_WDT_CLR); + writew(0, priv->base + REG_WDT_MAX_PRD_L); writew(0, priv->base + REG_WDT_MAX_PRD_H); writew(0, priv->base + REG_WDT_CLR); @@ -191,11 +197,15 @@ static int __maybe_unused msc313e_wdt_suspend(struct device *dev) static int __maybe_unused msc313e_wdt_resume(struct device *dev) { struct msc313e_wdt_priv *priv = dev_get_drvdata(dev); + int ret = 0; - if (watchdog_active(&priv->wdev) || watchdog_hw_running(&priv->wdev)) - msc313e_wdt_start(&priv->wdev); + if (watchdog_active(&priv->wdev) || watchdog_hw_running(&priv->wdev)) { + ret = msc313e_wdt_start(&priv->wdev); + if (ret) + dev_err(dev, "Failed to restart watchdog (err=%d)\n", ret); + } - return 0; + return ret; } static SIMPLE_DEV_PM_OPS(msc313e_wdt_pm_ops, msc313e_wdt_suspend, msc313e_wdt_resume); diff --git a/drivers/watchdog/rtd119x_wdt.c b/drivers/watchdog/rtd119x_wdt.c index 984905695dde..0bfadb58917b 100644 --- a/drivers/watchdog/rtd119x_wdt.c +++ b/drivers/watchdog/rtd119x_wdt.c @@ -98,6 +98,7 @@ static int rtd119x_wdt_probe(struct platform_device *pdev) { struct device *dev = &pdev->dev; struct rtd119x_watchdog_device *data; + unsigned long rate; data = devm_kzalloc(dev, sizeof(*data), GFP_KERNEL); if (!data) @@ -111,10 +112,14 @@ static int rtd119x_wdt_probe(struct platform_device *pdev) if (IS_ERR(data->clk)) return PTR_ERR(data->clk); + rate = clk_get_rate(data->clk); + if (!rate) + return -EINVAL; + data->wdt_dev.info = &rtd119x_wdt_info; data->wdt_dev.ops = &rtd119x_wdt_ops; data->wdt_dev.timeout = 120; - data->wdt_dev.max_timeout = 0xffffffff / clk_get_rate(data->clk); + data->wdt_dev.max_timeout = 0xffffffff / rate; data->wdt_dev.min_timeout = 1; data->wdt_dev.parent = dev; diff --git a/drivers/watchdog/rzv2h_wdt.c b/drivers/watchdog/rzv2h_wdt.c index 3b6abb66a1da..83540dd9a37b 100644 --- a/drivers/watchdog/rzv2h_wdt.c +++ b/drivers/watchdog/rzv2h_wdt.c @@ -278,6 +278,7 @@ static int rzv2h_wdt_probe(struct platform_device *pdev) struct device *dev = &pdev->dev; struct rzv2h_wdt_priv *priv; struct clk *count_clk; + unsigned long rate; int ret; priv = devm_kzalloc(dev, sizeof(*priv), GFP_KERNEL); @@ -314,8 +315,12 @@ static int rzv2h_wdt_probe(struct platform_device *pdev) return dev_err_probe(dev, -EINVAL, "Invalid count source\n"); } + rate = clk_get_rate(count_clk); + if (!rate) + return dev_err_probe(dev, -EINVAL, "Invalid clock rate\n"); + priv->wdev.max_hw_heartbeat_ms = (MILLI * priv->of_data->timeout_cycles * - priv->of_data->cks_div) / clk_get_rate(count_clk); + priv->of_data->cks_div) / rate; dev_dbg(dev, "max hw timeout of %dms\n", priv->wdev.max_hw_heartbeat_ms); ret = devm_pm_runtime_enable(dev); diff --git a/drivers/watchdog/sp5100_tco.c b/drivers/watchdog/sp5100_tco.c index 7e99c3b1f367..72ad59649ccc 100644 --- a/drivers/watchdog/sp5100_tco.c +++ b/drivers/watchdog/sp5100_tco.c @@ -605,8 +605,10 @@ static int __init sp5100_tco_init(void) pr_info("SP5100/SB800 TCO WatchDog Timer Driver\n"); err = platform_driver_register(&sp5100_tco_driver); - if (err) + if (err) { + pci_dev_put(sp5100_tco_pci); return err; + } sp5100_tco_platform_device = platform_device_register_simple(TCO_DRIVER_NAME, -1, NULL, 0); @@ -619,6 +621,7 @@ static int __init sp5100_tco_init(void) unreg_platform_driver: platform_driver_unregister(&sp5100_tco_driver); + pci_dev_put(sp5100_tco_pci); return err; } @@ -626,6 +629,7 @@ static void __exit sp5100_tco_exit(void) { platform_device_unregister(sp5100_tco_platform_device); platform_driver_unregister(&sp5100_tco_driver); + pci_dev_put(sp5100_tco_pci); } module_init(sp5100_tco_init); diff --git a/drivers/watchdog/starfive-wdt.c b/drivers/watchdog/starfive-wdt.c index af55adc4a3c6..5a3254c83d3b 100644 --- a/drivers/watchdog/starfive-wdt.c +++ b/drivers/watchdog/starfive-wdt.c @@ -371,7 +371,7 @@ static void starfive_wdt_stop(struct starfive_wdt *wdt) static int starfive_wdt_pm_start(struct watchdog_device *wdd) { struct starfive_wdt *wdt = watchdog_get_drvdata(wdd); - int ret = pm_runtime_get_sync(wdd->parent); + int ret = pm_runtime_resume_and_get(wdd->parent); if (ret < 0) return ret; diff --git a/fs/9p/vfs_addr.c b/fs/9p/vfs_addr.c index 1ac0b3dcc077..13cf87a5f90c 100644 --- a/fs/9p/vfs_addr.c +++ b/fs/9p/vfs_addr.c @@ -54,11 +54,37 @@ static void v9fs_begin_writeback(struct netfs_io_request *wreq) static void v9fs_issue_write(struct netfs_io_subrequest *subreq) { struct p9_fid *fid = subreq->rreq->netfs_priv; + struct inode *inode = subreq->rreq->inode; + struct netfs_inode *ictx = netfs_inode(inode); int err, len; len = p9_client_write(fid, subreq->start, &subreq->io_iter, &err); - if (len > 0) + if (len > 0) { + uoff_t end = subreq->start + len, i_size, remote, zp; + bool set = false; + + spin_lock(&inode->i_lock); + + /* We can read the sizes directly as we hold i_lock. */ + i_size = inode->i_size; + remote = ictx->_remote_i_size; + zp = ictx->_zero_point; + + if (end > i_size) { + i_size = end; + set = true; + } + if (end > remote) { + remote = end; + set = true; + } + + if (set) + netfs_write_sizes(inode, i_size, remote, zp); + spin_unlock(&inode->i_lock); + __set_bit(NETFS_SREQ_MADE_PROGRESS, &subreq->flags); + } netfs_write_subrequest_terminated(subreq, len ?: err); } diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c index af1b898029e8..0284be0e4e82 100644 --- a/fs/btrfs/dev-replace.c +++ b/fs/btrfs/dev-replace.c @@ -494,6 +494,7 @@ static int mark_block_group_to_copy(struct btrfs_fs_info *fs_info, path->reada = READA_FORWARD; path->search_commit_root = true; path->skip_locking = true; + path->need_commit_sem = true; key.objectid = src_dev->devid; key.type = BTRFS_DEV_EXTENT_KEY; diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 819727460bcf..dc7ad92876c0 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -2357,6 +2357,10 @@ static int validate_sys_chunk_array(const struct btrfs_fs_info *fs_info, key.type, cur); return -EUCLEAN; } + + if (unlikely(cur + sizeof(*chunk) > sys_array_size)) + goto short_read; + chunk = (struct btrfs_chunk *)(sb->sys_chunk_array + cur); num_stripes = btrfs_stack_chunk_num_stripes(chunk); if (unlikely(cur + btrfs_chunk_item_size(num_stripes) > sys_array_size)) diff --git a/fs/btrfs/file.c b/fs/btrfs/file.c index 20e15dc30bfb..f978c6524aa0 100644 --- a/fs/btrfs/file.c +++ b/fs/btrfs/file.c @@ -2509,8 +2509,10 @@ int btrfs_replace_file_extents(struct btrfs_inode *inode, inode_set_ctime_current(&inode->vfs_inode)); ret = btrfs_update_inode(trans, inode); - if (ret) + if (unlikely(ret)) { + btrfs_abort_transaction(trans, ret); break; + } btrfs_end_transaction(trans); btrfs_btree_balance_dirty(fs_info); diff --git a/fs/btrfs/free-space-tree.c b/fs/btrfs/free-space-tree.c index 1b3d82ae3de8..b7a4a6ade30f 100644 --- a/fs/btrfs/free-space-tree.c +++ b/fs/btrfs/free-space-tree.c @@ -1353,7 +1353,7 @@ int btrfs_rebuild_free_space_tree(struct btrfs_fs_info *fs_info) if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); btrfs_end_transaction(trans); - return ret; + goto out_clear; } node = rb_first_cached(&fs_info->block_group_cache_tree); @@ -1371,14 +1371,16 @@ int btrfs_rebuild_free_space_tree(struct btrfs_fs_info *fs_info) if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); btrfs_end_transaction(trans); - return ret; + goto out_clear; } next: if (btrfs_should_end_transaction(trans)) { btrfs_end_transaction(trans); trans = btrfs_start_transaction(free_space_root, 1); - if (IS_ERR(trans)) - return PTR_ERR(trans); + if (IS_ERR(trans)) { + ret = PTR_ERR(trans); + goto out_clear; + } } node = rb_next(node); } @@ -1390,6 +1392,10 @@ next: ret = btrfs_commit_transaction(trans); clear_bit(BTRFS_FS_FREE_SPACE_TREE_UNTRUSTED, &fs_info->flags); return ret; + +out_clear: + clear_bit(BTRFS_FS_CREATING_FREE_SPACE_TREE, &fs_info->flags); + return ret; } static int __add_block_group_free_space(struct btrfs_trans_handle *trans, diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 93ef3cec191e..558b4a3f9633 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -2339,12 +2339,27 @@ static int run_delalloc_inline(struct btrfs_inode *inode, struct folio *locked_f } else if (inode->prop_compress) { compress_type = inode->prop_compress; } + /* + * We need to pass blocksize and not i_size, otherwise we can't + * create compressed inline extents for data smaller than sector + * size with lzo. + */ cb = btrfs_compress_bio(inode, 0, blocksize, compress_type, compress_level, 0); if (IS_ERR(cb)) { cb = NULL; /* Just fall back to non-compressed case. */ } else { compressed_size = cb->bbio.bio.bi_iter.bi_size; + /* + * If we did not save space, it's pointless and wasteful + * to have an inline compressed extent, so fallback to + * an uncompressed inline extent. + */ + if (compressed_size >= i_size) { + cleanup_compressed_bio(cb); + cb = NULL; + compressed_size = 0; + } } } if (!can_cow_file_range_inline(inode, 0, i_size, compressed_size)) { @@ -3877,7 +3892,8 @@ int btrfs_orphan_cleanup(struct btrfs_root *root) if (ret) goto out; } - trans = btrfs_start_transaction(root, 1); + /* Only deletes the orphan. */ + trans = btrfs_start_transaction_fallback_global_rsv(root, 1); if (IS_ERR(trans)) { ret = PTR_ERR(trans); goto out; diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c index 464129b1b0d4..ddb620ac241b 100644 --- a/fs/btrfs/super.c +++ b/fs/btrfs/super.c @@ -1836,8 +1836,12 @@ static int btrfs_statfs(struct dentry *dentry, struct kstatfs *buf) f_fsid.val[0] ^= btrfs_root_id(BTRFS_I(d_inode(dentry))->root) >> 32; f_fsid.val[1] ^= btrfs_root_id(BTRFS_I(d_inode(dentry))->root); - /* Hash dev_t to avoid f_fsid collision with cloned filesystems. */ - if (fs_info->fs_devices->total_devices == 1) { + /* + * Hash dev_t to avoid f_fsid collisions with cloned filesystems. + * Only do this when a clone is present so the original filesystem + * (mounted first) maintains backward-compatible f_fsid behavior. + */ + if (fs_info->fs_devices->temp_fsid) { __kernel_fsid_t dev_fsid = u64_to_fsid(huge_encode_dev(fs_info->fs_devices->latest_dev->bdev->bd_dev)); diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index ab5abbb475e2..4b1e47173c63 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -2129,7 +2129,7 @@ static int check_dev_extent_item(const struct extent_buffer *leaf, sectorsize))) { generic_err(leaf, slot, "invalid dev extent chunk offset, has %llu not aligned to %u", - btrfs_dev_extent_chunk_objectid(leaf, de), + btrfs_dev_extent_chunk_offset(leaf, de), sectorsize); return -EUCLEAN; } @@ -2306,7 +2306,7 @@ static int check_free_space_extent(struct extent_buffer *leaf, struct btrfs_key if (unlikely(btrfs_item_size(leaf, slot) != 0)) { generic_err(leaf, slot, - "invalid item size for free space info, has %u expect 0", + "invalid item size for free space extent, has %u expect 0", btrfs_item_size(leaf, slot)); return -EUCLEAN; } diff --git a/fs/btrfs/verity.c b/fs/btrfs/verity.c index 4e0ab5842274..600337a84fbe 100644 --- a/fs/btrfs/verity.c +++ b/fs/btrfs/verity.c @@ -94,6 +94,20 @@ static loff_t merkle_file_pos(const struct inode *inode) } /* + * Start a transaction for removing verity items or the verity orphan. + * + * Like unlink, this only deletes items and frees space in the end, so the + * reservation may come from the global reserve when the filesystem is full + * (-ENOSPC) and is not subject to the qgroup limit (-EDQUOT). Otherwise a + * failed enable could never be cleaned up in either situation. + */ +static struct btrfs_trans_handle *start_verity_cleanup_trans(struct btrfs_root *root, + unsigned int num_items) +{ + return btrfs_start_transaction_fallback_global_rsv(root, num_items); +} + +/* * Drop all the items for this inode with this key_type. * * @inode: inode to drop items for @@ -120,7 +134,7 @@ static int drop_verity_items(struct btrfs_inode *inode, u8 key_type) while (1) { /* 1 for the item being dropped */ - trans = btrfs_start_transaction(root, 1); + trans = start_verity_cleanup_trans(root, 1); if (IS_ERR(trans)) return PTR_ERR(trans); @@ -466,7 +480,7 @@ static int rollback_verity(struct btrfs_inode *inode) * 1 for updating the inode flag * 1 for deleting the orphan */ - trans = btrfs_start_transaction(root, 2); + trans = start_verity_cleanup_trans(root, 2); if (IS_ERR(trans)) { ret = PTR_ERR(trans); trans = NULL; diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 74584669507f..85ea9c5d4536 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -749,6 +749,36 @@ const u8 *btrfs_sb_fsid_ptr(const struct btrfs_super_block *sb) return has_metadata_uuid ? sb->metadata_uuid : sb->fsid; } +static bool should_rename_device(const struct btrfs_device *dev) +{ + bool ret; + const char *old_name; + + rcu_read_lock(); + old_name = rcu_dereference(dev->name); + /* + * For systems booted without an initramfs, the rootfs has the device + * name "/dev/root". + * + * Although using btrfs without an initramfs is not recommended (if a + * new device is added to the rootfs, the system can no longer boot, as + * there is no way to register all devices), there is still a minority + * of users doing this. + * + * And after the system is up, a later device scan on the real block + * device file will never get this device's name updated, as the + * device->devt is still the same. + * + * Here we add one and only one exception for "/dev/root", to allow the + * device name to be updated even if the new path points to the same + * block device. + */ + ret = (strcmp(old_name, "/dev/root") == 0); + rcu_read_unlock(); + + return ret; +} + /* * Add new device to list of registered devices * @@ -869,7 +899,8 @@ static noinline struct btrfs_device *device_list_add(const char *path, MAJOR(path_devt), MINOR(path_devt), current->comm, task_pid_nr(current)); - } else if (!device->name || device->devt != path_devt) { + } else if (!device->name || device->devt != path_devt || + should_rename_device(device)) { const char *old_name; /* diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index 9cc2c9c1a606..08a15465a087 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -2688,6 +2688,11 @@ bool btrfs_can_activate_zone(struct btrfs_fs_devices *fs_devices, u64 flags) switch (flags & BTRFS_BLOCK_GROUP_PROFILE_MASK) { case 0: /* single */ + case BTRFS_BLOCK_GROUP_RAID0: + case BTRFS_BLOCK_GROUP_RAID1: + case BTRFS_BLOCK_GROUP_RAID1C3: + case BTRFS_BLOCK_GROUP_RAID1C4: + case BTRFS_BLOCK_GROUP_RAID10: ret = (atomic_read(&zinfo->active_zones_left) >= (1 + reserved)); break; case BTRFS_BLOCK_GROUP_DUP: @@ -480,11 +480,12 @@ static void dax_associate_entry(void *entry, struct address_space *mapping, unsigned long address, bool shared) { unsigned long size = dax_entry_size(entry), index; - struct folio *folio = dax_to_folio(entry); + struct folio *folio; if (dax_is_zero_entry(entry) || dax_is_empty_entry(entry)) return; + folio = dax_to_folio(entry); index = linear_page_index(vma, address & ~(size - 1)); if (shared && (folio->mapping || dax_folio_is_shared(folio))) { if (folio->mapping) @@ -505,21 +506,23 @@ static void dax_associate_entry(void *entry, struct address_space *mapping, static void dax_disassociate_entry(void *entry, struct address_space *mapping, bool trunc) { - struct folio *folio = dax_to_folio(entry); + struct folio *folio; if (dax_is_zero_entry(entry) || dax_is_empty_entry(entry)) return; + folio = dax_to_folio(entry); dax_folio_put(folio); } static struct page *dax_busy_page(void *entry) { - struct folio *folio = dax_to_folio(entry); + struct folio *folio; if (dax_is_zero_entry(entry) || dax_is_empty_entry(entry)) return NULL; + folio = dax_to_folio(entry); if (folio_ref_count(folio) - folio_mapcount(folio)) return &folio->page; else diff --git a/fs/exec.c b/fs/exec.c index d3081c8f7c10..819643408e6d 100644 --- a/fs/exec.c +++ b/fs/exec.c @@ -1115,6 +1115,17 @@ static struct file *bprm_identity_file(const struct linux_binprm *bprm) return bprm->file; } +static void posixtimer_exec(struct task_struct *me) +{ +#ifdef CONFIG_POSIX_TIMERS + spin_lock_irq(&me->sighand->siglock); + posix_cpu_timers_exit(me); + spin_unlock_irq(&me->sighand->siglock); + exit_itimers(me); + flush_itimer_signals(); +#endif +} + /* * Calling this is the point of no return. None of the failures will be * seen by userspace since either the process is already taking a fatal @@ -1152,6 +1163,16 @@ int begin_new_exec(struct linux_binprm * bprm) retval = de_thread(me); if (retval) goto out; + + /* + * This must be done here to ensure that POSIX CPU timers which were + * armed on the current task are dequeued from me::posix_cputimers. + * Otherwise in case of a TID switch the deletion of the related POSIX + * timer would not remove an enqueued timer because the TID lookup + * of the old TID fails. + */ + posixtimer_exec(me); + /* see the comment in check_unsafe_exec() */ current->fs->in_exec = 0; /* @@ -1206,14 +1227,6 @@ int begin_new_exec(struct linux_binprm * bprm) if (retval) goto out_unlock; -#ifdef CONFIG_POSIX_TIMERS - spin_lock_irq(&me->sighand->siglock); - posix_cpu_timers_exit(me); - spin_unlock_irq(&me->sighand->siglock); - exit_itimers(me); - flush_itimer_signals(); -#endif - /* * Make the signal table private. */ diff --git a/fs/nfsd/export.c b/fs/nfsd/export.c index a7ebce53faec..b6e0c543e028 100644 --- a/fs/nfsd/export.c +++ b/fs/nfsd/export.c @@ -1005,7 +1005,8 @@ static int nfsd_nl_parse_one_export(struct cache_detail *cd, goto out_uuid; err = 0; - nfsd4_setup_layout_type(&exp); + if (exp.ex_flags & NFSEXP_PNFS) + nfsd4_setup_layout_type(&exp); } expp = svc_export_lookup(&exp); diff --git a/fs/ntfs/attrib.c b/fs/ntfs/attrib.c index 848a0d338b89..c949ff765075 100644 --- a/fs/ntfs/attrib.c +++ b/fs/ntfs/attrib.c @@ -2917,7 +2917,7 @@ retry: attr_ni = NULL; /* Allocate new extent. */ - err = ntfs_mft_record_alloc(ni->vol, 0, &attr_ni, ni, NULL); + err = ntfs_mft_record_alloc(ni->vol, 0, &attr_ni, ni, NULL, -1); if (err) { ntfs_error(sb, "Failed to allocate extent record"); goto err_out; @@ -3550,7 +3550,7 @@ int ntfs_attr_record_move_away(struct ntfs_attr_search_ctx *ctx, int extra) * new extent and move attribute to it. */ ni = NULL; - err = ntfs_mft_record_alloc(base_ni->vol, 0, &ni, base_ni, NULL); + err = ntfs_mft_record_alloc(base_ni->vol, 0, &ni, base_ni, NULL, -1); if (err) { ntfs_error(sb, "Couldn't allocate MFT record, err : %d", err); return err; @@ -3574,7 +3574,8 @@ int ntfs_attr_record_move_away(struct ntfs_attr_search_ctx *ctx, int extra) * update allocated and compressed size. */ static int ntfs_attr_update_meta(struct attr_record *a, struct ntfs_inode *ni, - struct mft_record *m, struct ntfs_attr_search_ctx *ctx) + struct mft_record *m, struct ntfs_attr_search_ctx *ctx, + struct ntfs_inode *locked_ni, bool defer_attrlist) { int sparse, err = 0; struct ntfs_inode *base_ni; @@ -3610,6 +3611,8 @@ static int ntfs_attr_update_meta(struct attr_record *a, struct ntfs_inode *ni, le16_to_cpu(a->data.non_resident.mapping_pairs_offset) == 8) && !(le32_to_cpu(m->bytes_allocated) - le32_to_cpu(m->bytes_in_use))) { + if (defer_attrlist) + return -ENOSPC; if (!NInoAttrList(base_ni)) { err = ntfs_inode_add_attrlist(base_ni); if (err) @@ -3623,7 +3626,7 @@ static int ntfs_attr_update_meta(struct attr_record *a, struct ntfs_inode *ni, goto out; } - err = ntfs_attrlist_update(base_ni); + err = ntfs_attrlist_update_locked(base_ni, locked_ni); if (err) goto out; err = -EAGAIN; @@ -3703,6 +3706,8 @@ out: * ntfs_attr_update_mapping_pairs - update mapping pairs for ntfs attribute * @ni: non-resident ntfs inode for which we need update * @from_vcn: update runlist starting this VCN + * @locked_ni: inode whose runlist write lock is already held + * @defer_attrlist: return -ENOSPC instead of updating an attribute list * * Build mapping pairs from @na->rl and write them to the disk. Also, this * function updates sparse bit, allocated and compressed size (allocates/frees @@ -3712,7 +3717,10 @@ out: * call to this function. Vice-versa @na->compressed_size will be calculated and * set to correct value during this function. */ -int ntfs_attr_update_mapping_pairs(struct ntfs_inode *ni, s64 from_vcn) +static int __ntfs_attr_update_mapping_pairs(struct ntfs_inode *ni, + s64 from_vcn, + struct ntfs_inode *locked_ni, + bool defer_attrlist) { struct ntfs_attr_search_ctx *ctx; struct ntfs_inode *base_ni; @@ -3804,7 +3812,8 @@ retry: continue; } - err = ntfs_attr_update_meta(a, ni, m, ctx); + err = ntfs_attr_update_meta(a, ni, m, ctx, locked_ni, + defer_attrlist); if (err < 0) { if (err == -EAGAIN) { ntfs_attr_put_search_ctx(ctx); @@ -3844,18 +3853,28 @@ retry: */ if (ni->type == AT_ATTRIBUTE_LIST) { ntfs_attr_put_search_ctx(ctx); - if (ntfs_inode_free_space(base_ni, mp_size - - cur_max_mp_size)) { - ntfs_debug("Attribute list is too big. Defragment the volume\n"); - return -ENOSPC; + ctx = NULL; + if (locked_ni == ni || defer_attrlist) { + err = -ENOSPC; + goto put_err_out; } - if (ntfs_attrlist_update(base_ni)) - return -EIO; + err = ntfs_inode_free_space(base_ni, mp_size - + cur_max_mp_size); + if (err) + return err; + err = ntfs_attrlist_update_locked( + base_ni, locked_ni); + if (err) + return err; goto retry; } /* Add attribute list if it isn't present, and retry. */ if (!NInoAttrList(base_ni)) { + if (defer_attrlist) { + err = -ENOSPC; + goto put_err_out; + } ntfs_attr_put_search_ctx(ctx); if (ntfs_inode_add_attrlist(base_ni)) { ntfs_error(sb, "Can not add attrlist"); @@ -3883,13 +3902,21 @@ retry: } } + if (defer_attrlist && + (ctx->ntfs_ino->nr_extents == -1 || + NInoAttrList(ctx->ntfs_ino)) && + ctx->attr->type != AT_ATTRIBUTE_LIST) { + err = -ENOSPC; + goto put_err_out; + } + /* Update lowest vcn. */ a->data.non_resident.lowest_vcn = cpu_to_le64(stop_vcn); mark_mft_record_dirty(ctx->ntfs_ino); if ((ctx->ntfs_ino->nr_extents == -1 || NInoAttrList(ctx->ntfs_ino)) && ctx->attr->type != AT_ATTRIBUTE_LIST) { ctx->al_entry->lowest_vcn = cpu_to_le64(stop_vcn); - err = ntfs_attrlist_update(base_ni); + err = ntfs_attrlist_update_locked(base_ni, locked_ni); if (err) goto put_err_out; } @@ -3976,7 +4003,10 @@ retry: unsigned int de_cnt = 0; /* Allocate new mft record. */ - err = ntfs_mft_record_alloc(ni->vol, 0, &ext_ni, base_ni, NULL); + err = ntfs_mft_record_alloc(ni->vol, 0, &ext_ni, base_ni, NULL, + base_ni->mft_no == FILE_MFT && + ni->type == AT_DATA && + ni->name == AT_UNNAMED ? stop_vcn : -1); if (err) { ntfs_error(sb, "Failed to allocate extent record"); goto put_err_out; @@ -4061,6 +4091,19 @@ put_err_out: return err; } +int ntfs_attr_update_mapping_pairs_locked(struct ntfs_inode *ni, + s64 from_vcn, + struct ntfs_inode *locked_ni) +{ + return __ntfs_attr_update_mapping_pairs(ni, from_vcn, locked_ni, + false); +} + +int ntfs_attr_update_mapping_pairs(struct ntfs_inode *ni, s64 from_vcn) +{ + return ntfs_attr_update_mapping_pairs_locked(ni, from_vcn, NULL); +} + /* * ntfs_attr_make_resident - convert a non-resident to a resident attribute * @ni: open ntfs attribute to make resident @@ -4194,7 +4237,9 @@ static int ntfs_attr_make_resident(struct ntfs_inode *ni, struct ntfs_attr_searc * * Reduce the size of a non-resident, open ntfs attribute @na to @newsize bytes. */ -static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsize) +static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, + const s64 newsize, + struct ntfs_inode *locked_ni) { struct ntfs_volume *vol; struct ntfs_attr_search_ctx *ctx; @@ -4202,6 +4247,7 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz s64 nr_freed_clusters; int err; struct ntfs_inode *base_ni; + bool runlist_locked = locked_ni == ni; ntfs_debug("Inode 0x%llx attr 0x%x new size %lld\n", (unsigned long long)ni->mft_no, ni->type, (long long)newsize); @@ -4247,18 +4293,24 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz * clusters if there is a change. */ if (ntfs_bytes_to_cluster(vol, ni->allocated_size) != first_free_vcn) { - struct ntfs_attr_search_ctx *ctx; + /* + * ntfs_cluster_free() and ntfs_rl_truncate_nolock() + * both require this lock. + */ + if (!runlist_locked) + down_write(&ni->runlist.lock); err = ntfs_attr_map_whole_runlist(ni); if (err) { ntfs_debug("Eeek! ntfs_attr_map_whole_runlist failed.\n"); - return err; + goto unlock_runlist; } ctx = ntfs_attr_get_search_ctx(ni, NULL); if (!ctx) { ntfs_error(vol->sb, "%s: Failed to get search context", __func__); - return -ENOMEM; + err = -ENOMEM; + goto unlock_runlist; } /* Deallocate all clusters starting with the first free one. */ @@ -4266,7 +4318,8 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz if (nr_freed_clusters < 0) { ntfs_debug("Eeek! Freeing of clusters failed. Aborting...\n"); ntfs_attr_put_search_ctx(ctx); - return (int)nr_freed_clusters; + err = (int)nr_freed_clusters; + goto unlock_runlist; } ntfs_attr_put_search_ctx(ctx); @@ -4279,7 +4332,8 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz kvfree(ni->runlist.rl); ni->runlist.rl = NULL; ntfs_error(vol->sb, "Eeek! Run list truncation failed.\n"); - return -EIO; + err = -EIO; + goto unlock_runlist; } /* Prepare to mapping pairs update. */ @@ -4295,11 +4349,13 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz VFS_I(base_ni)->i_blocks = ni->allocated_size >> 9; /* Write mapping pairs for new runlist. */ - err = ntfs_attr_update_mapping_pairs(ni, 0 /*first_free_vcn*/); + err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni); if (err) { ntfs_debug("Eeek! Mapping pairs update failed. Leaving inconstant metadata. Run chkdsk.\n"); - return err; + goto unlock_runlist; } + if (!runlist_locked) + up_write(&ni->runlist.lock); } /* Get the first attribute record. */ @@ -4341,7 +4397,11 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz /* If the attribute now has zero size, make it resident. */ if (!newsize && !NInoEncrypted(ni) && !NInoCompressed(ni)) { + if (!runlist_locked) + down_write(&ni->runlist.lock); err = ntfs_attr_make_resident(ni, ctx); + if (!runlist_locked) + up_write(&ni->runlist.lock); if (err) { /* If couldn't make resident, just continue. */ if (err != -EPERM) @@ -4358,6 +4418,11 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz put_err_out: ntfs_attr_put_search_ctx(ctx); return err; + +unlock_runlist: + if (!runlist_locked) + up_write(&ni->runlist.lock); + return err; } /* @@ -4366,13 +4431,14 @@ put_err_out: * @prealloc_size: preallocation size (in bytes) to which to expand the attribute * @newsize: new size (in bytes) to which to expand the attribute * @holes: how to create a hole if expanding - * @need_lock: whether mrec lock is needed or not + * @locked_ni: inode whose runlist lock is already held * * Expand the size of a non-resident, open ntfs attribute @na to @newsize bytes, * by allocating new clusters. */ static int ntfs_non_resident_attr_expand(struct ntfs_inode *ni, const s64 newsize, - const s64 prealloc_size, unsigned int holes, bool need_lock) + const s64 prealloc_size, unsigned int holes, + struct ntfs_inode *locked_ni) { s64 lcn_seek_from; s64 first_free_vcn; @@ -4519,13 +4585,39 @@ static int ntfs_non_resident_attr_expand(struct ntfs_inode *ni, const s64 newsiz ntfs_bytes_to_cluster(vol, ni->allocated_size), first_free_vcn - ntfs_bytes_to_cluster(vol, ni->allocated_size), - lcn_seek_from, DATA_ZONE, false, false, false); + lcn_seek_from, DATA_ZONE, false, + ni->type == AT_ATTRIBUTE_LIST, false); if (IS_ERR(rl)) { ntfs_debug("Cluster allocation failed (%lld)", (long long)first_free_vcn - ntfs_bytes_to_cluster(vol, ni->allocated_size)); return PTR_ERR(rl); } + /* + * A contiguous ATTRIBUTE_LIST allocation keeps its mapping + * pairs small enough to fit in the base MFT record. The + * allocator can return a short run when contiguity was + * requested, so discard it and retry normally if necessary. + */ + if (ni->type == AT_ATTRIBUTE_LIST && + (rl->vcn != ntfs_bytes_to_cluster(vol, + ni->allocated_size) || + rl->length != first_free_vcn - + ntfs_bytes_to_cluster(vol, ni->allocated_size) || + rl[1].length)) { + ntfs_cluster_free_from_rl(vol, rl); + kvfree(rl); + rl = ntfs_cluster_alloc(vol, + ntfs_bytes_to_cluster(vol, + ni->allocated_size), + first_free_vcn - + ntfs_bytes_to_cluster(vol, + ni->allocated_size), + lcn_seek_from, DATA_ZONE, false, + false, false); + if (IS_ERR(rl)) + return PTR_ERR(rl); + } } if (!NInoCompressed(ni)) { @@ -4544,7 +4636,8 @@ static int ntfs_non_resident_attr_expand(struct ntfs_inode *ni, const s64 newsiz /* Prepare to mapping pairs update. */ ni->allocated_size = ntfs_cluster_to_bytes(vol, first_free_vcn); - err = ntfs_attr_update_mapping_pairs(ni, 0); + err = ntfs_attr_update_mapping_pairs_locked( + ni, 0, locked_ni); if (err) { ntfs_debug("Mapping pairs update failed"); goto rollback; @@ -4588,11 +4681,11 @@ rollback: ntfs_debug("Leaking clusters"); /* Now, truncate the runlist itself. */ - if (need_lock) + if (ni != locked_ni) down_write(&ni->runlist.lock); err2 = ntfs_rl_truncate_nolock(vol, &ni->runlist, ntfs_bytes_to_cluster(vol, org_alloc_size)); - if (need_lock) + if (ni != locked_ni) up_write(&ni->runlist.lock); if (err2) { /* @@ -4606,11 +4699,11 @@ rollback: /* Prepare to mapping pairs update. */ ni->allocated_size = org_alloc_size; /* Restore mapping pairs. */ - if (need_lock) + if (ni != locked_ni) down_read(&ni->runlist.lock); - if (ntfs_attr_update_mapping_pairs(ni, 0)) + if (__ntfs_attr_update_mapping_pairs(ni, 0, locked_ni, true)) ntfs_error(sb, "Failed to restore old mapping pairs"); - if (need_lock) + if (ni != locked_ni) up_read(&ni->runlist.lock); if (NInoSparse(ni) || NInoCompressed(ni)) { @@ -4715,7 +4808,8 @@ attr_resize_again: mark_mft_record_dirty(ctx->ntfs_ino); ntfs_attr_put_search_ctx(ctx); /* Resize non-resident attribute */ - return ntfs_non_resident_attr_expand(attr_ni, newsize, prealloc_size, holes, true); + return ntfs_non_resident_attr_expand( + attr_ni, newsize, prealloc_size, holes, NULL); } else if (err != -ENOSPC && err != -EPERM) { ntfs_error(sb, "Failed to make attribute non-resident"); goto put_err_out; @@ -4836,7 +4930,7 @@ attr_resize_again: } /* Allocate new mft record. */ - err = ntfs_mft_record_alloc(base_ni->vol, 0, &ext_ni, base_ni, NULL); + err = ntfs_mft_record_alloc(base_ni->vol, 0, &ext_ni, base_ni, NULL, -1); if (err) { ntfs_error(sb, "Couldn't allocate MFT record"); goto put_err_out; @@ -4890,13 +4984,14 @@ int __ntfs_attr_truncate_vfs(struct ntfs_inode *ni, const s64 newsize, if (NInoNonResident(ni)) { if (newsize > i_size) { down_write(&ni->runlist.lock); - err = ntfs_non_resident_attr_expand(ni, newsize, 0, - NVolDisableSparse(ni->vol) ? - HOLES_NO : HOLES_OK, - false); + err = ntfs_non_resident_attr_expand( + ni, newsize, 0, + NVolDisableSparse(ni->vol) ? + HOLES_NO : HOLES_OK, ni); up_write(&ni->runlist.lock); } else - err = ntfs_non_resident_attr_shrink(ni, newsize); + err = ntfs_non_resident_attr_shrink( + ni, newsize, NULL); } else err = ntfs_resident_attr_resize(ni, newsize, 0, NVolDisableSparse(ni->vol) ? @@ -4905,7 +5000,9 @@ int __ntfs_attr_truncate_vfs(struct ntfs_inode *ni, const s64 newsize, return err; } -int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 prealloc_size) +int ntfs_attr_expand_locked(struct ntfs_inode *ni, const s64 newsize, + const s64 prealloc_size, + struct ntfs_inode *locked_ni) { int err = 0; @@ -4918,7 +5015,8 @@ int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 preallo ntfs_debug("Entering for inode 0x%llx, attr 0x%x, size %lld\n", (unsigned long long)ni->mft_no, ni->type, newsize); - if (ni->data_size == newsize) { + if (ni->data_size == newsize && + (!prealloc_size || prealloc_size <= ni->allocated_size)) { ntfs_debug("Size is already ok\n"); return 0; } @@ -4933,10 +5031,11 @@ int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 preallo } if (NInoNonResident(ni)) { - if (newsize > ni->data_size) - err = ntfs_non_resident_attr_expand(ni, newsize, prealloc_size, - NVolDisableSparse(ni->vol) ? - HOLES_NO : HOLES_OK, true); + if (newsize > ni->data_size || prealloc_size > ni->allocated_size) + err = ntfs_non_resident_attr_expand( + ni, newsize, prealloc_size, + NVolDisableSparse(ni->vol) ? + HOLES_NO : HOLES_OK, locked_ni); } else err = ntfs_resident_attr_resize(ni, newsize, prealloc_size, NVolDisableSparse(ni->vol) ? @@ -4947,6 +5046,12 @@ int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 preallo return err; } +int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, + const s64 prealloc_size) +{ + return ntfs_attr_expand_locked(ni, newsize, prealloc_size, NULL); +} + /* * ntfs_attr_truncate_i - resize an ntfs attribute * @ni: open ntfs inode to resize @@ -4959,7 +5064,9 @@ int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 preallo * newly allocated space is marked as not initialised and no real allocation * on disk is performed. */ -int ntfs_attr_truncate_i(struct ntfs_inode *ni, const s64 newsize, unsigned int holes) +int ntfs_attr_truncate_i_locked(struct ntfs_inode *ni, const s64 newsize, + unsigned int holes, + struct ntfs_inode *locked_ni) { int err; @@ -4993,15 +5100,23 @@ int ntfs_attr_truncate_i(struct ntfs_inode *ni, const s64 newsize, unsigned int if (NInoNonResident(ni)) { if (newsize > ni->data_size) - err = ntfs_non_resident_attr_expand(ni, newsize, 0, holes, true); + err = ntfs_non_resident_attr_expand( + ni, newsize, 0, holes, locked_ni); else - err = ntfs_non_resident_attr_shrink(ni, newsize); + err = ntfs_non_resident_attr_shrink( + ni, newsize, locked_ni); } else err = ntfs_resident_attr_resize(ni, newsize, 0, holes); ntfs_debug("Return status %d\n", err); return err; } +int ntfs_attr_truncate_i(struct ntfs_inode *ni, const s64 newsize, + unsigned int holes) +{ + return ntfs_attr_truncate_i_locked(ni, newsize, holes, NULL); +} + /* * Resize an attribute, creating a hole if relevant */ @@ -5019,10 +5134,11 @@ int ntfs_attr_map_cluster(struct ntfs_inode *ni, s64 vcn_start, s64 *lcn_start, struct ntfs_volume *vol = ni->vol; struct ntfs_attr_search_ctx *ctx; struct runlist_element *rl, *rlc; + struct runlist_element *old_rl = NULL; s64 vcn = vcn_start, lcn, clu_count; s64 lcn_seek_from = -1; int err = 0; - size_t new_rl_count; + size_t new_rl_count, old_rl_count; err = ntfs_attr_map_whole_runlist(ni); if (err) @@ -5115,6 +5231,19 @@ int ntfs_attr_map_cluster(struct ntfs_inode *ni, s64 vcn_start, s64 *lcn_start, WARN_ON(rlc->vcn != vcn); lcn = rlc->lcn; clu_count = rlc->length; + old_rl_count = ni->runlist.count; + old_rl = kmemdup(ni->runlist.rl, + old_rl_count * sizeof(*old_rl), GFP_NOFS); + if (!old_rl) { + err = -ENOMEM; + if (ntfs_cluster_free_from_rl(vol, rlc)) { + ntfs_error(vol->sb, + "Failed to free cluster allocation after runlist backup failure."); + NVolSetErrors(vol); + } + kvfree(rlc); + goto out; + } rl = ntfs_runlists_merge(&ni->runlist, rlc, 0, &new_rl_count); if (IS_ERR(rl)) { @@ -5138,15 +5267,32 @@ int ntfs_attr_map_cluster(struct ntfs_inode *ni, s64 vcn_start, s64 *lcn_start, if (update_mp) { ntfs_attr_reinit_search_ctx(ctx); - err = ntfs_attr_update_mapping_pairs(ni, 0); + err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni); if (err) { int err2; err2 = ntfs_cluster_free(ni, vcn, clu_count, ctx); - if (err2 < 0) + if (err2 < 0 || err2 != clu_count) { ntfs_error(vol->sb, - "Failed to free cluster allocation. Leaving inconstant metadata.\n"); - goto out; + "Failed to free cluster allocation. Leaving inconsistent metadata.\n"); + NVolSetErrors(vol); + goto out; + } + + /* + * Restore the runlist before repairing the on-disk + * mapping pairs. + */ + kvfree(ni->runlist.rl); + ni->runlist.rl = old_rl; + ni->runlist.count = old_rl_count; + old_rl = NULL; + if (ntfs_attr_update_mapping_pairs_locked( + ni, 0, ni)) { + ntfs_error(vol->sb, + "Failed to restore mapping pairs after allocation rollback.\n"); + NVolSetErrors(vol); + } } } else { VFS_I(ni)->i_blocks += clu_count << (vol->cluster_size_bits - 9); @@ -5158,6 +5304,7 @@ int ntfs_attr_map_cluster(struct ntfs_inode *ni, s64 vcn_start, s64 *lcn_start, *lcn_count = clu_count; *balloc = true; out: + kvfree(old_rl); ntfs_attr_put_search_ctx(ctx); return err; } @@ -5401,7 +5548,7 @@ int ntfs_non_resident_attr_insert_range(struct ntfs_inode *ni, s64 start_vcn, s6 ni->data_size += ntfs_cluster_to_bytes(vol, len); if (ntfs_cluster_to_bytes(vol, start_vcn) < ni->initialized_size) ni->initialized_size += ntfs_cluster_to_bytes(vol, len); - ret = ntfs_attr_update_mapping_pairs(ni, 0); + ret = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni); up_write(&ni->runlist.lock); if (ret) return ret; @@ -5486,7 +5633,7 @@ int ntfs_non_resident_attr_collapse_range(struct ntfs_inode *ni, s64 start_vcn, } if (ni->allocated_size > 0) { - ret = ntfs_attr_update_mapping_pairs(ni, 0); + ret = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni); if (ret) { up_write(&ni->runlist.lock); goto out_rl; @@ -5564,7 +5711,7 @@ int ntfs_non_resident_attr_punch_hole(struct ntfs_inode *ni, s64 start_vcn, s64 ni->runlist.rl = rl; ni->runlist.count = new_rl_count; - ret = ntfs_attr_update_mapping_pairs(ni, 0); + ret = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni); up_write(&ni->runlist.lock); if (ret) { kvfree(punch_rl); @@ -5740,7 +5887,7 @@ int ntfs_attr_fallocate(struct ntfs_inode *ni, loff_t start, loff_t byte_len, bo if (NInoRunlistDirty(ni)) { mutex_lock_nested(&ni->mrec_lock, NTFS_INODE_MUTEX_NORMAL); down_write(&ni->runlist.lock); - err = ntfs_attr_update_mapping_pairs(ni, 0); + err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni); if (err) ntfs_error(ni->vol->sb, "Updating mapping pairs failed"); else diff --git a/fs/ntfs/attrib.h b/fs/ntfs/attrib.h index e2224fbfaabe..6b4fa9f57640 100644 --- a/fs/ntfs/attrib.h +++ b/fs/ntfs/attrib.h @@ -112,7 +112,13 @@ int ntfs_non_resident_attr_punch_hole(struct ntfs_inode *ni, s64 start_vcn, s64 int __ntfs_attr_truncate_vfs(struct ntfs_inode *ni, const s64 newsize, const s64 i_size); int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 prealloc_size); +int ntfs_attr_expand_locked(struct ntfs_inode *ni, const s64 newsize, + const s64 prealloc_size, + struct ntfs_inode *locked_ni); int ntfs_attr_truncate_i(struct ntfs_inode *ni, const s64 newsize, unsigned int holes); +int ntfs_attr_truncate_i_locked(struct ntfs_inode *ni, const s64 newsize, + unsigned int holes, + struct ntfs_inode *locked_ni); int ntfs_attr_truncate(struct ntfs_inode *ni, const s64 newsize); int ntfs_attr_rm(struct ntfs_inode *ni); int ntfs_attr_exist(struct ntfs_inode *ni, const __le32 type, __le16 *name, @@ -133,6 +139,9 @@ int ntfs_resident_attr_record_add(struct ntfs_inode *ni, __le32 type, __le16 *name, u8 name_len, u8 *val, u32 size, __le16 flags); int ntfs_attr_update_mapping_pairs(struct ntfs_inode *ni, s64 from_vcn); +int ntfs_attr_update_mapping_pairs_locked(struct ntfs_inode *ni, + s64 from_vcn, + struct ntfs_inode *locked_ni); struct runlist_element *ntfs_attr_vcn_to_rl(struct ntfs_inode *ni, s64 vcn, s64 *lcn); /* diff --git a/fs/ntfs/attrlist.c b/fs/ntfs/attrlist.c index be3086d34338..bb191953dcb1 100644 --- a/fs/ntfs/attrlist.c +++ b/fs/ntfs/attrlist.c @@ -12,6 +12,9 @@ #include "mft.h" #include "attrib.h" #include "attrlist.h" +#include "lcnalloc.h" + +#define NTFS_MAX_ATTR_LIST_SIZE (256 * 1024) /* * ntfs_attrlist_need - check whether inode need attribute list @@ -51,11 +54,155 @@ int ntfs_attrlist_need(struct ntfs_inode *ni) return 0; } -int ntfs_attrlist_update(struct ntfs_inode *base_ni) +/* + * Repack the $MFT/$ATTRIBUTE_LIST data into one run. + * + * The mapping pairs for an $ATTRIBUTE_LIST must remain in the base MFT + * record. Once that record has no room left, extending a fragmented list + * can require one more mapping-pairs byte than the record can hold. There + * is no attribute that can legally be moved out in that state: $STANDARD_ + * INFORMATION, $ATTRIBUTE_LIST, and the first $MFT/$DATA extent all have to + * stay in the base record. Move the list data to one contiguous run. The + * caller supplies the minimum allocation size so a recovery can use the + * smallest useful run while normal updates can still request the maximum + * legal list size as a reserve. + */ +static int ntfs_attrlist_repack(struct inode *attr_vi, + struct ntfs_inode *attr_ni, s64 min_alloc_size, + struct ntfs_inode *locked_ni) +{ + struct ntfs_volume *vol = attr_ni->vol; + struct runlist_element *old_rl, *new_rl; + u8 *data = NULL; + s64 data_size, alloc_size, nr_clusters, written; + s64 old_alloc_size; + size_t old_rl_count, new_rl_count; + unsigned long flags; + int err, restore_err; + if (attr_ni->mft_no != FILE_MFT || !NInoNonResident(attr_ni) || + min_alloc_size < 0) + return -EINVAL; + /* The buffered I/O below can reacquire the attribute runlist lock. */ + if (attr_ni == locked_ni) + return -ENOSPC; + + err = ntfs_attr_map_whole_runlist(attr_ni); + if (err) + return err; + + data_size = attr_ni->data_size; + if (data_size < 0) + return -EIO; + + if (data_size) { + data = kvmalloc(data_size, GFP_NOFS); + if (!data) + return -ENOMEM; + + written = ntfs_inode_attr_pread(attr_vi, 0, data_size, data); + if (written != data_size) { + err = written < 0 ? (int)written : -EIO; + goto out_free_data; + } + } + + old_alloc_size = attr_ni->allocated_size; + alloc_size = max_t(s64, old_alloc_size, min_alloc_size); + nr_clusters = ntfs_bytes_to_cluster(vol, + alloc_size + vol->cluster_size - 1); + if (nr_clusters <= 0) { + err = -EFBIG; + goto out_free_data; + } + + /* A single run keeps the mapping pairs at the minimum size. */ + new_rl = ntfs_cluster_alloc(vol, 0, nr_clusters, -1, DATA_ZONE, + true, true, false); + if (IS_ERR(new_rl)) { + err = PTR_ERR(new_rl); + goto out_free_data; + } + + new_rl_count = 0; + if (new_rl->vcn == 0 && new_rl->length == nr_clusters && + !new_rl[1].length) + new_rl_count = 2; + + if (new_rl_count != 2) { + ntfs_cluster_free_from_rl(vol, new_rl); + kvfree(new_rl); + err = -ENOSPC; + goto out_free_data; + } + old_rl = attr_ni->runlist.rl; + old_rl_count = attr_ni->runlist.count; + down_write(&attr_ni->runlist.lock); + attr_ni->runlist.rl = new_rl; + attr_ni->runlist.count = new_rl_count; + up_write(&attr_ni->runlist.lock); + + write_lock_irqsave(&attr_ni->size_lock, flags); + attr_ni->allocated_size = ntfs_cluster_to_bytes(vol, nr_clusters); + write_unlock_irqrestore(&attr_ni->size_lock, flags); + + /* Populate the replacement extent before publishing its mapping pairs. */ + if (data_size) { + written = ntfs_inode_attr_pwrite(attr_vi, 0, data_size, data, true); + if (written != data_size) { + err = written < 0 ? (int)written : -EIO; + goto restore_old_runlist; + } + } + + err = ntfs_attr_update_mapping_pairs_locked(attr_ni, 0, locked_ni); + if (err) + goto restore_old_runlist; + + /* The new mapping is now authoritative; release the old data runs. */ + if (ntfs_cluster_free_from_rl(vol, old_rl)) { + ntfs_error(vol->sb, + "Failed to free old ATTRIBUTE_LIST extent: inode %#llx", + (long long)attr_ni->mft_no); + NVolSetErrors(vol); + } + kvfree(old_rl); + kvfree(data); + return 0; + +restore_old_runlist: + down_write(&attr_ni->runlist.lock); + attr_ni->runlist.rl = old_rl; + attr_ni->runlist.count = old_rl_count; + up_write(&attr_ni->runlist.lock); + + write_lock_irqsave(&attr_ni->size_lock, flags); + attr_ni->allocated_size = old_alloc_size; + write_unlock_irqrestore(&attr_ni->size_lock, flags); + + restore_err = ntfs_attr_update_mapping_pairs_locked( + attr_ni, 0, locked_ni); + if (restore_err) { + ntfs_error(vol->sb, "Failed to restore ATTRIBUTE_LIST mapping pairs (%d)", + restore_err); + NVolSetErrors(vol); + } + + ntfs_cluster_free_from_rl(vol, new_rl); + kvfree(new_rl); + err = err ? err : restore_err; + +out_free_data: + kvfree(data); + return err; +} + +int ntfs_attrlist_update_locked(struct ntfs_inode *base_ni, + struct ntfs_inode *locked_ni) { struct inode *attr_vi; struct ntfs_inode *attr_ni; - int err; + s64 written; + int err, retry_err; /* * generic_shutdown_super() clears SB_ACTIVE before evicting cached @@ -72,23 +219,66 @@ int ntfs_attrlist_update(struct ntfs_inode *base_ni) return err; } attr_ni = NTFS_I(attr_vi); + /* Truncation and page-cache writes can reacquire this runlist lock. */ + if (attr_ni == locked_ni) { + iput(attr_vi); + return -ENOSPC; + } - err = ntfs_attr_truncate_i(attr_ni, base_ni->attr_list_size, HOLES_NO); - if (err == -ENOSPC && attr_ni->mft_no == FILE_MFT) { - err = ntfs_attr_truncate(attr_ni, 0); - if (err || ntfs_attr_truncate_i(attr_ni, base_ni->attr_list_size, HOLES_NO) != 0) { + err = ntfs_attr_truncate_i_locked( + attr_ni, base_ni->attr_list_size, HOLES_NO, locked_ni); + if (err == -ENOSPC && attr_ni->mft_no == FILE_MFT && + NInoNonResident(attr_ni)) { + retry_err = ntfs_attrlist_repack(attr_vi, attr_ni, + base_ni->attr_list_size, locked_ni); + if (retry_err) { + ntfs_error(base_ni->vol->sb, "Failed to repack attribute list"); iput(attr_vi); + return retry_err; + } + + retry_err = ntfs_attr_truncate_i_locked( + attr_ni, base_ni->attr_list_size, + HOLES_NO, locked_ni); + if (retry_err) { ntfs_error(base_ni->vol->sb, - "Failed to truncate attribute list of inode %#llx", - (long long)base_ni->mft_no); - return -EIO; + "Failed to resize attribute list after repack"); + iput(attr_vi); + return retry_err; } } else if (err) { iput(attr_vi); ntfs_error(base_ni->vol->sb, "Failed to truncate attribute list of inode %#llx", (long long)base_ni->mft_no); - return -EIO; + return err; + } + + /* + * Reserve the maximum legal list size while the MFT metadata area is + * still easy to allocate contiguously. This prevents a later list entry + * from needing another mapping-pairs byte in the full base MFT record. + * Failure to obtain the optional reserve must not reject the current + * metadata update; the repack retry above remains available if needed. + */ + if (base_ni->mft_no == FILE_MFT && NInoNonResident(attr_ni) && + attr_ni->allocated_size < NTFS_MAX_ATTR_LIST_SIZE) { + retry_err = ntfs_attr_expand_locked( + attr_ni, base_ni->attr_list_size, + NTFS_MAX_ATTR_LIST_SIZE, locked_ni); + if (retry_err == -ENOSPC) { + retry_err = ntfs_attrlist_repack( + attr_vi, attr_ni, + NTFS_MAX_ATTR_LIST_SIZE, locked_ni); + if (retry_err == -ENOSPC) + retry_err = 0; + } + if (retry_err) { + ntfs_error(base_ni->vol->sb, + "Failed to reserve attribute list space"); + iput(attr_vi); + return retry_err; + } } i_size_write(attr_vi, base_ni->attr_list_size); @@ -96,14 +286,15 @@ int ntfs_attrlist_update(struct ntfs_inode *base_ni) if (NInoNonResident(attr_ni) && !NInoAttrListNonResident(base_ni)) NInoSetAttrListNonResident(base_ni); - if (ntfs_inode_attr_pwrite(attr_vi, 0, base_ni->attr_list_size, - base_ni->attr_list, false) != - base_ni->attr_list_size) { + written = ntfs_inode_attr_pwrite(attr_vi, 0, base_ni->attr_list_size, + base_ni->attr_list, false); + if (written != base_ni->attr_list_size) { + err = written < 0 ? (int)written : -EIO; iput(attr_vi); ntfs_error(base_ni->vol->sb, "Failed to write attribute list of inode %#llx", (long long)base_ni->mft_no); - return -EIO; + return err; } NInoSetAttrListDirty(base_ni); @@ -111,6 +302,11 @@ int ntfs_attrlist_update(struct ntfs_inode *base_ni) return 0; } +int ntfs_attrlist_update(struct ntfs_inode *base_ni) +{ + return ntfs_attrlist_update_locked(base_ni, NULL); +} + /* * ntfs_attrlist_entry_add - add an attribute list attribute entry * @ni: opened ntfs inode, which contains that attribute diff --git a/fs/ntfs/attrlist.h b/fs/ntfs/attrlist.h index 1892a3934d3a..10cc2cc8e208 100644 --- a/fs/ntfs/attrlist.h +++ b/fs/ntfs/attrlist.h @@ -16,5 +16,7 @@ int ntfs_attrlist_need(struct ntfs_inode *ni); int ntfs_attrlist_entry_add(struct ntfs_inode *ni, struct attr_record *attr); int ntfs_attrlist_entry_rm(struct ntfs_attr_search_ctx *ctx); int ntfs_attrlist_update(struct ntfs_inode *base_ni); +int ntfs_attrlist_update_locked(struct ntfs_inode *base_ni, + struct ntfs_inode *locked_ni); #endif /* defined _NTFS_ATTRLIST_H */ diff --git a/fs/ntfs/compress.c b/fs/ntfs/compress.c index 99a3ea2b5c55..075b57fc1de6 100644 --- a/fs/ntfs/compress.c +++ b/fs/ntfs/compress.c @@ -1450,7 +1450,7 @@ static int ntfs_write_cb(struct ntfs_inode *ni, loff_t pos, struct page **pages, ni->runlist.rl = rl; rlc = NULL; - err = ntfs_attr_update_mapping_pairs(ni, 0); + err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni); up_write(&ni->runlist.lock); if (err) err = -EIO; diff --git a/fs/ntfs/file.c b/fs/ntfs/file.c index 8164326b7812..007d1614b9ac 100644 --- a/fs/ntfs/file.c +++ b/fs/ntfs/file.c @@ -111,7 +111,8 @@ static int ntfs_trim_prealloc(struct inode *vi) ntfs_error(vol->sb, "Preallocated block rollback failed"); } else { ni->allocated_size = ntfs_cluster_to_bytes(vol, vcn_tr); - err = ntfs_attr_update_mapping_pairs(ni, 0); + err = ntfs_attr_update_mapping_pairs_locked( + ni, 0, ni); if (err) ntfs_error(vol->sb, "Failed to rollback mapping pairs for prealloc"); diff --git a/fs/ntfs/inode.c b/fs/ntfs/inode.c index 5aedc045f65a..a777de8a80c7 100644 --- a/fs/ntfs/inode.c +++ b/fs/ntfs/inode.c @@ -170,17 +170,19 @@ struct inode *ntfs_iget(struct super_block *sb, u64 mft_no) /* If this is a freshly allocated inode, need to read it now. */ if (inode_state_read_once(vi) & I_NEW) { err = ntfs_read_locked_inode(vi); - unlock_new_inode(vi); + if (err) { + remove_inode_hash(vi); + discard_new_inode(vi); + } else + unlock_new_inode(vi); } /* * There is no point in keeping bad inodes around. This also * simplifies things in that we never need to check for bad inodes * elsewhere. */ - if (unlikely(err)) { - iput(vi); + if (unlikely(err)) vi = ERR_PTR(err); - } return vi; } @@ -231,17 +233,19 @@ struct inode *ntfs_attr_iget(struct inode *base_vi, __le32 type, /* If this is a freshly allocated inode, need to read it now. */ if (inode_state_read_once(vi) & I_NEW) { err = ntfs_read_locked_attr_inode(base_vi, vi); - unlock_new_inode(vi); + if (err) { + remove_inode_hash(vi); + discard_new_inode(vi); + } else + unlock_new_inode(vi); } /* * There is no point in keeping bad attribute inodes around. This also * simplifies things in that we never need to check for bad attribute * inodes elsewhere. */ - if (unlikely(err)) { - iput(vi); + if (unlikely(err)) vi = ERR_PTR(err); - } return vi; } @@ -286,17 +290,19 @@ struct inode *ntfs_index_iget(struct inode *base_vi, __le16 *name, /* If this is a freshly allocated inode, need to read it now. */ if (inode_state_read_once(vi) & I_NEW) { err = ntfs_read_locked_index_inode(base_vi, vi); - unlock_new_inode(vi); + if (err) { + remove_inode_hash(vi); + discard_new_inode(vi); + } else + unlock_new_inode(vi); } /* * There is no point in keeping bad index inodes around. This also * simplifies things in that we never need to check for bad index * inodes elsewhere. */ - if (unlikely(err)) { - iput(vi); + if (unlikely(err)) vi = ERR_PTR(err); - } return vi; } @@ -1241,7 +1247,8 @@ unm_err_out: if (m) unmap_mft_record(ni); err_out: - if (err != -EOPNOTSUPP && err != -ENOMEM && vol_err == true) { + if (err != -EOPNOTSUPP && err != -ENOMEM && + err != -EINTR && err != -ERESTARTSYS && vol_err == true) { ntfs_error(vol->sb, "Failed with error code %i. Marking corrupt inode 0x%llx as bad. Run chkdsk.", err, ni->mft_no); @@ -1467,12 +1474,13 @@ unm_err_out: ntfs_attr_put_search_ctx(ctx); unmap_mft_record(base_ni); err_out: - if (err != -ENOENT) + if (err != -ENOENT && err != -EINTR && err != -ERESTARTSYS) ntfs_error(vol->sb, "Failed with error code %i while reading attribute inode (mft_no 0x%llx, type 0x%x, name_len %i). Marking corrupt inode and base inode 0x%llx as bad. Run chkdsk.", err, ni->mft_no, ni->type, ni->name_len, base_ni->mft_no); - if (err != -ENOENT && err != -ENOMEM) + if (err != -ENOENT && err != -ENOMEM && + err != -EINTR && err != -ERESTARTSYS) NVolSetErrors(vol); return err; } @@ -1676,8 +1684,9 @@ static int ntfs_read_locked_index_inode(struct inode *base_vi, struct inode *vi) /* Get the index bitmap attribute inode. */ bvi = ntfs_attr_iget(base_vi, AT_BITMAP, ni->name, ni->name_len); if (IS_ERR(bvi)) { - ntfs_error(vi->i_sb, "Failed to get bitmap attribute."); err = PTR_ERR(bvi); + if (err != -EINTR && err != -ERESTARTSYS) + ntfs_error(vi->i_sb, "Failed to get bitmap attribute."); goto unm_err_out; } bni = NTFS_I(bvi); @@ -1721,10 +1730,12 @@ unm_err_out: if (m) unmap_mft_record(base_ni); err_out: - ntfs_error(vi->i_sb, - "Failed with error code %i while reading index inode (mft_no 0x%llx, name_len %i.", - err, ni->mft_no, ni->name_len); - if (err != -EOPNOTSUPP && err != -ENOMEM) + if (err != -EINTR && err != -ERESTARTSYS) + ntfs_error(vi->i_sb, + "Failed with error code %i while reading index inode (mft_no 0x%llx, name_len %i.", + err, ni->mft_no, ni->name_len); + if (err != -EOPNOTSUPP && err != -ENOMEM && + err != -EINTR && err != -ERESTARTSYS) NVolSetErrors(vol); return err; } @@ -2772,7 +2783,7 @@ int __ntfs_write_inode(struct inode *vi, int sync) if (NInoNonResident(ni) && NInoRunlistDirty(ni)) { down_write(&ni->runlist.lock); - err = ntfs_attr_update_mapping_pairs(ni, 0); + err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni); if (!err) NInoClearRunlistDirty(ni); up_write(&ni->runlist.lock); @@ -3713,7 +3724,7 @@ static s64 __ntfs_inode_non_resident_attr_pwrite(struct inode *vi, FGP_CREAT | FGP_LOCK, mapping_gfp_mask(mapping)); if (IS_ERR(folio)) { - ret = -ENOMEM; + ret = PTR_ERR(folio); break; } } else { @@ -3745,6 +3756,7 @@ static s64 __ntfs_inode_non_resident_attr_pwrite(struct inode *vi, u64 rl_length = 0; s64 vcn; struct runlist_element *rl; + int bio_err; lcn_count = max_t(s64, 1, ntfs_bytes_to_cluster(vol, attr_len)); vcn = ntfs_pidx_to_cluster(vol, folio->index); @@ -3787,8 +3799,15 @@ static s64 __ntfs_inode_non_resident_attr_pwrite(struct inode *vi, goto err_unlock_folio; } - submit_bio_wait(bio); + bio_err = submit_bio_wait(bio); bio_put(bio); + if (bio_err) { + ntfs_error(vi->i_sb, + "Synchronous attribute write failed (%d)", + bio_err); + ret = bio_err; + goto err_unlock_folio; + } vcn += rl_length; offset += length; } while (lcn_count != 0); diff --git a/fs/ntfs/mft.c b/fs/ntfs/mft.c index 98ab686a5ea2..4b7449495375 100644 --- a/fs/ntfs/mft.c +++ b/fs/ntfs/mft.c @@ -213,7 +213,8 @@ struct mft_record *map_mft_record(struct ntfs_inode *ni) return m; atomic_dec(&ni->count); - ntfs_error(ni->vol->sb, "Failed with error code %lu.", -PTR_ERR(m)); + if (PTR_ERR(m) != -EINTR && PTR_ERR(m) != -ERESTARTSYS) + ntfs_error(ni->vol->sb, "Failed with error code %lu.", -PTR_ERR(m)); return m; } @@ -462,7 +463,7 @@ int ntfs_sync_mft_mirror(struct ntfs_volume *vol, const u64 mft_no, { u8 *kmirr; struct folio *folio; - unsigned int folio_ofs, lcn_folio_off = 0; + unsigned int folio_ofs; int err = 0; struct bio *bio; @@ -492,15 +493,11 @@ int ntfs_sync_mft_mirror(struct ntfs_volume *vol, const u64 mft_no, memcpy(kmirr, m, vol->mft_record_size); kunmap_local(kmirr); - if (vol->cluster_size_bits > PAGE_SHIFT) { - lcn_folio_off = folio->index << PAGE_SHIFT; - lcn_folio_off &= vol->cluster_size_mask; - } - bio = bio_alloc(vol->sb->s_bdev, 1, REQ_OP_WRITE, GFP_NOIO); bio->bi_iter.bi_sector = ntfs_bytes_to_bio_sector(NTFS_CLU_TO_B(vol, vol->mftmirr_lcn) + - lcn_folio_off + folio_ofs); + ((u64)folio->index << PAGE_SHIFT) + + folio_ofs); if (bio_add_folio(bio, folio, vol->mft_record_size, folio_ofs)) err = submit_bio_wait(bio); @@ -841,12 +838,126 @@ static bool ntfs_may_write_mft_record(struct ntfs_volume *vol, const u64 mft_no, static const char *es = " Leaving inconsistent metadata. Unmount and run chkdsk."; -#define RESERVED_MFT_RECORDS 64 +#define FIRST_NORMAL_MFT_RECORD 24 +#define MFT_RECORD_RESERVE 4 /* - * ntfs_mft_bitmap_find_and_alloc_free_rec_nolock - see name + * Records 12-15 are marked in use by Windows but normally have no name + * and no links. Keep them as the last bootstrap option when a volume + * mounted without an in-memory tail reserve needs its first $MFT metadata + * extent. + */ +static bool mft_reserved_is_free(struct ntfs_volume *vol, + struct ntfs_inode *mft_ni, s64 mft_no) +{ + struct attr_record *a; + struct mft_record *m; + struct folio *folio; + void *mapped; + pgoff_t index = NTFS_MFT_NR_TO_PIDX(vol, mft_no); + unsigned int ofs = NTFS_MFT_NR_TO_POFS(vol, mft_no); + u32 attrs_offset, bytes_in_use; + bool available = false, have_std = false; + int i; + + for (i = 0; i < mft_ni->nr_extents; i++) { + if (mft_ni->ext.extent_ntfs_inos[i] && + mft_ni->ext.extent_ntfs_inos[i]->mft_no == mft_no) + return false; + } + m = kmalloc(vol->mft_record_size, GFP_NOFS); + if (!m) + return false; + + folio = read_mapping_folio(vol->mft_ino->i_mapping, index, NULL); + if (IS_ERR(folio)) + goto free_m; + + folio_lock(folio); + mapped = kmap_local_folio(folio, 0); + memcpy(m, (u8 *)mapped + ofs, vol->mft_record_size); + kunmap_local(mapped); + folio_unlock(folio); + folio_put(folio); + if (post_read_mst_fixup((struct ntfs_record *)m, vol->mft_record_size)) + goto free_m; + + if (!ntfs_is_mft_record(m->magic) || + !(m->flags & MFT_RECORD_IN_USE) || m->base_mft_record || + m->link_count) + goto out; + + attrs_offset = le16_to_cpu(m->attrs_offset); + bytes_in_use = le32_to_cpu(m->bytes_in_use); + if (attrs_offset > bytes_in_use || bytes_in_use > vol->mft_record_size || + bytes_in_use - attrs_offset < sizeof(a->type)) + goto out; + + for (a = (struct attr_record *)((u8 *)m + attrs_offset); + (u8 *)a + sizeof(a->type) <= (u8 *)m + bytes_in_use;) { + u32 len; + + if (a->type == AT_END) { + if ((u8 *)a + sizeof(a->type) + sizeof(a->length) > + (u8 *)m + bytes_in_use) + break; + /* Also accept a record emptied by an earlier bootstrap. */ + available = have_std || + (u8 *)a == (u8 *)m + attrs_offset; + break; + } + if (a->type == AT_FILE_NAME) + break; + len = le32_to_cpu(a->length); + if (len < offsetof(struct attr_record, data) || + (u8 *)a + len > (u8 *)m + bytes_in_use) + break; + if (a->type == AT_STANDARD_INFORMATION) { + u32 value_len, value_ofs; + + if (have_std || a->non_resident || + len < offsetof(struct attr_record, + data.resident.reserved) + 1) + break; + value_len = le32_to_cpu(a->data.resident.value_length); + value_ofs = le16_to_cpu(a->data.resident.value_offset); + if (value_ofs > len || value_len > len - value_ofs) + break; + have_std = true; + } + a = (struct attr_record *)((u8 *)a + len); + } +out: + kfree(m); + return available; +free_m: + kfree(m); + return false; +} + +static s64 mft_reserve_end(const u8 *buf, s64 buf_start, s64 buf_end, + s64 start, s64 pass_end, s64 initialized_mft_records) +{ + s64 end = start + 1; + s64 limit = min_t(s64, start + MFT_RECORD_RESERVE, pass_end); + + if (limit > initialized_mft_records) + limit = initialized_mft_records; + if (limit > buf_end) + limit = buf_end; + while (end < limit && + !(buf[(end - buf_start) >> 3] & + (1 << ((end - buf_start) & 7)))) + end++; + return end; +} + +/* + * mft_bitmap_alloc_free_rec - find and allocate a free MFT record * @vol: volume on which to search for a free mft record * @base_ni: open base inode if allocating an extent mft record or NULL + * @max_mft_no: first record which must not be allocated, or -1 + * @new_reserve_end: if not NULL, end of a free run starting after the result * * Search for a free mft record in the mft bitmap attribute on the ntfs volume * @vol. @@ -862,10 +973,12 @@ static const char *es = " Leaving inconsistent metadata. Unmount and run chkds * * Locking: Caller must hold vol->mftbmp_lock for writing. */ -static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vol, - struct ntfs_inode *base_ni) +static s64 mft_bitmap_alloc_free_rec(struct ntfs_volume *vol, + struct ntfs_inode *base_ni, + s64 max_mft_no, s64 *new_reserve_end) { s64 pass_end, ll, data_pos, pass_start, ofs, bit; + s64 initialized_mft_records; unsigned long flags; struct address_space *mftbmp_mapping; u8 *buf = NULL, *byte; @@ -882,30 +995,36 @@ static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vo read_lock_irqsave(&NTFS_I(vol->mft_ino)->size_lock, flags); pass_end = NTFS_I(vol->mft_ino)->allocated_size >> vol->mft_record_size_bits; + initialized_mft_records = NTFS_I(vol->mft_ino)->initialized_size >> + vol->mft_record_size_bits; read_unlock_irqrestore(&NTFS_I(vol->mft_ino)->size_lock, flags); read_lock_irqsave(&NTFS_I(vol->mftbmp_ino)->size_lock, flags); ll = NTFS_I(vol->mftbmp_ino)->initialized_size << 3; read_unlock_irqrestore(&NTFS_I(vol->mftbmp_ino)->size_lock, flags); if (pass_end > ll) pass_end = ll; - pass = 1; - if (!base_ni) - data_pos = vol->mft_data_pos; - else - data_pos = base_ni->mft_no + 1; - if (data_pos < RESERVED_MFT_RECORDS) - data_pos = RESERVED_MFT_RECORDS; - if (data_pos >= pass_end) { - data_pos = RESERVED_MFT_RECORDS; + if (max_mft_no >= 0 && pass_end > max_mft_no) + pass_end = max_mft_no; + if (base_ni && base_ni->mft_no == FILE_MFT) { + data_pos = FILE_first_user; pass = 2; - /* This happens on a freshly formatted volume. */ if (data_pos >= pass_end) return -ENOSPC; - } - - if (base_ni && base_ni->mft_no == FILE_MFT) { - data_pos = 0; - pass = 2; + } else { + pass = 1; + if (!base_ni) + data_pos = vol->mft_data_pos; + else + data_pos = base_ni->mft_no + 1; + if (data_pos < FIRST_NORMAL_MFT_RECORD) + data_pos = FIRST_NORMAL_MFT_RECORD; + if (data_pos >= pass_end) { + data_pos = FIRST_NORMAL_MFT_RECORD; + pass = 2; + /* This happens on a freshly formatted volume. */ + if (data_pos >= pass_end) + return -ENOSPC; + } } pass_start = data_pos; @@ -940,38 +1059,28 @@ static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vo size, data_pos, bit); for (; bit < size && data_pos + bit < pass_end; bit &= ~7ull, bit += 8) { - /* - * If we're extending $MFT and running out of the first - * mft record (base record) then give up searching since - * no guarantee that the found record will be accessible. - */ - if (base_ni && base_ni->mft_no == FILE_MFT && bit > 400) { - folio_unlock(folio); - kunmap_local(buf); - folio_put(folio); - return -ENOSPC; - } - byte = buf + (bit >> 3); if (*byte == 0xff) continue; - b = ffz((unsigned long)*byte); - if (b < 8 && b >= (bit & 7)) { + b = bit & 7; + for (; b < 8; b++) { + if (*byte & (1 << b)) + continue; ll = data_pos + (bit & ~7ull) + b; + if (ll >= pass_end) + break; + /* Keep the dynamic tail reserve for $MFT metadata. */ + if ((!base_ni || base_ni->mft_no != FILE_MFT) && + ll >= vol->mft_record_reserve_pos && + ll < vol->mft_record_reserve_end) + continue; if (unlikely(ll >= (1ll << 32))) { folio_unlock(folio); kunmap_local(buf); folio_put(folio); return -ENOSPC; } - *byte |= 1 << b; - folio_mark_dirty(folio); - folio_unlock(folio); - kunmap_local(buf); - folio_put(folio); - ntfs_debug("Done. (Found and allocated mft record 0x%llx.)", - ll); - return ll; + goto found; } } ntfs_debug("After inner for loop: size 0x%x, data_pos 0x%llx, bit 0x%llx", @@ -994,7 +1103,8 @@ static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vo * part of the zone which we omitted earlier. */ pass_end = pass_start; - data_pos = pass_start = RESERVED_MFT_RECORDS; + data_pos = FIRST_NORMAL_MFT_RECORD; + pass_start = FIRST_NORMAL_MFT_RECORD; ntfs_debug("pass %i, pass_start 0x%llx, pass_end 0x%llx.", pass, pass_start, pass_end); if (data_pos >= pass_end) @@ -1004,9 +1114,22 @@ static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vo /* No free mft records in currently initialized mft bitmap. */ ntfs_debug("Done. (No free mft records left in currently initialized mft bitmap.)"); return -ENOSPC; +found: + if (new_reserve_end) + *new_reserve_end = mft_reserve_end(buf, data_pos, + data_pos + size, ll, pass_end, + initialized_mft_records); + *byte |= 1 << b; + folio_mark_dirty(folio); + folio_unlock(folio); + kunmap_local(buf); + folio_put(folio); + ntfs_debug("Done. (Found and allocated mft record 0x%llx.)", ll); + return ll; } -static int ntfs_mft_attr_extend(struct ntfs_inode *ni) +static int ntfs_mft_attr_extend(struct ntfs_inode *ni, + struct ntfs_inode *locked_ni) { int ret = 0; struct ntfs_inode *base_ni; @@ -1027,7 +1150,7 @@ static int ntfs_mft_attr_extend(struct ntfs_inode *ni) } } - ret = ntfs_attr_update_mapping_pairs(ni, 0); + ret = ntfs_attr_update_mapping_pairs_locked(ni, 0, locked_ni); if (ret) pr_err("MP update failed\n"); @@ -1215,7 +1338,7 @@ static int ntfs_mft_bitmap_extend_allocation_nolock(struct ntfs_volume *vol) ret = ntfs_attr_record_resize(ctx->mrec, a, mp_size + le16_to_cpu(a->data.non_resident.mapping_pairs_offset)); if (unlikely(ret)) { - ret = ntfs_mft_attr_extend(mftbmp_ni); + ret = ntfs_mft_attr_extend(mftbmp_ni, mftbmp_ni); if (!ret) goto extended_ok; if (ret != -EAGAIN) @@ -1326,7 +1449,9 @@ undo_alloc: NVolSetErrors(vol); } mark_mft_record_dirty(ctx->ntfs_ino); - } else if (status.mp_extended && ntfs_attr_update_mapping_pairs(mftbmp_ni, 0)) { + } else if (status.mp_extended && + ntfs_attr_update_mapping_pairs_locked(mftbmp_ni, 0, + mftbmp_ni)) { ntfs_error(vol->sb, "Failed to restore mapping pairs.%s", es); NVolSetErrors(vol); } @@ -1414,7 +1539,6 @@ static int ntfs_mft_bitmap_extend_initialized_nolock(struct ntfs_volume *vol) ret = ntfs_attr_set(mftbmp_ni, old_initialized_size, 8, 0); if (likely(!ret)) { ntfs_debug("Done. (Wrote eight initialized bytes to mft bitmap."); - ntfs_inc_free_mft_records(vol, 8 * 8); return 0; } ntfs_error(vol->sb, "Failed to write to mft bitmap."); @@ -1471,8 +1595,9 @@ err_out: * @vol: volume on which to extend the mft data attribute * * Extend the mft data attribute on the ntfs volume @vol by 16 mft records - * worth of clusters or if not enough space for this by one mft record worth - * of clusters. + * worth of clusters or if not enough space for this by two mft records worth + * of clusters. Keeping at least two new records breaks the recursion between + * extending $MFT and allocating a record for a new $MFT attribute extent. * * Note: Only changes allocated_size, i.e. does not touch initialized_size or * data_size. @@ -1526,10 +1651,8 @@ static int ntfs_mft_data_extend_allocation_nolock(struct ntfs_volume *vol) } lcn = rl->lcn + rl->length; ntfs_debug("Last lcn of mft data attribute is 0x%llx.", lcn); - /* Minimum allocation is one mft record worth of clusters. */ - min_nr = NTFS_B_TO_CLU(vol, vol->mft_record_size); - if (!min_nr) - min_nr = 1; + /* Keep room for the allocating record and at least one MFT reserve. */ + min_nr = DIV_ROUND_UP_ULL((u64)vol->mft_record_size * 2, vol->cluster_size); /* Want to allocate 16 mft records worth of clusters. */ nr = vol->mft_record_size << 4 >> vol->cluster_size_bits; if (!nr) @@ -1653,7 +1776,7 @@ static int ntfs_mft_data_extend_allocation_nolock(struct ntfs_volume *vol) ret = ntfs_attr_record_resize(ctx->mrec, a, mp_size + le16_to_cpu(a->data.non_resident.mapping_pairs_offset)); if (unlikely(ret)) { - ret = ntfs_mft_attr_extend(mft_ni); + ret = ntfs_mft_attr_extend(mft_ni, NULL); if (!ret) goto extended_ok; if (ret != -EAGAIN) @@ -1928,6 +2051,7 @@ static int ntfs_mft_record_format(const struct ntfs_volume *vol, const s64 mft_n * @ni: [OUT] on success, set to the allocated ntfs inode * @base_ni: [IN] open base inode if allocating an extent mft record or NULL * @ni_mrec: [OUT] on successful return this is the mapped mft record + * @mft_data_vcn: [IN] lowest VCN of a new $MFT/$DATA extent, or -1 * * Allocate an mft record in $MFT/$DATA of an open ntfs volume @vol. * @@ -1955,30 +2079,23 @@ static int ntfs_mft_record_format(const struct ntfs_volume *vol, const s64 mft_n * optimize this we start scanning at the place specified by @base_ni or if * @base_ni is NULL we start where we last stopped and we perform wrap around * when we reach the end. Note, we do not try to allocate mft records below - * number 64 because numbers 0 to 15 are the defined system files anyway and 16 - * to 64 are special in that they are used for storing extension mft records - * for the $DATA attribute of $MFT. This is required to avoid the possibility - * of creating a runlist with a circular dependency which once written to disk - * can never be read in again. Windows will only use records 16 to 24 for - * normal files if the volume is completely out of space. We never use them - * which means that when the volume is really out of space we cannot create any - * more files while Windows can still create up to 8 small files. We can start - * doing this at some later time, it does not matter much for now. + * number 24 because numbers 0 to 15 are the defined system files and records + * 16 to 23 are kept for metadata compatibility. Records reserved dynamically + * at the initialized MFT tail are skipped by normal allocation and consumed by + * $MFT metadata extent allocation. * * When scanning the mft bitmap, we only search up to the last allocated mft - * record. If there are no free records left in the range 64 to number of + * record. If there are no free records left in the range 24 to number of * allocated mft records, then we extend the $MFT/$DATA attribute in order to * create free mft records. We extend the allocated size of $MFT/$DATA by 16 * records at a time or one cluster, if cluster size is above 16kiB. If there - * is not sufficient space to do this, we try to extend by a single mft record - * or one cluster, if cluster size is above the mft record size. + * is not sufficient space to do this, we try to extend by two mft records or + * one cluster, if a cluster already contains at least two mft records. * - * No matter how many mft records we allocate, we initialize only the first - * allocated mft record, incrementing mft data size and initialized size - * accordingly, open an struct ntfs_inode for it and return it to the caller, unless - * there are less than 64 mft records, in which case we allocate and initialize - * mft records until we reach record 64 which we consider as the first free mft - * record for use by normal files. + * When extending the initialized MFT tail, we also initialize up to four + * additional records and reserve them in memory for future $MFT metadata + * extents. If there are less than 24 mft records, records are initialized + * until record 24, which is the first record used for normal files. * * If during any stage we overflow the initialized data in the mft bitmap, we * extend the initialized size (and data size) by 8 bytes, allocating another @@ -2014,9 +2131,13 @@ static int ntfs_mft_record_format(const struct ntfs_volume *vol, const s64 mft_n */ int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode, struct ntfs_inode **ni, struct ntfs_inode *base_ni, - struct mft_record **ni_mrec) + struct mft_record **ni_mrec, const s64 mft_data_vcn) { s64 ll, bit, old_data_initialized, old_data_size; + s64 nr_new_mft_records = 0; + s64 max_mft_no = -1, reserve_start = -1, reserve_end = -1; + s64 candidate_reserve_end = -1; + s64 *reserve_endp; unsigned long flags; struct folio *folio; struct ntfs_inode *mft_ni, *mftbmp_ni; @@ -2027,7 +2148,9 @@ int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode, unsigned int ofs; int err; __le16 seq_no, usn; - bool record_formatted = false; + bool record_formatted = false, from_reserve = false, tail_alloc = false; + bool reserve_created = false; + bool forced_reserved_record = false; unsigned int memalloc_flags; if (base_ni && *ni) @@ -2036,6 +2159,21 @@ int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode, /* @mode and @base_ni are mutually exclusive. */ if (mode && base_ni) return -EINVAL; + if (mft_data_vcn >= 0 && + (!base_ni || base_ni->mft_no != FILE_MFT)) + return -EINVAL; + if (mft_data_vcn >= 0) { + u64 vbo; + + if ((u64)mft_data_vcn > (U64_MAX >> vol->cluster_size_bits)) + return -EOVERFLOW; + vbo = (u64)mft_data_vcn << vol->cluster_size_bits; + /* + * The whole extent record must be reachable without this + * extent, including when an MFT record spans multiple clusters. + */ + max_mft_no = vbo >> vol->mft_record_size_bits; + } if (base_ni) ntfs_debug("Entering (allocating an extent mft record for base mft record 0x%llx).", @@ -2050,10 +2188,39 @@ int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode, mutex_lock(&mft_ni->mrec_lock); mftbmp_ni = NTFS_I(vol->mftbmp_ino); search_free_rec: + from_reserve = false; + reserve_created = false; + candidate_reserve_end = -1; if (!base_ni || base_ni->mft_no != FILE_MFT) down_write(&vol->mftbmp_lock); - bit = ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(vol, base_ni); + if (base_ni && base_ni->mft_no == FILE_MFT && + vol->mft_record_reserve_pos < vol->mft_record_reserve_end && + (max_mft_no < 0 || vol->mft_record_reserve_pos < max_mft_no)) { + bit = vol->mft_record_reserve_pos; + err = ntfs_bitmap_set_bit(vol->mftbmp_ino, bit); + if (unlikely(err)) { + ntfs_error(vol->sb, + "Failed to allocate reserved MFT record 0x%llx.", + bit); + goto err_out; + } + vol->mft_record_reserve_pos++; + from_reserve = true; + ntfs_debug("Allocated MFT metadata record 0x%llx from tail reserve.", + bit); + goto have_alloc_rec; + } + reserve_endp = vol->mft_record_reserve_pos >= + vol->mft_record_reserve_end ? &candidate_reserve_end : NULL; + bit = mft_bitmap_alloc_free_rec(vol, base_ni, max_mft_no, reserve_endp); if (bit >= 0) { + if (candidate_reserve_end > bit + 1) { + vol->mft_record_reserve_pos = bit + 1; + vol->mft_record_reserve_end = candidate_reserve_end; + reserve_created = true; + ntfs_debug("Reserved free MFT records [0x%llx, 0x%llx) for metadata.", + bit + 1, candidate_reserve_end); + } ntfs_debug("Found and allocated free record (#1), bit 0x%llx.", (long long)bit); goto have_alloc_rec; @@ -2068,6 +2235,24 @@ search_free_rec: } if (base_ni && base_ni->mft_no == FILE_MFT) { + static const u8 bootstrap_records[] = { + FILE_reserved15, FILE_reserved12, FILE_reserved13, + FILE_reserved14, + }; + int i; + + for (i = 0; i < ARRAY_SIZE(bootstrap_records); i++) { + if (max_mft_no >= 0 && bootstrap_records[i] >= max_mft_no) + continue; + if (!mft_reserved_is_free(vol, mft_ni, + bootstrap_records[i])) + continue; + bit = bootstrap_records[i]; + forced_reserved_record = true; + ntfs_debug("Using reserved MFT record %lld to bootstrap metadata extension.", + bit); + goto have_alloc_rec; + } memalloc_nofs_restore(memalloc_flags); return bit; } @@ -2087,10 +2272,10 @@ search_free_rec: old_data_initialized = mftbmp_ni->initialized_size; read_unlock_irqrestore(&mftbmp_ni->size_lock, flags); if (old_data_initialized << 3 > ll && - old_data_initialized > RESERVED_MFT_RECORDS / 8) { + old_data_initialized << 3 > FIRST_NORMAL_MFT_RECORD) { bit = ll; - if (bit < RESERVED_MFT_RECORDS) - bit = RESERVED_MFT_RECORDS; + if (bit < FIRST_NORMAL_MFT_RECORD) + bit = FIRST_NORMAL_MFT_RECORD; if (unlikely(bit >= (1ll << 32))) goto max_err_out; ntfs_debug("Found free record (#2), bit 0x%llx.", @@ -2176,6 +2361,11 @@ have_alloc_rec: read_lock_irqsave(&mft_ni->size_lock, flags); old_data_initialized = mft_ni->initialized_size; read_unlock_irqrestore(&mft_ni->size_lock, flags); + tail_alloc = (!base_ni || base_ni->mft_no != FILE_MFT) && + bit >= (old_data_initialized >> vol->mft_record_size_bits) && + vol->mft_record_reserve_pos >= vol->mft_record_reserve_end; + if (tail_alloc) + ll = (bit + 2) << vol->mft_record_size_bits; if (ll <= old_data_initialized) { ntfs_debug("Allocated mft record already initialized."); goto mft_rec_already_initialized; @@ -2208,6 +2398,29 @@ have_alloc_rec: mft_ni->initialized_size); } read_unlock_irqrestore(&mft_ni->size_lock, flags); + if (tail_alloc) { + s64 bitmap_records; + + read_lock_irqsave(&mft_ni->size_lock, flags); + reserve_end = mft_ni->allocated_size >> + vol->mft_record_size_bits; + read_unlock_irqrestore(&mft_ni->size_lock, flags); + read_lock_irqsave(&mftbmp_ni->size_lock, flags); + bitmap_records = mftbmp_ni->initialized_size << 3; + read_unlock_irqrestore(&mftbmp_ni->size_lock, flags); + if (reserve_end > bitmap_records) + reserve_end = bitmap_records; + if (reserve_end > bit + 1 + MFT_RECORD_RESERVE) + reserve_end = bit + 1 + MFT_RECORD_RESERVE; + reserve_start = bit + 1; + if (reserve_end > reserve_start) { + ll = reserve_end << vol->mft_record_size_bits; + } else { + reserve_start = -1; + reserve_end = -1; + ll = (bit + 1) << vol->mft_record_size_bits; + } + } } else if (ll > mft_ni->allocated_size) { err = -ENOSPC; goto undo_mftbmp_alloc_nolock; @@ -2276,14 +2489,25 @@ have_alloc_rec: mark_mft_record_dirty(ctx->ntfs_ino); ntfs_attr_put_search_ctx(ctx); unmap_mft_record(mft_ni); + if (reserve_start >= 0 && reserve_end > reserve_start) { + vol->mft_record_reserve_pos = reserve_start; + vol->mft_record_reserve_end = reserve_end; + ntfs_debug("Reserved MFT records [0x%llx, 0x%llx) for metadata.", + reserve_start, reserve_end); + } read_lock_irqsave(&mft_ni->size_lock, flags); ntfs_debug("Status of mft data after mft record initialization: allocated_size 0x%llx, data_size 0x%llx, initialized_size 0x%llx.", mft_ni->allocated_size, i_size_read(vol->mft_ino), mft_ni->initialized_size); WARN_ON(i_size_read(vol->mft_ino) > mft_ni->allocated_size); WARN_ON(mft_ni->initialized_size > i_size_read(vol->mft_ino)); + nr_new_mft_records = (i_size_read(vol->mft_ino) - old_data_size) >> + vol->mft_record_size_bits; read_unlock_irqrestore(&mft_ni->size_lock, flags); mft_rec_already_initialized: + /* Account for newly visible MFT records before dropping the lock. */ + if (nr_new_mft_records > 0) + ntfs_inc_free_mft_records(vol, nr_new_mft_records); /* * We can finally drop the mft bitmap lock as the mft data attribute * has been fully updated. The only disparity left is that the @@ -2315,8 +2539,8 @@ mft_rec_already_initialized: /* If we just formatted the mft record no need to do it again. */ if (!record_formatted) { /* Sanity check that the mft record is really not in use. */ - if (ntfs_is_file_record(m->magic) && - (m->flags & MFT_RECORD_IN_USE)) { + if (!forced_reserved_record && ntfs_is_file_record(m->magic) && + (m->flags & MFT_RECORD_IN_USE)) { ntfs_warning(vol->sb, "Mft record 0x%llx was marked free in mft bitmap but is marked used itself. Unmount and run chkdsk.", bit); @@ -2389,9 +2613,13 @@ mft_rec_already_initialized: ntfs_error(vol->sb, "Failed to map allocated extent mft record 0x%llx.", bit); err = PTR_ERR(m_tmp); - /* Set the mft record itself not in use. */ - m->flags &= cpu_to_le16( - ~le16_to_cpu(MFT_RECORD_IN_USE)); + if (forced_reserved_record) { + m->base_mft_record = 0; + m->flags |= MFT_RECORD_IN_USE; + } else { + /* Set the mft record itself not in use. */ + m->flags &= cpu_to_le16(~le16_to_cpu(MFT_RECORD_IN_USE)); + } /* Make sure the mft record is written out to disk. */ ntfs_mft_mark_dirty(folio); folio_unlock(folio); @@ -2463,7 +2691,8 @@ mft_rec_already_initialized: (*ni)->mft_no = bit; if (ni_mrec) *ni_mrec = (*ni)->mrec; - ntfs_dec_free_mft_records(vol, 1); + if (!forced_reserved_record) + ntfs_dec_free_mft_records(vol, 1); return 0; undo_data_init: write_lock_irqsave(&mft_ni->size_lock, flags); @@ -2475,10 +2704,13 @@ undo_mftbmp_alloc: if (!base_ni || base_ni->mft_no != FILE_MFT) down_write(&vol->mftbmp_lock); undo_mftbmp_alloc_nolock: - if (ntfs_bitmap_clear_bit(vol->mftbmp_ino, bit)) { + if (!forced_reserved_record && ntfs_bitmap_clear_bit(vol->mftbmp_ino, bit)) { ntfs_error(vol->sb, "Failed to clear bit in mft bitmap.%s", es); NVolSetErrors(vol); } + if ((from_reserve || reserve_created) && + vol->mft_record_reserve_pos == bit + 1) + vol->mft_record_reserve_pos = bit; if (!base_ni || base_ni->mft_no != FILE_MFT) up_write(&vol->mftbmp_lock); err_out: @@ -2514,9 +2746,11 @@ int ntfs_mft_record_free(struct ntfs_volume *vol, struct ntfs_inode *ni) int err; u16 seq_no; __le16 old_seq_no; + __le64 old_base_mft_record; struct mft_record *ni_mrec; unsigned int memalloc_flags; struct ntfs_inode *base_ni; + bool keep_reserved; if (!vol || !ni) return -EINVAL; @@ -2529,9 +2763,23 @@ int ntfs_mft_record_free(struct ntfs_volume *vol, struct ntfs_inode *ni) /* Cache the mft reference for later. */ mft_no = ni->mft_no; - - /* Mark the mft record as not in use. */ - ni_mrec->flags &= ~MFT_RECORD_IN_USE; + if (likely(ni->nr_extents >= 0)) + base_ni = ni; + else + base_ni = ni->ext.base_ntfs_ino; + keep_reserved = mft_no >= FILE_reserved12 && + mft_no <= FILE_reserved15 && + base_ni->mft_no == FILE_MFT; + + old_base_mft_record = ni_mrec->base_mft_record; + if (keep_reserved) { + /* Restore the special, unnamed form used by reserved records. */ + ni_mrec->base_mft_record = 0; + ni_mrec->flags |= MFT_RECORD_IN_USE; + } else { + /* Mark the mft record as not in use. */ + ni_mrec->flags &= ~MFT_RECORD_IN_USE; + } /* Increment the sequence number, skipping zero, if it is not zero. */ old_seq_no = ni_mrec->sequence_number; @@ -2560,24 +2808,28 @@ int ntfs_mft_record_free(struct ntfs_volume *vol, struct ntfs_inode *ni) if (err) goto sync_rollback; - if (likely(ni->nr_extents >= 0)) - base_ni = ni; - else - base_ni = ni->ext.base_ntfs_ino; + if (keep_reserved) { + unmap_mft_record(ni); + return 0; + } /* Clear the bit in the $MFT/$BITMAP corresponding to this record. */ memalloc_flags = memalloc_nofs_save(); if (base_ni->mft_no != FILE_MFT) down_write(&vol->mftbmp_lock); err = ntfs_bitmap_clear_bit(vol->mftbmp_ino, mft_no); + if (!err) + ntfs_inc_free_mft_records(vol, 1); + if (!err && base_ni->mft_no == FILE_MFT && + mft_no + 1 == vol->mft_record_reserve_pos && + mft_no < vol->mft_record_reserve_end) + vol->mft_record_reserve_pos = mft_no; if (base_ni->mft_no != FILE_MFT) up_write(&vol->mftbmp_lock); memalloc_nofs_restore(memalloc_flags); if (err) goto bitmap_rollback; - unmap_mft_record(ni); - ntfs_inc_free_mft_records(vol, 1); return 0; /* Rollback what we did... */ @@ -2595,6 +2847,7 @@ sync_rollback: "Eeek! Rollback failed in %s. Leaving inconsistent metadata!\n", __func__); ni_mrec->flags |= MFT_RECORD_IN_USE; ni_mrec->sequence_number = old_seq_no; + ni_mrec->base_mft_record = old_base_mft_record; NInoSetDirty(ni); write_mft_record(ni, ni_mrec, 0); unmap_mft_record(ni); diff --git a/fs/ntfs/mft.h b/fs/ntfs/mft.h index 75a51a98d0f6..ed5c1d595c0d 100644 --- a/fs/ntfs/mft.h +++ b/fs/ntfs/mft.h @@ -78,7 +78,7 @@ static inline int write_mft_record(struct ntfs_inode *ni, struct mft_record *m, int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode, struct ntfs_inode **ni, struct ntfs_inode *base_ni, - struct mft_record **ni_mrec); + struct mft_record **ni_mrec, const s64 mft_data_vcn); int ntfs_mft_record_free(struct ntfs_volume *vol, struct ntfs_inode *ni); int ntfs_mft_records_write(const struct ntfs_volume *vol, const u64 mref, const s64 count, struct mft_record *b); diff --git a/fs/ntfs/namei.c b/fs/ntfs/namei.c index 7091b2496fac..fdf52fac4329 100644 --- a/fs/ntfs/namei.c +++ b/fs/ntfs/namei.c @@ -480,7 +480,7 @@ static struct ntfs_inode *__ntfs_create(struct mnt_idmap *idmap, struct inode *d mark_inode_dirty(dir); err = ntfs_mft_record_alloc(dir_ni->vol, mode, &ni, NULL, - &ni_mrec); + &ni_mrec, -1); if (err) { iput(vi); return ERR_PTR(err); diff --git a/fs/ntfs/super.c b/fs/ntfs/super.c index 5aad2d2a36bb..4066bacabe37 100644 --- a/fs/ntfs/super.c +++ b/fs/ntfs/super.c @@ -2064,8 +2064,7 @@ static unsigned long __get_nr_free_mft_records(struct ntfs_volume *vol, /* If errors occurred we may well have gone below zero, fix this. */ if (nr_free < 0) nr_free = 0; - else - atomic64_set(&vol->free_mft_records, nr_free); + atomic64_set(&vol->free_mft_records, nr_free); ntfs_debug("Exiting."); return nr_free; @@ -2131,7 +2130,14 @@ static int ntfs_statfs(struct dentry *dentry, struct kstatfs *sfs) read_unlock_irqrestore(&mft_ni->size_lock, flags); /* Free inodes in fs (based on current total count). */ - sfs->f_ffree = atomic64_read(&vol->free_mft_records); + size = atomic64_read(&vol->free_mft_records); + if (unlikely(size < 0 || size > (s64)sfs->f_files)) + ntfs_warning(vol->sb, "Invalid free MFT record count %lld.", size); + if (size < 0) + size = 0; + else if (size > (s64)sfs->f_files) + size = sfs->f_files; + sfs->f_ffree = size; /* * File system id. This is extremely *nix flavour dependent and even diff --git a/fs/ntfs/volume.h b/fs/ntfs/volume.h index 65fd3908af26..bc85a9592245 100644 --- a/fs/ntfs/volume.h +++ b/fs/ntfs/volume.h @@ -55,6 +55,10 @@ * @attrdef_size: Size of the attribute definition table in bytes. * @attrdef: Table of attribute definitions. Obtained from FILE_AttrDef. * @mft_data_pos: Mft record number at which to allocate the next mft record. + * @mft_record_reserve_pos: First record in the in-memory MFT metadata reserve + * (protected by mftbmp_lock). + * @mft_record_reserve_end: First record beyond the MFT metadata reserve + * (protected by mftbmp_lock). * @mft_zone_start: First cluster of the mft zone. * @mft_zone_end: First cluster beyond the mft zone. * @mft_zone_pos: Current position in the mft zone. @@ -119,6 +123,8 @@ struct ntfs_volume { s32 attrdef_size; struct attr_def *attrdef; s64 mft_data_pos; + s64 mft_record_reserve_pos; + s64 mft_record_reserve_end; s64 mft_zone_start; s64 mft_zone_end; s64 mft_zone_pos; @@ -252,17 +258,11 @@ static inline void ntfs_dec_free_clusters(struct ntfs_volume *vol, s64 nr) static inline void ntfs_inc_free_mft_records(struct ntfs_volume *vol, s64 nr) { - if (!NVolFreeClusterKnown(vol)) - return; - atomic64_add(nr, &vol->free_mft_records); } static inline void ntfs_dec_free_mft_records(struct ntfs_volume *vol, s64 nr) { - if (!NVolFreeClusterKnown(vol)) - return; - atomic64_sub(nr, &vol->free_mft_records); } diff --git a/fs/smb/client/cifssmb.c b/fs/smb/client/cifssmb.c index f9aff0712794..6dddbd84b93b 100644 --- a/fs/smb/client/cifssmb.c +++ b/fs/smb/client/cifssmb.c @@ -3080,7 +3080,7 @@ int cifs_query_reparse_point(const unsigned int xid, end = 2 + get_bcc(&io_rsp->hdr) + (__u8 *)&io_rsp->ByteCount; start = (__u8 *)&io_rsp->hdr.Protocol + data_offset; - if (start >= end) { + if (start >= end || (size_t)(end - start) < sizeof(*buf)) { rc = smb_EIO2(smb_eio_trace_qreparse_data_area, (unsigned long)start - (unsigned long)io_rsp, (unsigned long)end - (unsigned long)io_rsp); diff --git a/fs/smb/client/connect.c b/fs/smb/client/connect.c index b6e98eb31673..28e1ddeb6182 100644 --- a/fs/smb/client/connect.c +++ b/fs/smb/client/connect.c @@ -174,6 +174,8 @@ cifs_signal_cifsd_for_reconnect(struct TCP_Server_Info *server, nserver = ses->chans[i].server; if (!nserver) continue; + if (!list_empty(&nserver->rlist)) + continue; nserver->srv_count++; list_add(&nserver->rlist, &reco); } @@ -182,11 +184,15 @@ cifs_signal_cifsd_for_reconnect(struct TCP_Server_Info *server, } } + spin_lock(&cifs_tcp_ses_lock); list_for_each_entry_safe(server, nserver, &reco, rlist) { list_del_init(&server->rlist); set_need_reco(server); + spin_unlock(&cifs_tcp_ses_lock); cifs_put_tcp_session(server, 0); + spin_lock(&cifs_tcp_ses_lock); } + spin_unlock(&cifs_tcp_ses_lock); } /* @@ -1067,6 +1073,7 @@ clean_demultiplex_info(struct TCP_Server_Info *server) spin_unlock(&server->srv_lock); cancel_delayed_work_sync(&server->echo); + cancel_delayed_work_sync(&server->reconnect); spin_lock(&server->srv_lock); server->tcpStatus = CifsExiting; @@ -1823,6 +1830,7 @@ cifs_get_tcp_session(struct smb3_fs_context *ctx, spin_lock_init(&tcp_ses->mid_counter_lock); INIT_LIST_HEAD(&tcp_ses->tcp_ses_list); INIT_LIST_HEAD(&tcp_ses->smb_ses_list); + INIT_LIST_HEAD(&tcp_ses->rlist); INIT_DELAYED_WORK(&tcp_ses->echo, cifs_echo_request); INIT_DELAYED_WORK(&tcp_ses->reconnect, smb2_reconnect_server); mutex_init(&tcp_ses->reconnect_mutex); @@ -1926,6 +1934,7 @@ out_err: kfree(tcp_ses->leaf_fullpath); if (tcp_ses->ssocket) sock_release(tcp_ses->ssocket); + smbd_destroy(tcp_ses); kfree(tcp_ses); } return ERR_PTR(rc); diff --git a/fs/smb/client/file.c b/fs/smb/client/file.c index 1aa4844f8b8a..0d428517f454 100644 --- a/fs/smb/client/file.c +++ b/fs/smb/client/file.c @@ -3354,8 +3354,8 @@ void cifs_oplock_break(struct work_struct *work) wait_on_bit(&cinode->flags, CIFS_INODE_PENDING_WRITERS, TASK_UNINTERRUPTIBLE); - tlink = cifs_sb_tlink(cifs_sb); - if (IS_ERR(tlink)) { + tlink = cifs_get_tlink(cfile->tlink); + if (IS_ERR_OR_NULL(tlink)) { /* drop the reference taken when the break was queued */ _cifsFileInfo_put(cfile, false /* do not wait for ourself */, false); goto out; diff --git a/fs/smb/client/misc.c b/fs/smb/client/misc.c index 945194fe7a97..05168284f205 100644 --- a/fs/smb/client/misc.c +++ b/fs/smb/client/misc.c @@ -788,7 +788,11 @@ parse_dfs_referrals(struct get_dfs_referral_rsp *rsp, u32 rsp_size, node->ref_flag = le16_to_cpu(ref->ReferralEntryFlags); /* copy DfsPath */ - if (le16_to_cpu(ref->DfsPathOffset) > data_end - (char *)ref) { + if (le16_to_cpu(ref->DfsPathOffset) < sizeof(*ref) || + le16_to_cpu(ref->DfsPathOffset) > data_end - (char *)ref) { + cifs_dbg(VFS, "%s: DfsPathOffset %u out of range [%zu, %td]\n", + __func__, le16_to_cpu(ref->DfsPathOffset), + sizeof(*ref), data_end - (char *)ref); rc = -EINVAL; goto parse_DFS_referrals_exit; } @@ -802,7 +806,11 @@ parse_dfs_referrals(struct get_dfs_referral_rsp *rsp, u32 rsp_size, } /* copy link target UNC */ - if (le16_to_cpu(ref->NetworkAddressOffset) > data_end - (char *)ref) { + if (le16_to_cpu(ref->NetworkAddressOffset) < sizeof(*ref) || + le16_to_cpu(ref->NetworkAddressOffset) > data_end - (char *)ref) { + cifs_dbg(VFS, "%s: NetworkAddressOffset %u out of range [%zu, %td]\n", + __func__, le16_to_cpu(ref->NetworkAddressOffset), + sizeof(*ref), data_end - (char *)ref); rc = -EINVAL; goto parse_DFS_referrals_exit; } diff --git a/fs/smb/client/reparse.c b/fs/smb/client/reparse.c index 8a1b9e8be5ba..3a27773186ae 100644 --- a/fs/smb/client/reparse.c +++ b/fs/smb/client/reparse.c @@ -3,6 +3,7 @@ * Copyright (c) 2024 Paulo Alcantara <pc@manguebit.com> */ +#include <linux/ctype.h> #include <linux/fs.h> #include <linux/stat.h> #include <linux/slab.h> @@ -159,15 +160,24 @@ static int create_native_symlink(const unsigned int xid, struct inode *inode, convert_delimiter(sym, sep); /* - * For absolute NT symlinks it is required to pass also leading - * backslash and to not mangle NT object prefix "\\??\\" and not to - * mangle colon in drive letter. But cifs_convert_path_to_utf16() - * removes leading backslash and replaces '?' and ':'. So temporary - * mask these characters in NT object prefix by '_' and then change - * them back. + * Absolute NT symlinks must retain the leading backslash, "\\??\\" + * prefix and drive-letter colon. cifs_convert_path_to_utf16() strips + * the leading backslash and maps '?' and ':', so temporarily mask + * these characters with '_' and restore them after conversion. + * + * When symlinkroot is unset, sym comes directly from the caller. + * Validate the complete "\\??\\X:" prefix before using fixed offsets + * or subtracting the NT prefix length below. Require an ASCII drive + * letter so the prefix occupies six characters in UTF-16 too. */ - if (!(sbflags & CIFS_MOUNT_POSIX_PATHS) && symname[0] == '/') + if (!(sbflags & CIFS_MOUNT_POSIX_PATHS) && symname[0] == '/') { + if (!strstarts(sym, "\\??\\") || !isascii(sym[4]) || + !isalpha(sym[4]) || sym[5] != ':') { + rc = -EINVAL; + goto out; + } sym[0] = sym[1] = sym[2] = sym[5] = '_'; + } /* * On a POSIX paths mount the symlink target is stored verbatim, so @@ -1139,29 +1149,30 @@ static bool wsl_to_fattr(struct cifs_open_info_data *data, u32 tag, struct cifs_fattr *fattr) { unsigned int sbflags = cifs_sb_flags(cifs_sb); + kuid_t uid = cifs_sb->ctx->linux_uid; + kgid_t gid = cifs_sb->ctx->linux_gid; struct smb2_file_full_ea_info *ea; bool have_xattr_dev = false; + dev_t rdev = 0; + umode_t mode; u32 next = 0; - fattr->cf_uid = cifs_sb->ctx->linux_uid; - fattr->cf_gid = cifs_sb->ctx->linux_gid; - - fattr->cf_mode &= ~S_IFMT; + mode = fattr->cf_mode & ~S_IFMT; switch (tag) { case IO_REPARSE_TAG_LX_SYMLINK: - fattr->cf_mode |= S_IFLNK; + mode |= S_IFLNK; break; case IO_REPARSE_TAG_LX_FIFO: - fattr->cf_mode |= S_IFIFO; + mode |= S_IFIFO; break; case IO_REPARSE_TAG_AF_UNIX: - fattr->cf_mode |= S_IFSOCK; + mode |= S_IFSOCK; break; case IO_REPARSE_TAG_LX_CHR: - fattr->cf_mode |= S_IFCHR; + mode |= S_IFCHR; break; case IO_REPARSE_TAG_LX_BLK: - fattr->cf_mode |= S_IFBLK; + mode |= S_IFBLK; break; } @@ -1185,26 +1196,29 @@ static bool wsl_to_fattr(struct cifs_open_info_data *data, if (!strncmp(name, SMB2_WSL_XATTR_UID, nlen)) { if (!(sbflags & CIFS_MOUNT_OVERR_UID)) - fattr->cf_uid = wsl_make_kuid(cifs_sb, v); + uid = wsl_make_kuid(cifs_sb, v); } else if (!strncmp(name, SMB2_WSL_XATTR_GID, nlen)) { if (!(sbflags & CIFS_MOUNT_OVERR_GID)) - fattr->cf_gid = wsl_make_kgid(cifs_sb, v); + gid = wsl_make_kgid(cifs_sb, v); } else if (!strncmp(name, SMB2_WSL_XATTR_MODE, nlen)) { /* File type in reparse point tag and in xattr mode must match. */ - if (S_DT(fattr->cf_mode) != S_DT(le32_to_cpu(*(__le32 *)v))) + if (S_DT(mode) != S_DT(get_unaligned_le32(v))) return false; - fattr->cf_mode = (umode_t)le32_to_cpu(*(__le32 *)v); + mode = get_unaligned_le32(v); } else if (!strncmp(name, SMB2_WSL_XATTR_DEV, nlen)) { - fattr->cf_rdev = reparse_mkdev(v); + rdev = reparse_mkdev(v); have_xattr_dev = true; } } while (next); out: - /* Major and minor numbers for char and block devices are mandatory. */ if (!have_xattr_dev && (tag == IO_REPARSE_TAG_LX_CHR || tag == IO_REPARSE_TAG_LX_BLK)) return false; + fattr->cf_uid = uid; + fattr->cf_gid = gid; + fattr->cf_mode = mode; + fattr->cf_rdev = rdev; return true; } diff --git a/fs/smb/client/reparse.h b/fs/smb/client/reparse.h index 49efd85b1e94..05b2cecb4495 100644 --- a/fs/smb/client/reparse.h +++ b/fs/smb/client/reparse.h @@ -9,6 +9,7 @@ #include <linux/fs.h> #include <linux/stat.h> #include <linux/uidgid.h> +#include <linux/unaligned.h> #include "fs_context.h" #include "cifsglob.h" #include "../common/smbfsctl.h" @@ -23,7 +24,7 @@ static inline dev_t reparse_mkdev(void *ptr) { - u64 v = le64_to_cpu(*(__le64 *)ptr); + u64 v = get_unaligned_le64(ptr); return MKDEV(v & 0xffffffff, v >> 32); } @@ -31,7 +32,7 @@ static inline dev_t reparse_mkdev(void *ptr) static inline kuid_t wsl_make_kuid(struct cifs_sb_info *cifs_sb, void *ptr) { - u32 uid = le32_to_cpu(*(__le32 *)ptr); + u32 uid = get_unaligned_le32(ptr); if (cifs_sb_flags(cifs_sb) & CIFS_MOUNT_OVERR_UID) return cifs_sb->ctx->linux_uid; @@ -41,7 +42,7 @@ static inline kuid_t wsl_make_kuid(struct cifs_sb_info *cifs_sb, static inline kgid_t wsl_make_kgid(struct cifs_sb_info *cifs_sb, void *ptr) { - u32 gid = le32_to_cpu(*(__le32 *)ptr); + u32 gid = get_unaligned_le32(ptr); if (cifs_sb_flags(cifs_sb) & CIFS_MOUNT_OVERR_GID) return cifs_sb->ctx->linux_gid; diff --git a/fs/smb/client/sess.c b/fs/smb/client/sess.c index 7cf7dd104f7c..e095f41b5882 100644 --- a/fs/smb/client/sess.c +++ b/fs/smb/client/sess.c @@ -149,9 +149,9 @@ int cifs_try_adding_channels(struct cifs_ses *ses) int old_chan_count, new_chan_count; int left; int rc = 0; - int tries = 0; + int tries = 0, attempts; size_t iface_weight = 0, iface_min_speed = 0; - struct cifs_server_iface *iface = NULL, *niface = NULL; + struct cifs_server_iface *iface = NULL, *candidate = NULL; struct cifs_server_iface *last_iface = NULL; spin_lock(&ses->chan_lock); @@ -197,67 +197,89 @@ int cifs_try_adding_channels(struct cifs_ses *ses) break; } - if (!iface) - iface = list_first_entry(&ses->iface_list, struct cifs_server_iface, - iface_head); last_iface = list_last_entry(&ses->iface_list, struct cifs_server_iface, iface_head); iface_min_speed = last_iface->speed; + spin_unlock(&ses->iface_lock); - list_for_each_entry_safe_from(iface, niface, &ses->iface_list, - iface_head) { - /* do not mix rdma and non-rdma interfaces */ - if (iface->rdma_capable != ses->server->rdma) - continue; - - /* skip ifaces that are unusable */ - if (!iface->is_active || - (is_ses_using_iface(ses, iface) && - !iface->rss_capable)) - continue; + attempts = 0; + while (left > 0) { + spin_lock(&ses->iface_lock); - /* check if we already allocated enough channels */ - iface_weight = iface->speed / iface_min_speed; + /* + * iface_lock must be dropped while opening a channel, + * and a concurrent interface refresh may remove and + * free entries during that window, so no list entry + * may be kept across it without a reference. Scan + * the list from the beginning each time and only pass + * a referenced candidate to cifs_ses_add_channel(); + * weight_fulfilled tracks the progress so that no + * iface is selected beyond its weight. + */ + candidate = NULL; + list_for_each_entry(iface, &ses->iface_list, iface_head) { + /* do not mix rdma and non-rdma interfaces */ + if (iface->rdma_capable != ses->server->rdma) + continue; + + /* skip ifaces that are unusable */ + if (!iface->is_active || + (is_ses_using_iface(ses, iface) && + !iface->rss_capable)) + continue; + + /* check if we already allocated enough channels */ + iface_weight = iface->speed / iface_min_speed; + + if (iface->weight_fulfilled >= iface_weight) + continue; + + /* take ref before unlock */ + kref_get(&iface->refcount); + candidate = iface; + break; + } - if (iface->weight_fulfilled >= iface_weight) - continue; + if (!candidate) { + /* no usable iface. reset weight_fulfilled and start over */ + list_for_each_entry(iface, &ses->iface_list, iface_head) + iface->weight_fulfilled = 0; + spin_unlock(&ses->iface_lock); + break; + } - /* take ref before unlock */ - kref_get(&iface->refcount); + attempts++; + if (attempts > 3 * ses->chan_max) { + kref_put(&candidate->refcount, release_iface); + spin_unlock(&ses->iface_lock); + break; + } spin_unlock(&ses->iface_lock); - rc = cifs_ses_add_channel(ses, iface); + rc = cifs_ses_add_channel(ses, candidate); spin_lock(&ses->iface_lock); if (rc) { cifs_dbg(VFS, "failed to open extra channel on iface:%pIS rc=%d\n", - &iface->sockaddr, + &candidate->sockaddr, rc); /* failure to add chan should increase weight */ - iface->weight_fulfilled++; - kref_put(&iface->refcount, release_iface); + candidate->weight_fulfilled++; + kref_put(&candidate->refcount, release_iface); + spin_unlock(&ses->iface_lock); continue; } - iface->num_channels++; - iface->weight_fulfilled++; + candidate->num_channels++; + candidate->weight_fulfilled++; cifs_info("successfully opened new channel on iface:%pIS\n", - &iface->sockaddr); - break; - } - - /* reached end of list. reset weight_fulfilled and start over */ - if (list_entry_is_head(iface, &ses->iface_list, iface_head)) { - list_for_each_entry(iface, &ses->iface_list, iface_head) - iface->weight_fulfilled = 0; + &candidate->sockaddr); spin_unlock(&ses->iface_lock); - iface = NULL; - continue; - } - spin_unlock(&ses->iface_lock); - left--; - new_chan_count++; + left--; + new_chan_count++; + break; + } } return new_chan_count - old_chan_count; diff --git a/fs/smb/client/smb2inode.c b/fs/smb/client/smb2inode.c index 96063e355186..13fe8e3b48f3 100644 --- a/fs/smb/client/smb2inode.c +++ b/fs/smb/client/smb2inode.c @@ -77,6 +77,17 @@ static int parse_posix_sids(struct cifs_open_info_data *data, sidsbuf = (u8 *)qi + le16_to_cpu(qi->OutputBufferOffset) + qi_len; sidsbuf_end = sidsbuf + out_len - qi_len; + if (sidsbuf_end < sidsbuf) { + cifs_dbg(VFS, "%s: server-supplied out_len %u caused pointer wraparound\n", + __func__, out_len); + return -EINVAL; + } + if (sidsbuf_end > (u8 *)rsp_iov->iov_base + rsp_iov->iov_len) { + cifs_dbg(VFS, "%s: server-supplied out_len %u overruns iov by %td bytes\n", + __func__, out_len, + sidsbuf_end - ((u8 *)rsp_iov->iov_base + rsp_iov->iov_len)); + return -EINVAL; + } owner_len = posix_info_sid_size(sidsbuf, sidsbuf_end); if (owner_len == -1) diff --git a/fs/smb/client/smb2misc.c b/fs/smb/client/smb2misc.c index 9068175e57cd..0cfe60ae42c3 100644 --- a/fs/smb/client/smb2misc.c +++ b/fs/smb/client/smb2misc.c @@ -85,6 +85,36 @@ static const __le16 smb2_rsp_struct_sizes[NUMBER_OF_SMB2_COMMANDS] = { /* SMB2_OPLOCK_BREAK */ cpu_to_le16(24) }; +/* + * Minimum received PDU size for commands whose response carries a + * variable-length data area. A non-zero entry marks the command as + * having one, and gives the length smb2_check_message() requires + * before smb2_get_data_area_len() reads the offset and length fields + * out of the fixed response struct. + */ +static const size_t smb2_min_pdu_len[NUMBER_OF_SMB2_COMMANDS] = { + /* SMB2_NEGOTIATE */ sizeof(struct smb2_negotiate_rsp), + /* SMB2_SESSION_SETUP */ sizeof(struct smb2_sess_setup_rsp), + /* SMB2_LOGOFF */ 0, + /* SMB2_TREE_CONNECT */ 0, + /* SMB2_TREE_DISCONNECT */ 0, + /* SMB2_CREATE */ sizeof(struct smb2_create_rsp), + /* SMB2_CLOSE */ 0, + /* SMB2_FLUSH */ 0, + /* SMB2_READ */ sizeof(struct smb2_read_rsp), + /* SMB2_WRITE */ 0, + /* SMB2_LOCK */ 0, + /* SMB2_IOCTL */ sizeof(struct smb2_ioctl_rsp), + /* SMB2_CANCEL */ 0, + /* SMB2_ECHO */ 0, + /* SMB2_QUERY_DIRECTORY */ sizeof(struct smb2_query_directory_rsp), + /* SMB2_CHANGE_NOTIFY */ sizeof(struct smb2_change_notify_rsp), + /* SMB2_QUERY_INFO */ sizeof(struct smb2_query_info_rsp), + /* SMB2_SET_INFO */ 0, + /* SMB2_OPLOCK_BREAK */ 0, +}; + +#define smb2_has_data_area(cmd) (smb2_min_pdu_len[cmd] != 0) #define SMB311_NEGPROT_BASE_SIZE (sizeof(struct smb2_hdr) + sizeof(struct smb2_negotiate_rsp)) static __u32 get_neg_ctxt_len(struct smb2_hdr *hdr, __u32 len, @@ -233,6 +263,16 @@ smb2_check_message(char *buf, unsigned int pdu_len, unsigned int len, } } + if ((shdr->Status == STATUS_SUCCESS || + shdr->Status == STATUS_MORE_PROCESSING_REQUIRED || + pdu->StructureSize2 != SMB2_ERROR_STRUCTURE_SIZE2_LE) && + smb2_has_data_area(command) && + len < smb2_min_pdu_len[command]) { + cifs_server_dbg(VFS, "SMB2 command %d response too short: %u < %zu\n", + command, len, smb2_min_pdu_len[command]); + return 1; + } + have_data = false; data_area_overlap = false; calc_len = __smb2_calc_size(buf, &have_data, &data_area_overlap); @@ -299,33 +339,6 @@ smb2_check_message(char *buf, unsigned int pdu_len, unsigned int len, } /* - * The size of the variable area depends on the offset and length fields - * located in different fields for various SMB2 responses. SMB2 responses - * with no variable length info, show an offset of zero for the offset field. - */ -static const bool has_smb2_data_area[NUMBER_OF_SMB2_COMMANDS] = { - /* SMB2_NEGOTIATE */ true, - /* SMB2_SESSION_SETUP */ true, - /* SMB2_LOGOFF */ false, - /* SMB2_TREE_CONNECT */ false, - /* SMB2_TREE_DISCONNECT */ false, - /* SMB2_CREATE */ true, - /* SMB2_CLOSE */ false, - /* SMB2_FLUSH */ false, - /* SMB2_READ */ true, - /* SMB2_WRITE */ false, - /* SMB2_LOCK */ false, - /* SMB2_IOCTL */ true, - /* SMB2_CANCEL */ false, /* BB CHECK this not listed in documentation */ - /* SMB2_ECHO */ false, - /* SMB2_QUERY_DIRECTORY */ true, - /* SMB2_CHANGE_NOTIFY */ true, - /* SMB2_QUERY_INFO */ true, - /* SMB2_SET_INFO */ false, - /* SMB2_OPLOCK_BREAK */ false -}; - -/* * Returns the pointer to the beginning of the data area. Length of the data * area and the offset to it (from the beginning of the smb are also returned. */ @@ -451,7 +464,7 @@ __smb2_calc_size(void *buf, bool *have_data, bool *data_area_overlap) */ len += le16_to_cpu(pdu->StructureSize2); - if (has_smb2_data_area[le16_to_cpu(shdr->Command)] == false) + if (!smb2_has_data_area(le16_to_cpu(shdr->Command))) goto calc_size_exit; smb2_get_data_area_len(&offset, &data_length, shdr); diff --git a/fs/smb/client/smb2ops.c b/fs/smb/client/smb2ops.c index cb4fd09f996e..3464470d3297 100644 --- a/fs/smb/client/smb2ops.c +++ b/fs/smb/client/smb2ops.c @@ -785,9 +785,9 @@ next_iface: break; } /* Validate that Next doesn't point beyond the buffer */ - if (next > bytes_left) { - cifs_dbg(VFS, "%s: invalid Next pointer %zu > %zd\n", - __func__, next, bytes_left); + if (next < sizeof(*p) || next > bytes_left) { + cifs_dbg(VFS, "%s: invalid Next pointer %zu out of range [%zu, %zd]\n", + __func__, next, sizeof(*p), bytes_left); rc = -EINVAL; goto out; } @@ -1053,8 +1053,9 @@ move_smb2_ea_to_cifs(char *dst, size_t dst_size, char *name, *value; size_t buf_size = dst_size; size_t name_len, value_len, user_name_len; + u32 next_off; - while (src_size > 0) { + while (src_size >= sizeof(*src)) { name_len = (size_t)src->ea_name_length; value_len = (size_t)le16_to_cpu(src->ea_value_length); @@ -1110,14 +1111,22 @@ move_smb2_ea_to_cifs(char *dst, size_t dst_size, if (!src->next_entry_offset) break; - if (src_size < le32_to_cpu(src->next_entry_offset)) { - /* stop before overrun buffer */ - rc = -ERANGE; - break; + next_off = le32_to_cpu(src->next_entry_offset); + if (next_off < sizeof(*src) || src_size < next_off) { + cifs_dbg(FYI, "EA next_entry_offset %u out of range [%zu, %zu]\n", + next_off, sizeof(*src), src_size); + rc = smb_EIO2(smb_eio_trace_ea_next_offset, + next_off, src_size); + goto out; + } + src_size -= next_off; + src = (void *)((char *)src + next_off); + if (src_size > 0 && src_size < sizeof(*src)) { + cifs_dbg(FYI, "EA next_entry_offset %u left truncated entry (%zu bytes)\n", + next_off, src_size); + rc = smb_EIO2(smb_eio_trace_ea_next_offset, next_off, src_size); + goto out; } - src_size -= le32_to_cpu(src->next_entry_offset); - src = (void *)((char *)src + - le32_to_cpu(src->next_entry_offset)); } /* didn't find the named attribute */ @@ -2454,8 +2463,14 @@ smb3_enum_snapshots(const unsigned int xid, struct cifs_tcon *tcon, * and retry the ioctl again with larger array size sufficient * to hold all of the snapshot GMT tokens on the second try. */ - if (snapshot_in.snapshot_array_size < GMT_TOKEN_SIZE) + if (snapshot_in.snapshot_array_size < GMT_TOKEN_SIZE) { + if (ret_data_len < sizeof(struct smb_snapshot_array)) { + rc = -EIO; + kfree(retbuf); + return rc; + } ret_data_len = sizeof(struct smb_snapshot_array); + } /* * We return struct SRV_SNAPSHOT_ARRAY, followed by @@ -5365,11 +5380,13 @@ receive_encrypted_standard(struct TCP_Server_Info *server, length = decrypt_raw_data(server, buf, buf_size, NULL, false); if (length) return length; + pdu_length = buf_size; next_is_large = server->large_buf; one_more: shdr = (struct smb2_hdr *)buf; next_cmd = le32_to_cpu(shdr->NextCommand); + server->total_read = next_cmd ? next_cmd : pdu_length; if (*num_mids >= MAX_COMPOUND) { cifs_server_dbg(VFS, "too many PDUs in compound\n"); @@ -5377,8 +5394,15 @@ one_more: } if (next_cmd) { - if (WARN_ON_ONCE(next_cmd > pdu_length)) + if (next_cmd < MID_HEADER_SIZE(server) || + next_cmd > pdu_length || + pdu_length - next_cmd < MID_HEADER_SIZE(server)) { + unsigned int max_next = pdu_length > (unsigned int)MID_HEADER_SIZE(server) ? + pdu_length - (unsigned int)MID_HEADER_SIZE(server) : 0; + cifs_server_dbg(VFS, "invalid NextCommand offset %u out of range [%zu, %u]\n", + next_cmd, MID_HEADER_SIZE(server), max_next); return -1; + } if (next_is_large) next_buffer = (char *)cifs_buf_get(); else @@ -5414,6 +5438,7 @@ one_more: server->bigbuf = buf = next_buffer; else server->smallbuf = buf = next_buffer; + next_buffer = NULL; goto one_more; } else if (ret != 0) { /* diff --git a/fs/smb/client/smb2pdu.c b/fs/smb/client/smb2pdu.c index dea05aeb53a1..880ce12f50c4 100644 --- a/fs/smb/client/smb2pdu.c +++ b/fs/smb/client/smb2pdu.c @@ -189,18 +189,19 @@ cifs_chan_skip_or_disable(struct cifs_ses *ses, spin_unlock(&ses->chan_lock); /* - * the above reference of server by channel - * needs to be dropped without holding chan_lock - * as cifs_put_tcp_session takes a higher lock - * i.e. cifs_tcp_ses_lock + * signal the channel and its primary server to + * reconnect before dropping the above reference of + * server by channel, which is done without holding + * chan_lock as cifs_put_tcp_session takes a higher + * lock i.e. cifs_tcp_ses_lock */ - cifs_put_tcp_session(server, from_reconnect); - cifs_signal_cifsd_for_reconnect(server, false); /* mark primary server as needing reconnect */ pserver = server->primary_server; cifs_signal_cifsd_for_reconnect(pserver, false); + + cifs_put_tcp_session(server, from_reconnect); skip_terminate: return -EHOSTDOWN; } diff --git a/fs/smb/client/trace.h b/fs/smb/client/trace.h index b442cccd1530..bb8d0197cb54 100644 --- a/fs/smb/client/trace.h +++ b/fs/smb/client/trace.h @@ -27,6 +27,7 @@ EM(smb_eio_trace_copychunk_overcopy_c, "copychunk_overcopy_c") \ EM(smb_eio_trace_create_rsp_too_small, "create_rsp_too_small") \ EM(smb_eio_trace_dfsref_no_rsp, "dfsref_no_rsp") \ + EM(smb_eio_trace_ea_next_offset, "ea_next_offset") \ EM(smb_eio_trace_ea_overrun, "ea_overrun") \ EM(smb_eio_trace_extract_will_pin, "extract_will_pin") \ EM(smb_eio_trace_forced_shutdown, "forced_shutdown") \ diff --git a/fs/smb/server/connection.c b/fs/smb/server/connection.c index 4cb92d6599ee..d211861ff86f 100644 --- a/fs/smb/server/connection.c +++ b/fs/smb/server/connection.c @@ -22,6 +22,8 @@ static DEFINE_MUTEX(init_lock); static struct ksmbd_conn_ops default_conn_ops; +static struct delayed_work session_expiration_work; +static bool stopping_session_expiration_work; DEFINE_HASHTABLE(conn_list, CONN_HASH_BITS); DECLARE_RWSEM(conn_list_lock); @@ -158,18 +160,28 @@ static void delete_proc_clients(void) {} static struct workqueue_struct *ksmbd_conn_wq; +static void ksmbd_session_expiration_worker(struct work_struct *work); + int ksmbd_conn_wq_init(void) { ksmbd_conn_wq = alloc_workqueue("ksmbd-conn-release", WQ_UNBOUND | WQ_MEM_RECLAIM, 0); if (!ksmbd_conn_wq) return -ENOMEM; + + WRITE_ONCE(stopping_session_expiration_work, false); + INIT_DELAYED_WORK(&session_expiration_work, + ksmbd_session_expiration_worker); + queue_delayed_work(ksmbd_conn_wq, &session_expiration_work, + KSMBD_SESSION_EXPIRATION_INTERVAL); return 0; } void ksmbd_conn_wq_destroy(void) { if (ksmbd_conn_wq) { + WRITE_ONCE(stopping_session_expiration_work, true); + cancel_delayed_work_sync(&session_expiration_work); destroy_workqueue(ksmbd_conn_wq); ksmbd_conn_wq = NULL; } @@ -279,6 +291,7 @@ struct ksmbd_conn *ksmbd_conn_alloc(void) return NULL; conn->need_neg = true; + conn->creation_time = jiffies; ksmbd_conn_set_new(conn); conn->local_nls = load_nls("utf8"); if (!conn->local_nls) @@ -588,6 +601,20 @@ bool ksmbd_conn_alive(struct ksmbd_conn *conn) if (kthread_should_stop()) return false; + /* + * Stale connections that have not completed NEGOTIATE and SESSION_SETUP + * must be disconnected. Do not race a request that is currently + * completing authentication. + */ + if (!atomic_read(&conn->req_running) && + time_after(jiffies, conn->creation_time + + KSMBD_UNAUTHENTICATED_CONN_TIMEOUT) && + (READ_ONCE(conn->need_neg) || + !ksmbd_conn_has_valid_or_expired_session(conn))) { + ksmbd_debug(CONN, "Connection setup timed out\n"); + return false; + } + if (atomic_read(&conn->stats.open_files_count) > 0) return true; @@ -605,6 +632,51 @@ bool ksmbd_conn_alive(struct ksmbd_conn *conn) return true; } +static void ksmbd_session_expiration_worker(struct work_struct *work) +{ + struct ksmbd_conn *conn, *target; + int bkt; + + if (!ksmbd_server_running()) + goto reschedule; + + ksmbd_expire_sessions(); + + /* + * An old connection without a Valid or Expired session must be + * disconnected. Process one connection at a time without holding + * conn_list_lock across transport shutdown. + */ +again: + target = NULL; + down_read(&conn_list_lock); + hash_for_each(conn_list, bkt, conn, hlist) { + if (ksmbd_conn_exiting(conn) || ksmbd_conn_releasing(conn) || + atomic_read(&conn->req_running) || + time_before_eq(jiffies, conn->creation_time + + KSMBD_UNAUTHENTICATED_CONN_TIMEOUT) || + (!READ_ONCE(conn->need_neg) && + ksmbd_conn_has_valid_or_expired_session(conn))) + continue; + + target = ksmbd_conn_get(conn); + break; + } + up_read(&conn_list_lock); + + if (target) { + ksmbd_debug(CONN, "Connection setup timed out\n"); + ksmbd_conn_abort(target); + ksmbd_conn_put(target); + goto again; + } + +reschedule: + if (!READ_ONCE(stopping_session_expiration_work)) + queue_delayed_work(ksmbd_conn_wq, &session_expiration_work, + KSMBD_SESSION_EXPIRATION_INTERVAL); +} + /* "+2" for BCC field (ByteCount, 2 bytes) */ #define SMB1_MIN_SUPPORTED_PDU_SIZE (sizeof(struct smb_hdr) + 2) #define SMB2_MIN_SUPPORTED_PDU_SIZE (sizeof(struct smb2_pdu)) diff --git a/fs/smb/server/connection.h b/fs/smb/server/connection.h index 63484c8efbbd..371f17b4f02a 100644 --- a/fs/smb/server/connection.h +++ b/fs/smb/server/connection.h @@ -77,6 +77,7 @@ struct ksmbd_conn { struct rw_semaphore session_lock; /* smb session 1 per user */ struct xarray sessions; + unsigned long creation_time; unsigned long last_active; /* How many request are running currently */ atomic_t req_running; @@ -192,6 +193,8 @@ struct ksmbd_transport { #define KSMBD_TCP_RECV_TIMEOUT (7 * HZ) #define KSMBD_TCP_SEND_TIMEOUT (5 * HZ) +#define KSMBD_SESSION_EXPIRATION_INTERVAL (5 * HZ) +#define KSMBD_UNAUTHENTICATED_CONN_TIMEOUT (45 * HZ) #define KSMBD_TCP_PEER_SOCKADDR(c) ((struct sockaddr *)&((c)->peer_addr)) #define CONN_HASH_BITS 12 diff --git a/fs/smb/server/mgmt/user_session.c b/fs/smb/server/mgmt/user_session.c index 2eb8f730e99e..44dc3f800cd4 100644 --- a/fs/smb/server/mgmt/user_session.c +++ b/fs/smb/server/mgmt/user_session.c @@ -22,6 +22,7 @@ static DEFINE_IDA(session_ida); #define SESSION_HASH_BITS 12 +#define KSMBD_MAX_PENDING_SESSIONS 1 static DEFINE_HASHTABLE(sessions_table, SESSION_HASH_BITS); static DECLARE_RWSEM(sessions_table_lock); @@ -432,26 +433,31 @@ struct ksmbd_session *__session_lookup(unsigned long long id) return NULL; } -static void ksmbd_expire_session(struct ksmbd_conn *conn) +static bool ksmbd_too_many_session_setups(struct ksmbd_conn *conn) { unsigned long id; struct ksmbd_session *sess; + unsigned int pending = 0; down_write(&sessions_table_lock); down_write(&conn->session_lock); xa_for_each(&conn->sessions, id, sess) { + if (READ_ONCE(sess->state) != SMB2_SESSION_IN_PROGRESS) + continue; + if (atomic_read(&sess->refcnt) <= 1 && - (sess->state != SMB2_SESSION_VALID || - time_after(jiffies, - sess->last_active + SMB2_SESSION_TIMEOUT))) { + time_after(jiffies, sess->last_active + + KSMBD_UNAUTHENTICATED_CONN_TIMEOUT)) { xa_erase(&conn->sessions, sess->id); ksmbd_session_remove_from_table(sess); ksmbd_session_destroy(sess); continue; } + pending++; } up_write(&conn->session_lock); up_write(&sessions_table_lock); + return pending >= KSMBD_MAX_PENDING_SESSIONS; } int ksmbd_session_register(struct ksmbd_conn *conn, @@ -461,9 +467,12 @@ int ksmbd_session_register(struct ksmbd_conn *conn, sess->dialect = conn->dialect; memcpy(sess->ClientGUID, conn->ClientGUID, SMB2_CLIENT_GUID_SIZE); - ksmbd_expire_session(conn); - ret = xa_err(xa_store(&conn->sessions, sess->id, sess, - KSMBD_DEFAULT_GFP)); + /* Bound abandoned SessionId-zero authentication exchanges. */ + if (ksmbd_too_many_session_setups(conn)) + ret = -ENOSPC; + else + ret = xa_err(xa_store(&conn->sessions, sess->id, sess, + KSMBD_DEFAULT_GFP)); if (ret) { down_write(&sessions_table_lock); ksmbd_session_remove_from_table(sess); @@ -474,6 +483,105 @@ int ksmbd_session_register(struct ksmbd_conn *conn, return ret; } +void ksmbd_session_unregister(struct ksmbd_conn *conn, + struct ksmbd_session *sess) +{ + struct ksmbd_conn *session_conns[KSMBD_MAX_CHANNELS]; + struct channel *chann; + unsigned long index; + unsigned int nr_conns = 0, i; + bool removed = false; + + down_write(&sessions_table_lock); + if (!hlist_unhashed(&sess->hlist)) { + /* Keep each channel connection stable under sessions_table_lock. */ + down_read(&sess->chann_lock); + xa_for_each(&sess->ksmbd_chann_list, index, chann) { + if (nr_conns == ARRAY_SIZE(session_conns)) + break; + session_conns[nr_conns++] = chann->conn; + } + up_read(&sess->chann_lock); + + ksmbd_session_remove_from_table(sess); + removed = true; + } + + down_write(&conn->session_lock); + if (xa_load(&conn->sessions, sess->id) == sess) + xa_erase(&conn->sessions, sess->id); + up_write(&conn->session_lock); + for (i = 0; i < nr_conns; i++) { + if (session_conns[i] == conn) + continue; + down_write(&session_conns[i]->session_lock); + if (xa_load(&session_conns[i]->sessions, sess->id) == sess) + xa_erase(&session_conns[i]->sessions, sess->id); + up_write(&session_conns[i]->session_lock); + } + up_write(&sessions_table_lock); + + if (removed) + ksmbd_user_session_put(sess); +} + +bool ksmbd_conn_has_valid_or_expired_session(struct ksmbd_conn *conn) +{ + struct ksmbd_session *sess; + unsigned long id; + int state, bkt; + bool found = false; + + down_read(&conn->session_lock); + xa_for_each(&conn->sessions, id, sess) { + state = READ_ONCE(sess->state); + if (state == SMB2_SESSION_VALID || + state == SMB2_SESSION_EXPIRED) { + found = true; + break; + } + } + up_read(&conn->session_lock); + if (found) + return true; + + /* A session bound through SMB3 multichannel is not in conn->sessions. */ + down_read(&sessions_table_lock); + hash_for_each(sessions_table, bkt, sess, hlist) { + state = READ_ONCE(sess->state); + if (state != SMB2_SESSION_VALID && + state != SMB2_SESSION_EXPIRED) + continue; + + down_read(&sess->chann_lock); + found = xa_load(&sess->ksmbd_chann_list, (long)conn); + up_read(&sess->chann_lock); + if (found) + break; + } + up_read(&sessions_table_lock); + return found; +} + +void ksmbd_expire_sessions(void) +{ + struct ksmbd_session *sess; + u64 now = ktime_get_real_seconds(); + int bkt; + + down_read(&sessions_table_lock); + hash_for_each(sessions_table, bkt, sess, hlist) { + if (READ_ONCE(sess->state) != SMB2_SESSION_VALID || + !sess->kerberos_expiry || now < sess->kerberos_expiry) + continue; + + if (cmpxchg(&sess->state, SMB2_SESSION_VALID, + SMB2_SESSION_EXPIRED) == SMB2_SESSION_VALID) + ksmbd_counter_inc(KSMBD_COUNTER_SESSION_TIMEOUTS); + } + up_read(&sessions_table_lock); +} + static int ksmbd_chann_del(struct ksmbd_conn *conn, struct ksmbd_session *sess) { struct channel *chann; @@ -488,7 +596,7 @@ static int ksmbd_chann_del(struct ksmbd_conn *conn, struct ksmbd_session *sess) return 0; } -void ksmbd_sessions_deregister(struct ksmbd_conn *conn) +void ksmbd_conn_sessions_cleanup(struct ksmbd_conn *conn) { struct ksmbd_session *sess; unsigned long id; diff --git a/fs/smb/server/mgmt/user_session.h b/fs/smb/server/mgmt/user_session.h index 3e52d4cc1324..217258551d6d 100644 --- a/fs/smb/server/mgmt/user_session.h +++ b/fs/smb/server/mgmt/user_session.h @@ -72,6 +72,8 @@ struct ksmbd_session { struct rw_semaphore rpc_lock; }; +#define KSMBD_MAX_CHANNELS 32 + static inline int test_session_flag(struct ksmbd_session *sess, int bit) { return sess->flags & bit; @@ -98,7 +100,11 @@ bool is_ksmbd_session_in_connection(struct ksmbd_conn *conn, unsigned long long id); int ksmbd_session_register(struct ksmbd_conn *conn, struct ksmbd_session *sess); -void ksmbd_sessions_deregister(struct ksmbd_conn *conn); +void ksmbd_session_unregister(struct ksmbd_conn *conn, + struct ksmbd_session *sess); +void ksmbd_conn_sessions_cleanup(struct ksmbd_conn *conn); +bool ksmbd_conn_has_valid_or_expired_session(struct ksmbd_conn *conn); +void ksmbd_expire_sessions(void); struct ksmbd_session *__session_lookup(unsigned long long id); struct ksmbd_session *ksmbd_session_lookup_all(struct ksmbd_conn *conn, unsigned long long id); diff --git a/fs/smb/server/proc.c b/fs/smb/server/proc.c index 826353ed0553..19f0f2cfbf54 100644 --- a/fs/smb/server/proc.c +++ b/fs/smb/server/proc.c @@ -178,6 +178,8 @@ static int proc_show_ksmbd_stats(struct seq_file *m, void *v) proc_show_runtime_totals(m); seq_printf(m, "sessions:\t%lld\n", ksmbd_counter_sum(KSMBD_COUNTER_SESSIONS)); + seq_printf(m, "session_timeouts:\t%lld\n", + ksmbd_counter_sum(KSMBD_COUNTER_SESSION_TIMEOUTS)); seq_printf(m, "tree_connects:\t%lld\n", ksmbd_counter_sum(KSMBD_COUNTER_TREE_CONNS)); seq_printf(m, "requests:\t%lld\n", diff --git a/fs/smb/server/server.c b/fs/smb/server/server.c index 0069d4e6a60a..0827c8c51006 100644 --- a/fs/smb/server/server.c +++ b/fs/smb/server/server.c @@ -414,7 +414,7 @@ static int ksmbd_server_process_request(struct ksmbd_conn *conn) static int ksmbd_server_terminate_conn(struct ksmbd_conn *conn) { - ksmbd_sessions_deregister(conn); + ksmbd_conn_sessions_cleanup(conn); destroy_lease_table(conn); return 0; } diff --git a/fs/smb/server/smb2pdu.c b/fs/smb/server/smb2pdu.c index b7ce67094626..6b8809f67b92 100644 --- a/fs/smb/server/smb2pdu.c +++ b/fs/smb/server/smb2pdu.c @@ -85,8 +85,6 @@ struct channel *lookup_chann_list(struct ksmbd_session *sess, struct ksmbd_conn return chann; } -#define KSMBD_MAX_CHANNELS 32 - static int register_session_channel(struct ksmbd_session *sess, struct ksmbd_conn *conn, const char *sess_key) @@ -933,8 +931,14 @@ static bool smb2_session_expired_cmd_allowed(struct ksmbd_work *work, static bool smb2_session_kerberos_expired(struct ksmbd_session *sess) { - return sess->kerberos_expiry && - ktime_get_real_seconds() >= sess->kerberos_expiry; + if (!sess->kerberos_expiry || + ktime_get_real_seconds() < sess->kerberos_expiry) + return false; + + if (cmpxchg(&sess->state, SMB2_SESSION_VALID, + SMB2_SESSION_EXPIRED) == SMB2_SESSION_VALID) + ksmbd_counter_inc(KSMBD_COUNTER_SESSION_TIMEOUTS); + return true; } /** @@ -969,9 +973,8 @@ int smb2_check_user_session(struct ksmbd_work *work) if (!work->next_smb2_rcv_hdr_off && sess_id) work->sess = ksmbd_session_lookup_all_states(conn, sess_id); if (work->sess) { - if (smb2_session_kerberos_expired(work->sess)) { - work->sess->state = SMB2_SESSION_EXPIRED; - } else if (work->sess->state != SMB2_SESSION_VALID) { + if (!smb2_session_kerberos_expired(work->sess) && + work->sess->state != SMB2_SESSION_VALID) { ksmbd_user_session_put(work->sess); work->sess = NULL; } @@ -996,8 +999,7 @@ int smb2_check_user_session(struct ksmbd_work *work) sess_id, work->sess->id); return -EINVAL; } - if (smb2_session_kerberos_expired(work->sess)) - work->sess->state = SMB2_SESSION_EXPIRED; + smb2_session_kerberos_expired(work->sess); if (work->sess->state != SMB2_SESSION_VALID) { pr_err("compound request on a non-valid session (state %d)\n", work->sess->state); @@ -1014,7 +1016,6 @@ int smb2_check_user_session(struct ksmbd_work *work) work->sess = ksmbd_session_lookup_all_states(conn, sess_id); if (work->sess) { if (smb2_session_kerberos_expired(work->sess)) { - work->sess->state = SMB2_SESSION_EXPIRED; return smb2_session_expired_cmd_allowed(work, cmd) ? 1 : -EKEYEXPIRED; } @@ -2436,7 +2437,7 @@ int smb2_sess_setup(struct ksmbd_work *work) struct ksmbd_conn *conn = work->conn; struct smb2_sess_setup_req *req; struct smb2_sess_setup_rsp *rsp; - struct ksmbd_session *sess; + struct ksmbd_session *sess = NULL; struct negotiate_message *negblob; unsigned int negblob_len, negblob_off; int rc = 0; @@ -2592,6 +2593,9 @@ int smb2_sess_setup(struct ksmbd_work *work) goto out_err; } + if (work->session_setup_reauth) + WRITE_ONCE(sess->state, SMB2_SESSION_IN_PROGRESS); + conn->binding = false; } work->sess = sess; @@ -2703,6 +2707,14 @@ out_err: } if (rc < 0) { + bool setup_in_progress = sess && + READ_ONCE(sess->state) == SMB2_SESSION_IN_PROGRESS && + !(req->Flags & SMB2_SESSION_REQ_FLAG_BINDING); + + /* Authentication errors must not leave the new session published. */ + if (setup_in_progress) + ksmbd_session_unregister(conn, sess); + if (sess && conn->dialect == SMB311_PROT_ID && (req->Flags & SMB2_SESSION_REQ_FLAG_BINDING)) { struct preauth_session *preauth_sess; @@ -2736,7 +2748,8 @@ out_err: * For binding requests, session belongs to another * connection. Do not expire it. */ - if (!(req->Flags & SMB2_SESSION_REQ_FLAG_BINDING)) { + if (!(req->Flags & SMB2_SESSION_REQ_FLAG_BINDING) && + !setup_in_progress) { sess->last_active = jiffies; sess->kerberos_expiry = 0; sess->state = SMB2_SESSION_EXPIRED; @@ -6262,21 +6275,18 @@ err_out2: * @reqOutputBufferLength: max buffer length expected in command response * @fixed_len: minimum fixed response length * @rsp: query info response buffer contains output buffer length - * @rsp_org: base response buffer pointer in case of chained response * * Return: 0 on success, otherwise error */ static int buffer_check_err(int reqOutputBufferLength, unsigned int fixed_len, - struct smb2_query_info_rsp *rsp, - void *rsp_org) + struct smb2_query_info_rsp *rsp) { unsigned int output_len = le32_to_cpu(rsp->OutputBufferLength); if (reqOutputBufferLength < fixed_len) { pr_err("Invalid Buffer Size Requested\n"); rsp->hdr.Status = STATUS_INFO_LENGTH_MISMATCH; - *(__be32 *)rsp_org = cpu_to_be32(sizeof(struct smb2_hdr)); return -EINVAL; } @@ -6287,8 +6297,7 @@ static int buffer_check_err(int reqOutputBufferLength, return 0; } -static void get_standard_info_pipe(struct smb2_query_info_rsp *rsp, - void *rsp_org) +static void get_standard_info_pipe(struct smb2_query_info_rsp *rsp) { struct smb2_file_standard_info *sinfo; @@ -6303,8 +6312,7 @@ static void get_standard_info_pipe(struct smb2_query_info_rsp *rsp, cpu_to_le32(sizeof(struct smb2_file_standard_info)); } -static void get_internal_info_pipe(struct smb2_query_info_rsp *rsp, u64 num, - void *rsp_org) +static void get_internal_info_pipe(struct smb2_query_info_rsp *rsp, u64 num) { struct smb2_file_internal_info *file_info; @@ -6318,8 +6326,7 @@ static void get_internal_info_pipe(struct smb2_query_info_rsp *rsp, u64 num, static int smb2_get_info_file_pipe(struct ksmbd_session *sess, struct smb2_query_info_req *req, - struct smb2_query_info_rsp *rsp, - void *rsp_org) + struct smb2_query_info_rsp *rsp) { u64 id; int rc; @@ -6344,16 +6351,16 @@ static int smb2_get_info_file_pipe(struct ksmbd_session *sess, switch (req->FileInfoClass) { case FILE_STANDARD_INFORMATION: - get_standard_info_pipe(rsp, rsp_org); + get_standard_info_pipe(rsp); rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength), le32_to_cpu(rsp->OutputBufferLength), - rsp, rsp_org); + rsp); break; case FILE_INTERNAL_INFORMATION: - get_internal_info_pipe(rsp, id, rsp_org); + get_internal_info_pipe(rsp, id); rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength), le32_to_cpu(rsp->OutputBufferLength), - rsp, rsp_org); + rsp); break; default: ksmbd_debug(SMB, "smb2_info_file_pipe for %u not supported\n", @@ -7189,8 +7196,7 @@ static int smb2_get_info_file(struct ksmbd_work *work, if (test_share_config_flag(work->tcon->share_conf, KSMBD_SHARE_FLAG_PIPE)) { /* smb2 info file called for pipe */ - rc = smb2_get_info_file_pipe(work->sess, req, rsp, - work->response_buf); + rc = smb2_get_info_file_pipe(work->sess, req, rsp); goto iov_pin_out; } @@ -7299,13 +7305,16 @@ static int smb2_get_info_file(struct ksmbd_work *work, case FILE_ALTERNATE_NAME_INFORMATION: fixed_len = FILE_ALTERNATE_NAME_INFORMATION_SIZE; break; + case FILE_NORMALIZED_NAME_INFORMATION: + fixed_len = FILE_NORMALIZED_NAME_INFORMATION_SIZE; + break; case FILE_STREAM_INFORMATION: fixed_len = FILE_STREAM_INFORMATION_SIZE; break; } rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength), fixed_len, - rsp, work->response_buf); + rsp); } ksmbd_fd_put(work, fp); @@ -7576,7 +7585,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work, } rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength), fixed_len, - rsp, work->response_buf); + rsp); path_put(&path); if (!rc) @@ -7690,7 +7699,7 @@ release_acl: rsp->OutputBufferLength = cpu_to_le32(secdesclen); rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength), le32_to_cpu(rsp->OutputBufferLength), - rsp, work->response_buf); + rsp); if (rc) goto err_out; diff --git a/fs/smb/server/smb2pdu.h b/fs/smb/server/smb2pdu.h index 3f08d1ca5a38..ca8e27f7b712 100644 --- a/fs/smb/server/smb2pdu.h +++ b/fs/smb/server/smb2pdu.h @@ -61,8 +61,6 @@ struct preauth_integrity_info { #define SMB2_SESSION_IN_PROGRESS BIT(0) #define SMB2_SESSION_VALID BIT(1) -#define SMB2_SESSION_TIMEOUT (10 * HZ) - /* Apple Defined Contexts */ #define SMB2_CREATE_AAPL "AAPL" @@ -214,6 +212,7 @@ struct file_sparse { #define FILE_ALLOCATION_INFORMATION_SIZE 19 #define FILE_END_OF_FILE_INFORMATION_SIZE 20 #define FILE_ALTERNATE_NAME_INFORMATION_SIZE 8 +#define FILE_NORMALIZED_NAME_INFORMATION_SIZE 8 #define FILE_STREAM_INFORMATION_SIZE 32 #define FILE_PIPE_INFORMATION_SIZE 23 #define FILE_PIPE_LOCAL_INFORMATION_SIZE 24 diff --git a/fs/smb/server/stats.h b/fs/smb/server/stats.h index bc864efa0d46..8b32b8b4e8be 100644 --- a/fs/smb/server/stats.h +++ b/fs/smb/server/stats.h @@ -15,6 +15,7 @@ enum { KSMBD_COUNTER_SESSIONS = 0, + KSMBD_COUNTER_SESSION_TIMEOUTS, KSMBD_COUNTER_TREE_CONNS, KSMBD_COUNTER_REQUESTS, KSMBD_COUNTER_STATUS_SUCCESS, diff --git a/fs/super.c b/fs/super.c index 01db6124e409..9d4025213521 100644 --- a/fs/super.c +++ b/fs/super.c @@ -172,19 +172,6 @@ static void super_wake(struct super_block *sb, unsigned int flag) } /* - * The s_op->nr_cached_objects hooks (used for example by btrfs and xfs) - * operate on filesystem-global state and ignore sc->memcg. Driving them - * from per-memcg shrink_slab_memcg() invocations only burns CPU walking - * per-cpu counters and queueing duplicate work: the actual reclaim happens on - * the global path (kswapd or root direct reclaim) regardless. Restrict them - * to that path. - */ -static inline bool super_fs_objects_eligible(struct shrink_control *sc) -{ - return !sc->memcg || mem_cgroup_is_root(sc->memcg); -} - -/* * One thing we have to be careful of with a per-sb shrinker is that we don't * drop the last active reference to the superblock from within the shrinker. * If that happens we could trigger unregistering the shrinker from within the @@ -213,7 +200,7 @@ static unsigned long super_cache_scan(struct shrinker *shrink, if (!super_trylock_shared(sb)) return SHRINK_STOP; - if (sb->s_op->nr_cached_objects && super_fs_objects_eligible(sc)) + if (sb->s_op->nr_cached_objects) fs_objects = sb->s_op->nr_cached_objects(sb, sc); inodes = list_lru_shrink_count(&sb->s_inode_lru, sc); @@ -274,8 +261,7 @@ static unsigned long super_cache_count(struct shrinker *shrink, return 0; smp_rmb(); - if (sb->s_op && sb->s_op->nr_cached_objects && - super_fs_objects_eligible(sc)) + if (sb->s_op && sb->s_op->nr_cached_objects) total_objects = sb->s_op->nr_cached_objects(sb, sc); total_objects += list_lru_shrink_count(&sb->s_dentry_lru, sc); diff --git a/include/keys/request_key_auth-type.h b/include/keys/request_key_auth-type.h index 01e42ee5f409..464636278c4f 100644 --- a/include/keys/request_key_auth-type.h +++ b/include/keys/request_key_auth-type.h @@ -22,7 +22,7 @@ struct request_key_auth { const struct cred *cred; void *callout_info; size_t callout_len; - pid_t pid; + struct pid *pid; char op[8]; } __randomize_layout; diff --git a/include/linux/dma-fence.h b/include/linux/dma-fence.h index 158cd609f103..ffa99b930843 100644 --- a/include/linux/dma-fence.h +++ b/include/linux/dma-fence.h @@ -141,6 +141,9 @@ struct dma_fence_ops { * compute the name at runtime, without having it to store permanently * for each fence, or build a cache of some sort. * + * The returned string is RCU protected and can be freed after the fence + * signaled and a RCU grace period passed. + * * This callback is mandatory. */ const char * (*get_driver_name)(struct dma_fence *fence); @@ -153,6 +156,9 @@ struct dma_fence_ops { * having it to store permanently for each fence, or build a cache of * some sort. * + * The returned string is RCU protected and can be freed after the fence + * signaled and a RCU grace period passed. + * * This callback is mandatory. */ const char * (*get_timeline_name)(struct dma_fence *fence); diff --git a/include/linux/ieee80211-mesh.h b/include/linux/ieee80211-mesh.h index 7eb15834531c..9e548b9173df 100644 --- a/include/linux/ieee80211-mesh.h +++ b/include/linux/ieee80211-mesh.h @@ -361,8 +361,7 @@ ieee80211_mesh_hwmp_perr_get_rcode(const u8 *ie, u8 dst_idx) /* IEEE Std 802.11-2016 9.4.2.113 PREQ element */ static inline bool ieee80211_mesh_preq_size_ok(const u8 *pos, u8 elen) { - struct ieee80211_mesh_hwmp_preq_bottom *preq_elem_bottom = - ieee80211_mesh_hwmp_preq_get_bottom(pos); + struct ieee80211_mesh_hwmp_preq_bottom *preq_elem_bottom; u8 target_count; int needed; @@ -378,6 +377,7 @@ static inline bool ieee80211_mesh_preq_size_ok(const u8 *pos, u8 elen) if (elen < needed) return false; + preq_elem_bottom = ieee80211_mesh_hwmp_preq_get_bottom(pos); target_count = preq_elem_bottom->target_count; /* IEEE Std 802.11-2016 Table 14-10 to 14-16 */ if (target_count < 1) diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index bec0eab6ef03..b8a13b8d36a4 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -630,6 +630,8 @@ static inline void mmap_read_unlock(struct mm_struct *mm) DEFINE_GUARD(mmap_read_lock, struct mm_struct *, mmap_read_lock(_T), mmap_read_unlock(_T)) DEFINE_GUARD_COND(mmap_read_lock, _try, mmap_read_trylock(_T)) +DEFINE_GUARD(mmap_write_lock, struct mm_struct *, + mmap_write_lock(_T), mmap_write_unlock(_T)) static inline void mmap_read_unlock_non_owner(struct mm_struct *mm) { diff --git a/include/net/bluetooth/coredump.h b/include/net/bluetooth/coredump.h index 1f071ab55416..acc1849f66c0 100644 --- a/include/net/bluetooth/coredump.h +++ b/include/net/bluetooth/coredump.h @@ -70,6 +70,7 @@ struct hci_devcoredump { const char *hci_devcd_state_name(enum devcoredump_state state); void hci_devcd_reset(struct hci_dev *hdev); +void hci_devcd_shutdown(struct hci_dev *hdev); void hci_devcd_rx(struct work_struct *work); void hci_devcd_timeout(struct work_struct *work); @@ -89,6 +90,7 @@ static inline const char *hci_devcd_state_name(enum devcoredump_state state) } static inline void hci_devcd_reset(struct hci_dev *hdev) {} +static inline void hci_devcd_shutdown(struct hci_dev *hdev) {} static inline void hci_devcd_rx(struct work_struct *work) {} static inline void hci_devcd_timeout(struct work_struct *work) {} diff --git a/include/net/codel.h b/include/net/codel.h index aa80f744826c..183d43c2bd43 100644 --- a/include/net/codel.h +++ b/include/net/codel.h @@ -140,6 +140,11 @@ struct codel_vars { /* needed shift to get a Q0.32 number from rec_inv_sqrt */ #define REC_INV_SQRT_SHIFT (32 - REC_INV_SQRT_BITS) +/* Cap on drops per codel_dequeue() call: the loop's work depends on the + * idle gap and backlog, both outside our control; resync when exceeded. + */ +#define CODEL_MAX_DROPS_PER_DEQUEUE 256 + /** * struct codel_stats - contains codel shared variables and stats * @maxpacket: largest packet we've seen so far diff --git a/include/net/codel_impl.h b/include/net/codel_impl.h index 2c1f0ec309e9..8f26132d45b7 100644 --- a/include/net/codel_impl.h +++ b/include/net/codel_impl.h @@ -93,12 +93,17 @@ static void codel_Newton_step(struct codel_vars *vars) * CoDel control_law is t + interval/sqrt(count) * We maintain in rec_inv_sqrt the reciprocal value of sqrt(count) to avoid * both sqrt() and divide operation. + * + * Clamp the increment to at least 1 tick: a very small interval (or a + * large count) can truncate it to zero, stalling the dropping loop. */ static codel_time_t codel_control_law(codel_time_t t, codel_time_t interval, u32 rec_inv_sqrt) { - return t + reciprocal_scale(interval, rec_inv_sqrt << REC_INV_SQRT_SHIFT); + return t + max_t(u32, 1, + reciprocal_scale(interval, + rec_inv_sqrt << REC_INV_SQRT_SHIFT)); } static bool codel_should_drop(const struct sk_buff *skb, @@ -154,6 +159,7 @@ static struct sk_buff *codel_dequeue(void *ctx, codel_skb_dequeue_t dequeue_func) { struct sk_buff *skb = dequeue_func(vars, ctx); + unsigned int drops = 0; codel_time_t now; bool drop; @@ -180,6 +186,14 @@ static struct sk_buff *codel_dequeue(void *ctx, */ while (vars->dropping && codel_time_after_eq(now, vars->drop_next)) { + if (++drops > CODEL_MAX_DROPS_PER_DEQUEUE) { + /* fell far behind the schedule */ + WRITE_ONCE(vars->drop_next, + codel_control_law(now, + params->interval, + vars->rec_inv_sqrt)); + break; + } /* dont care of possible wrap * since there is no more divide. */ diff --git a/include/net/mac80211.h b/include/net/mac80211.h index 9d1fac6e8082..ed6a5874ff96 100644 --- a/include/net/mac80211.h +++ b/include/net/mac80211.h @@ -7638,11 +7638,14 @@ bool ieee80211_tx_prepare_skb(struct ieee80211_hw *hw, * * @skb: packet injected by userspace * @dev: the &struct device of this 802.11 device + * @chandef: the channel definition the frame will be transmitted on, or + * %NULL to skip the bandwidth checks * * Return: %true if the radiotap header was parsed, %false otherwise */ bool ieee80211_parse_tx_radiotap(struct sk_buff *skb, - struct net_device *dev); + struct net_device *dev, + const struct cfg80211_chan_def *chandef); /** * struct ieee80211_noa_data - holds temporary data for tracking P2P NoA state diff --git a/include/net/sock.h b/include/net/sock.h index 51185222aac2..60ea55dc1885 100644 --- a/include/net/sock.h +++ b/include/net/sock.h @@ -2312,6 +2312,8 @@ static inline void sk_gso_disable(struct sock *sk) sk->sk_route_caps &= ~NETIF_F_GSO_MASK; } +bool sk_has_decrypt_user(const struct sock *sk); + static inline int skb_do_copy_data_nocache(struct sock *sk, struct sk_buff *skb, struct iov_iter *from, char *to, int copy, int offset) diff --git a/include/rdma/uverbs_types.h b/include/rdma/uverbs_types.h index 5a07f9a6dcd1..6f3622892c0c 100644 --- a/include/rdma/uverbs_types.h +++ b/include/rdma/uverbs_types.h @@ -180,8 +180,6 @@ struct ib_uverbs_file { struct page *disassociate_page; struct xarray idr; - - struct mutex disassociation_lock; }; extern const struct uverbs_obj_type_class uverbs_idr_class; diff --git a/include/sound/soc.h b/include/sound/soc.h index f46b2bc2a022..5afc34b147b5 100644 --- a/include/sound/soc.h +++ b/include/sound/soc.h @@ -699,7 +699,8 @@ struct snd_soc_dai_link_component { struct snd_soc_dai_link_ch_map { unsigned int cpu; unsigned int codec; - unsigned int ch_mask; + unsigned int cpu_ch_mask; + unsigned int codec_ch_mask; }; struct snd_soc_dai_link { diff --git a/include/sound/soc_sdw_utils.h b/include/sound/soc_sdw_utils.h index 9b28e9aef4f1..9fbb69b9052d 100644 --- a/include/sound/soc_sdw_utils.h +++ b/include/sound/soc_sdw_utils.h @@ -250,8 +250,6 @@ int asoc_sdw_cs_amp_init(struct snd_soc_card *card, struct snd_soc_dai_link *dai_links, struct asoc_sdw_codec_info *info, bool playback); -int asoc_sdw_cs_spk_feedback_rtd_init(struct snd_soc_pcm_runtime *rtd, - struct snd_soc_dai *dai); int asoc_sdw_cs35l56_volume_limit(struct snd_soc_card *card, const char *name_prefix); /* MAXIM codec support */ diff --git a/include/trace/events/dma.h b/include/trace/events/dma.h index 9df02c1511de..b06d8f99922d 100644 --- a/include/trace/events/dma.h +++ b/include/trace/events/dma.h @@ -134,7 +134,7 @@ DECLARE_EVENT_CLASS(dma_alloc_class, TP_fast_assign( __assign_str(device); __entry->virt_addr = virt_addr; - __entry->dma_addr = dma_addr; + __entry->dma_addr = virt_addr ? dma_addr : 0; __entry->size = size; __entry->flags = flags; __entry->dir = dir; diff --git a/include/uapi/linux/input-event-codes.h b/include/uapi/linux/input-event-codes.h index 3528168f7c6d..4217950e5d16 100644 --- a/include/uapi/linux/input-event-codes.h +++ b/include/uapi/linux/input-event-codes.h @@ -970,6 +970,12 @@ /* * LEDs + * + * Do not add any new LED definitions to the input subsystem. The existing + * definitions are legacy and grandfathered for backwards compatibility with + * userspace (via evdev). Any new LEDs should be implemented using the + * LED subsystem (struct led_classdev). The input core provides a bridge to + * the LED subsystem in drivers/input/input-leds.c. */ #define LED_NUML 0x00 diff --git a/include/uapi/rdma/bnxt_re-abi.h b/include/uapi/rdma/bnxt_re-abi.h index 856a1b3036e9..15ed2a51f8f1 100644 --- a/include/uapi/rdma/bnxt_re-abi.h +++ b/include/uapi/rdma/bnxt_re-abi.h @@ -250,7 +250,7 @@ struct bnxt_re_query_device_ex_resp { struct bnxt_re_db_region { __u32 dpi; __u32 reserved; - __aligned_u64 umdbr; + __aligned_u64 reserved2; }; enum bnxt_re_obj_dbr_alloc_attrs { diff --git a/kernel/cgroup/cgroup.c b/kernel/cgroup/cgroup.c index 2d532bf2c0c7..227d09704ca5 100644 --- a/kernel/cgroup/cgroup.c +++ b/kernel/cgroup/cgroup.c @@ -5303,10 +5303,13 @@ struct task_struct *css_task_iter_next(struct css_task_iter *it) if (it->flags & CSS_TASK_ITER_SKIPPED) css_task_iter_advance(it); - if (it->task_pos) { + while (it->task_pos && !it->cur_task) { it->cur_task = list_entry(it->task_pos, struct task_struct, cg_list); - get_task_struct(it->cur_task); + /* a task on dying_tasks with zero refcount is only valid for + * RCU readers, not even interesting for + * CSS_TASK_ITER_WITH_DEAD, find another one */ + it->cur_task = tryget_task_struct(it->cur_task); css_task_iter_advance(it); } diff --git a/kernel/dma/coherent.c b/kernel/dma/coherent.c index 45bbae947f4b..4d0266893bcc 100644 --- a/kernel/dma/coherent.c +++ b/kernel/dma/coherent.c @@ -352,8 +352,7 @@ static int rmem_dma_device_init(struct reserved_mem *rmem, struct device *dev) min_not_zero(dev->coherent_dma_mask, dev->bus_dma_limit)) dev_warn(dev, "reserved memory is beyond device's set DMA address range\n"); - dma_assign_coherent_memory(dev, mem); - return 0; + return dma_assign_coherent_memory(dev, mem); } static void rmem_dma_device_release(struct reserved_mem *rmem, diff --git a/kernel/dma/swiotlb.c b/kernel/dma/swiotlb.c index ded7016a46a7..aa2f1c4588b9 100644 --- a/kernel/dma/swiotlb.c +++ b/kernel/dma/swiotlb.c @@ -1019,7 +1019,6 @@ static void swiotlb_bounce(struct device *dev, phys_addr_t tlb_addr, size_t size int index = (tlb_addr - mem->start) >> IO_TLB_SHIFT; phys_addr_t orig_addr = mem->slots[index].orig_addr; size_t alloc_size = mem->slots[index].alloc_size; - unsigned long pfn = PFN_DOWN(orig_addr); unsigned char *vaddr = mem->vaddr + tlb_addr - mem->start; int tlb_offset; @@ -1052,7 +1051,8 @@ static void swiotlb_bounce(struct device *dev, phys_addr_t tlb_addr, size_t size size = alloc_size; } - if (PageHighMem(pfn_to_page(pfn))) { + if (PhysHighMem(orig_addr)) { + unsigned long pfn = PFN_DOWN(orig_addr); unsigned int offset = orig_addr & ~PAGE_MASK; struct page *page; unsigned int sz = 0; diff --git a/kernel/events/core.c b/kernel/events/core.c index fe33fe15689d..db7b76d6b68a 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -6350,6 +6350,9 @@ static DEFINE_MUTEX(perf_mediated_pmu_mutex); /* !exclude_guest event of PMU with PERF_PMU_CAP_MEDIATED_VPMU */ static inline bool is_include_guest_event(struct perf_event *event) { + if (!event->pmu) + return false; + if ((event->pmu->capabilities & PERF_PMU_CAP_MEDIATED_VPMU) && !event->attr.exclude_guest) return true; @@ -13002,6 +13005,7 @@ static void __pmu_detach_event(struct pmu *pmu, struct perf_event *event, exclusive_event_destroy(event); module_put(pmu->module); + mediated_pmu_unaccount_event(event); event->pmu = NULL; /* force fault instead of UAF */ } diff --git a/kernel/exit.c b/kernel/exit.c index 4e028f157597..424c44a42a4d 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -302,12 +302,13 @@ repeat: free_pids(post.pids); release_thread(p); /* - * This task was already removed from the process/thread/pid lists - * and lock_task_sighand(p) can't succeed. Nobody else can touch - * ->pending or, if group dead, signal->shared_pending. We can call - * flush_sigqueue() lockless. + * This task was already removed from the process/thread/pid lists and + * lock_task_sighand(p) can't succeed. If it's the group leader then + * flush tsk->signal->shared_pending. tsk->pending has been flushed + * already in exit_signals(). Nothing else can touch + * signal->shared_pending anymore, so flush_sigqueue() can be invoked + * lockless. */ - flush_sigqueue(&p->pending); if (thread_group_leader(p)) flush_sigqueue(&p->signal->shared_pending); diff --git a/kernel/fork.c b/kernel/fork.c index a5934a317634..5ef413368912 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1996,9 +1996,9 @@ static bool need_futex_hash_allocate_default(u64 clone_flags) { /* * Allocate a default futex hash for any sibling that will - * share the parent's mm, except vfork. + * share the parent's mm. */ - return (clone_flags & (CLONE_VM | CLONE_VFORK)) == CLONE_VM; + return clone_flags & CLONE_VM; } /* diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 7885ff76e69f..0b846a13c628 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -3351,6 +3351,8 @@ void relax_compatible_cpus_allowed_ptr(struct task_struct *p) void set_task_cpu(struct task_struct *p, unsigned int new_cpu) { unsigned int state = READ_ONCE(p->__state); + bool proxy_migrated = sched_proxy_exec() && p->is_blocked && + task_cpu(p) != p->wake_cpu; /* * We should never call set_task_cpu() on a blocked task, @@ -3386,7 +3388,12 @@ void set_task_cpu(struct task_struct *p, unsigned int new_cpu) */ WARN_ON_ONCE(!cpu_online(new_cpu)); - WARN_ON_ONCE(is_migration_disabled(p)); + /* + * Proxy execution can move a blocked task's scheduling context to any + * CPU without moving its migration-disabled execution context. The + * wakeup path will return the task to a CPU where it can execute. + */ + WARN_ON_ONCE(is_migration_disabled(p) && !proxy_migrated); trace_sched_migrate_task(p, new_cpu); diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index 51de1d8b72a1..3219f0da0fe4 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -2919,7 +2919,7 @@ static inline void maybe_queue_balance_callback(struct rq *rq) static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev) { - struct scx_sched *sch = scx_root_protected_live(); + struct scx_sched *root_sch = scx_root_protected_live(); enum scx_dsp_verdict verdict; s32 cpu = cpu_of(rq); @@ -2928,7 +2928,7 @@ static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev scx_process_sync_ecaps(rq, prev); - if ((sch->ops.flags & SCX_OPS_HAS_CPU_PREEMPT) && + if ((root_sch->ops.flags & SCX_OPS_HAS_CPU_PREEMPT) && unlikely(rq->scx.cpu_released)) { /* * If the previous sched_class for the current CPU was not SCX, @@ -2936,8 +2936,8 @@ static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev * core. This callback complements ->cpu_release(), which is * emitted in switch_class(). */ - if (sch->ops.cpu_acquire) - SCX_CALL_OP(sch, cpu_acquire, rq, cpu, NULL); + if (root_sch->ops.cpu_acquire) + SCX_CALL_OP(root_sch, cpu_acquire, rq, cpu, NULL); rq->scx.cpu_released = false; } @@ -2955,7 +2955,7 @@ static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev * test. */ if ((prev->scx.flags & SCX_TASK_QUEUED) && prev->scx.slice && - !scx_bypassing(sch, cpu)) { + !scx_bypassing(scx_task_sched(prev), cpu)) { verdict = SCX_DSP_PREV; goto has_tasks; } @@ -2967,20 +2967,25 @@ static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev goto has_tasks; } - verdict = scx_dispatch_sched(sch, rq, prev, false); + verdict = scx_dispatch_sched(root_sch, rq, prev, false); if (verdict != SCX_DSP_NONE) goto has_tasks; /* - * Didn't find another task to run. Keep running @prev unless - * %SCX_OPS_ENQ_LAST is in effect. + * Didn't find another task to run. Keep running @prev unless its own + * scheduler set %SCX_OPS_ENQ_LAST and takes the enqueue instead, see + * put_prev_task_scx(). Read the scheduler here as the dispatch above + * may have dropped the rq lock while @prev changed class or scheduler. */ - if ((prev->scx.flags & SCX_TASK_QUEUED) && - (!(sch->ops.flags & SCX_OPS_ENQ_LAST) || scx_bypassing(sch, cpu)) && - scx_task_can_stay_on_cpu(rq, prev)) { - __scx_add_event(sch, SCX_EV_DISPATCH_KEEP_LAST, 1); - verdict = SCX_DSP_PREV; - goto has_tasks; + if (prev->scx.flags & SCX_TASK_QUEUED) { + struct scx_sched *prev_sch = scx_task_sched(prev); + + if ((!(prev_sch->ops.flags & SCX_OPS_ENQ_LAST) || + scx_bypassing(prev_sch, cpu)) && scx_task_can_stay_on_cpu(rq, prev)) { + __scx_add_event(prev_sch, SCX_EV_DISPATCH_KEEP_LAST, 1); + verdict = SCX_DSP_PREV; + goto has_tasks; + } } rq->scx.flags &= ~SCX_RQ_IN_DISPATCH; return SCX_DSP_NONE; @@ -3665,8 +3670,20 @@ static void handle_hotplug(struct rq *rq, bool online) s16 *tbl = rcu_dereference_check(scx_cpu_to_cid_tbl, lockdep_is_cpus_held()); - if (tbl) + if (tbl) { + struct scx_sched *pos; + cpu_or_cid = tbl[cpu]; + + guard(raw_spinlock_irqsave)(&scx_sched_lock); + list_for_each_entry(pos, &scx_sched_all, all) { + struct scx_cmask *mask = pos->online_cmask; + + if (mask) + __assign_bit(cpu_or_cid, (unsigned long *)mask->bits, + online); + } + } } if (online && SCX_HAS_OP(sch, cpu_online)) @@ -4766,7 +4783,8 @@ int scx_tg_online(struct task_group *tg) { .weight = tg->scx.weight, .bw_period_us = tg->scx.bw_period_us, .bw_quota_us = tg->scx.bw_quota_us, - .bw_burst_us = tg->scx.bw_burst_us }; + .bw_burst_us = tg->scx.bw_burst_us, + .sched_idle = tg->scx.idle }; ret = SCX_CALL_OP_RET(sch, cgroup_init, NULL, tg->css.cgroup, &args); @@ -4932,7 +4950,8 @@ void scx_group_set_idle(struct task_group *tg, bool idle) percpu_down_read(&scx_cgroup_ops_rwsem); sch = scx_tg_knob_sched(tg); - if (scx_cgroup_enabled && sch && SCX_HAS_OP(sch, cgroup_set_idle)) + if (scx_cgroup_enabled && sch && SCX_HAS_OP(sch, cgroup_set_idle) && + tg->scx.idle != idle) SCX_CALL_OP(sch, cgroup_set_idle, NULL, tg_cgrp(tg), idle); /* Update the task group's idle state */ @@ -5187,6 +5206,7 @@ static int scx_cgroup_init(struct scx_sched *sch) .bw_period_us = tg->scx.bw_period_us, .bw_quota_us = tg->scx.bw_quota_us, .bw_burst_us = tg->scx.bw_burst_us, + .sched_idle = tg->scx.idle, }; ret = SCX_CALL_OP_RET(sch, cgroup_init, NULL, css->cgroup, &args); @@ -5272,12 +5292,17 @@ static void free_exit_info(struct scx_exit_info *ei); static const char *scx_exit_reason(enum scx_exit_kind kind); static bool scx_claim_exit(struct scx_sched *sch, enum scx_exit_kind kind); -s32 scx_set_cmask_scratch_alloc(struct scx_sched *sch) +s32 scx_alloc_kern_arena_objs(struct scx_sched *sch) { size_t size = struct_size_t(struct scx_cmask, bits, SCX_CMASK_NR_WORDS(num_possible_cpus())); + struct scx_cmask *online; + struct scx_cmask_ref ref; int cpu; + /* hotplug stays excluded until the online mask is published */ + lockdep_assert_cpus_held(); + if (!sch->is_cid_type || !sch->arena_pool) return 0; @@ -5293,15 +5318,28 @@ s32 scx_set_cmask_scratch_alloc(struct scx_sched *sch) return -ENOMEM; scx_cmask_init(*slot, 0, num_possible_cpus()); } + + /* pack the online mask alongside the scratch masks */ + online = scx_arena_alloc(sch, size); + if (!online) + return -ENOMEM; + + scoped_guard(rcu) { + scx_cmask_ref_init_kern(sch, online, 0, num_possible_cpus(), &ref); + scx_cmask_ref_from_cpumask(&ref, cpu_active_mask); + } + sch->online_cmask = online; + return 0; } -static void scx_set_cmask_scratch_free(struct scx_sched *sch) +static void scx_free_kern_arena_objs(struct scx_sched *sch) { size_t size = struct_size_t(struct scx_cmask, bits, SCX_CMASK_NR_WORDS(num_possible_cpus())); int cpu; + scx_arena_free(sch, sch->online_cmask, size); if (!sch->set_cmask_scratch) return; @@ -5388,7 +5426,7 @@ static void scx_sched_free_rcu_work(struct work_struct *work) rhashtable_free_and_destroy(&sch->dsq_hash, NULL, NULL); free_exit_info(sch->exit_info); - scx_set_cmask_scratch_free(sch); + scx_free_kern_arena_objs(sch); scx_arena_pool_destroy(sch); if (sch->arena_map) bpf_map_put(sch->arena_map); @@ -7508,22 +7546,24 @@ static void scx_root_enable_workfn(struct kthread_work *work) #ifdef CONFIG_EXT_SUB_SCHED cgroup_get(cgrp); #endif + /* + * Transition to ENABLING to arm the disable path. Allocation failure + * still unwinds locally. Full disabling on failure applies only after + * scx_alloc_and_add_sched() succeeds. + */ + WARN_ON_ONCE(scx_set_enable_state(SCX_ENABLING) != SCX_DISABLED); + WARN_ON_ONCE(scx_root); + sch = scx_alloc_and_add_sched(cmd, cgrp, NULL); if (IS_ERR(sch)) { ret = PTR_ERR(sch); + WARN_ON_ONCE(scx_set_enable_state(SCX_DISABLED) != SCX_ENABLING); goto err_free_tid_hash; } if (sch->is_cid_type) static_branch_enable(&__scx_is_cid_type); - /* - * Transition to ENABLING and clear exit info to arm the disable path. - * Failure triggers full disabling from here on. - */ - WARN_ON_ONCE(scx_set_enable_state(SCX_ENABLING) != SCX_DISABLED); - WARN_ON_ONCE(scx_root); - atomic_long_set(&scx_nr_rejected, 0); for_each_possible_cpu(cpu) { @@ -7591,7 +7631,7 @@ static void scx_root_enable_workfn(struct kthread_work *work) goto err_disable; } - ret = scx_set_cmask_scratch_alloc(sch); + ret = scx_alloc_kern_arena_objs(sch); if (ret) { cpus_read_unlock(); goto err_disable; @@ -8946,10 +8986,17 @@ __bpf_kfunc void scx_bpf_dsq_insert_vtime(struct task_struct *p, u64 dsq_id, #ifdef CONFIG_EXT_SUB_SCHED /* * Disallow if any sub-scheds are attached. There is no way to tell - * which scheduler called us, just error out @p's scheduler. + * which scheduler called us, so error out @p's scheduler -- read it + * under RCU as @p's locks aren't necessarily held here. @p may be a + * task past sched_ext_dead() or an idle task, in which case its + * scheduler can't be determined and there is nothing obviously wrong + * to report; just refuse the call. */ if (unlikely(!list_empty(&sch->children))) { - scx_error(scx_task_sched(p), "__scx_bpf_dsq_insert_vtime() must be used"); + struct scx_sched *tsch = scx_task_sched_rcu(p); + + if (tsch) + scx_error(tsch, "__scx_bpf_dsq_insert_vtime() must be used"); return; } #endif @@ -10321,7 +10368,8 @@ __bpf_kfunc u32 scx_bpf_nr_cids(void) * hotplug, which lets schedulers treat [0, nr_online_cids) as the online * range. Schedulers that prefer to handle hotplug without a restart should * install a custom mapping via scx_bpf_cid_override() and track onlining - * through the ops.cid_online / ops.cid_offline callbacks. + * through the ops.cid_online / ops.cid_offline callbacks, starting from the + * mask scx_bpf_online_cmask() returns. */ __bpf_kfunc u32 scx_bpf_nr_online_cids(void) { @@ -10329,6 +10377,37 @@ __bpf_kfunc u32 scx_bpf_nr_online_cids(void) } /** + * scx_bpf_online_cmask - Return the online cid mask in the scheduler arena + * @aux: implicit BPF argument to access bpf_prog_aux hidden from BPF progs + * + * Return a kernel-maintained cmask covering [0, scx_bpf_nr_cids()), or NULL if + * the calling program is not associated with a live cid-form scheduler or the + * mask is not allocated yet, as in ops.init_cids(). Treat the mask as read-only + * even though arena memory stays writable by the BPF scheduler. The mask + * follows the SCX hotplug notifications: a cid's bit is updated before + * ops.cid_online/offline() runs for it. The pointer is valid from ops.init() + * through ops.exit(). Root ops.init() runs with hotplug excluded. Other + * contexts can observe concurrent updates. + */ +__bpf_kfunc const void *scx_bpf_online_cmask(const struct bpf_prog_aux *aux) +{ + struct scx_sched *sch; + struct scx_cmask *online; + + guard(rcu)(); + + sch = scx_prog_sched(aux); + if (unlikely(!sch)) + return NULL; + online = sch->online_cmask; + if (unlikely(!online)) + return NULL; + + /* BPF rebases by the low 32 bits, like __arena callback args */ + return (void *)((unsigned long)online - sch->arena_kern_base); +} + +/** * scx_bpf_this_cid - Return the cid of the CPU this program is running on * * cid-addressed equivalent of bpf_get_smp_processor_id() for scx programs. @@ -10691,6 +10770,7 @@ BTF_ID_FLAGS(func, scx_bpf_nr_node_ids) BTF_ID_FLAGS(func, scx_bpf_nr_cpu_ids) BTF_ID_FLAGS(func, scx_bpf_nr_cids) BTF_ID_FLAGS(func, scx_bpf_nr_online_cids) +BTF_ID_FLAGS(func, scx_bpf_online_cmask, KF_IMPLICIT_ARGS | KF_ARENA_RET) BTF_ID_FLAGS(func, scx_bpf_this_cid) BTF_ID_FLAGS(func, scx_bpf_get_possible_cpumask, KF_ACQUIRE) BTF_ID_FLAGS(func, scx_bpf_get_online_cpumask, KF_ACQUIRE) diff --git a/kernel/sched/ext/idle.c b/kernel/sched/ext/idle.c index d2973fb3af6d..aa9fb6de0ad6 100644 --- a/kernel/sched/ext/idle.c +++ b/kernel/sched/ext/idle.c @@ -1142,10 +1142,17 @@ __bpf_kfunc s32 scx_bpf_select_cpu_and(struct task_struct *p, s32 prev_cpu, u64 #ifdef CONFIG_EXT_SUB_SCHED /* * Disallow if any sub-scheds are attached. There is no way to tell - * which scheduler called us, just error out @p's scheduler. + * which scheduler called us, so error out @p's scheduler -- read it + * under RCU as @p's locks aren't necessarily held here. @p may be a + * task past sched_ext_dead() or an idle task, in which case its + * scheduler can't be determined and there is nothing obviously wrong + * to report; just refuse the call. */ if (unlikely(!list_empty(&sch->children))) { - scx_error(scx_task_sched(p), "__scx_bpf_select_cpu_and() must be used"); + struct scx_sched *tsch = scx_task_sched_rcu(p); + + if (tsch) + scx_error(tsch, "__scx_bpf_select_cpu_and() must be used"); return -EINVAL; } #endif diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h index 0967b99a4948..3464e0f113c1 100644 --- a/kernel/sched/ext/internal.h +++ b/kernel/sched/ext/internal.h @@ -259,6 +259,9 @@ struct scx_cgroup_init_args { u64 bw_period_us; u64 bw_quota_us; u64 bw_burst_us; + + /* whether the cgroup is configured SCHED_IDLE via cpu.idle */ + bool sched_idle; }; enum scx_cpu_preempt_reason { @@ -569,6 +572,12 @@ struct sched_ext_ops { * * Specify the %SCX_OPS_KEEP_BUILTIN_IDLE flag to keep the built-in idle * tracking. + * + * Only actual transitions are reported. A CPU that is claimed with an + * idle pick and kicked but dispatches no task returns to idle without a + * transition. A scheduler tracking idle CPUs itself must restore the + * idle state from ops.dispatch() when it returns without the next task + * to run. */ void (*update_idle)(s32 cpu, bool idle); @@ -1552,6 +1561,7 @@ struct scx_sched { * and passes it to the callback's __arena argument. */ struct scx_cmask * __percpu *set_cmask_scratch; + struct scx_cmask *online_cmask; DECLARE_BITMAP(has_op, SCX_OPI_END); @@ -2078,7 +2088,7 @@ void scx_disable_and_exit_task(struct scx_sched *sch, struct task_struct *p); void scx_cgroup_lock(void); void scx_cgroup_unlock(void); #endif -s32 scx_set_cmask_scratch_alloc(struct scx_sched *sch); +s32 scx_alloc_kern_arena_objs(struct scx_sched *sch); void scx_disable_bypass_dsp(struct scx_sched *sch); void scx_bypass(struct scx_sched *sch, bool bypass); s32 scx_link_sched(struct scx_sched *sch); diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c index 9e7040482bde..34e642a1a403 100644 --- a/kernel/sched/ext/sub.c +++ b/kernel/sched/ext/sub.c @@ -1361,6 +1361,7 @@ static s32 scx_cgroup_claim_subtree(struct scx_sched *sch) .bw_period_us = tg->scx.bw_period_us, .bw_quota_us = tg->scx.bw_quota_us, .bw_burst_us = tg->scx.bw_burst_us, + .sched_idle = tg->scx.idle, }; if (tg->scx.sched != parent || @@ -1464,6 +1465,7 @@ static void scx_cgroup_return_subtree(struct scx_sched *sch) .bw_period_us = tg->scx.bw_period_us, .bw_quota_us = tg->scx.bw_quota_us, .bw_burst_us = tg->scx.bw_burst_us, + .sched_idle = tg->scx.idle, }; /* the first pass must have transferred everything */ @@ -1803,6 +1805,12 @@ void scx_sub_enable_workfn(struct kthread_work *work) goto err_disable; } + scoped_guard(cpus_read_lock) { + ret = scx_alloc_kern_arena_objs(sch); + if (ret) + goto err_disable; + } + if (sch->ops.init) { ret = SCX_CALL_OP_RET(sch, init, NULL); if (ret) { @@ -1813,10 +1821,6 @@ void scx_sub_enable_workfn(struct kthread_work *work) sch->exit_info->flags |= SCX_EFLAG_INITIALIZED; } - ret = scx_set_cmask_scratch_alloc(sch); - if (ret) - goto err_disable; - struct scx_sub_attach_args sub_attach_args = { .ops = &sch->ops, .cgroup_path = sch->cgrp_path, diff --git a/kernel/signal.c b/kernel/signal.c index ec30550951ec..d31ebcb6ed4d 100644 --- a/kernel/signal.c +++ b/kernel/signal.c @@ -457,18 +457,42 @@ static void __sigqueue_free(struct sigqueue *q) kmem_cache_free(sigqueue_cachep, q); } -void flush_sigqueue(struct sigpending *queue) +/* + * flush_sigqueue_list() can only be invoked without holding sighand::siglock in + * the following cases: + * + * 1) When flushing task::pending _after_ setting task::flags PF_EXITING + * + * All functions which try to send a signal to @task will observe PF_EXITING + * and drop the signal. + * + * 2) When flushing task::signal::shared_pending _after_ the last task in a + * thread group was unhashed and task::sighand is NULL. + * + * Nothing can queue a signal anymore because sighand is NULL. + */ +static void flush_sigqueue_list(struct list_head *head) { - struct sigqueue *q; + struct sigqueue *q, *tmp; - sigemptyset(&queue->signal); - while (!list_empty(&queue->list)) { - q = list_entry(queue->list.next, struct sigqueue , list); + list_for_each_entry_safe(q, tmp, head, list) { list_del_init(&q->list); __sigqueue_free(q); } } +void flush_sigqueue(struct sigpending *queue) +{ + sigemptyset(&queue->signal); + flush_sigqueue_list(&queue->list); +} + +static void sigqueue_dequeue_pending(struct sigpending *queue, struct list_head *head) +{ + sigemptyset(&queue->signal); + list_splice_init(&queue->list, head); +} + /* * Flush all pending signals for this kthread. */ @@ -1019,6 +1043,21 @@ static inline bool legacy_queue(struct sigpending *signals, int sig) return (sig < SIGRTMIN) && sigismember(&signals->signal, sig); } +/* + * When PF_EXITING is set the task is on the way out and has t::pending + * flushed already. Prevent queueing of PIDTYPE_PID signals as they would + * be leaked. + */ +static inline bool task_can_queue_signal(struct task_struct *t, enum pid_type type) +{ + lockdep_assert_held(&t->sighand->siglock); + + if (!(t->flags & PF_EXITING)) + return true; + + return type != PIDTYPE_PID; +} + static int __send_signal_locked(int sig, struct kernel_siginfo *info, struct task_struct *t, enum pid_type type, bool force) { @@ -1030,6 +1069,10 @@ static int __send_signal_locked(int sig, struct kernel_siginfo *info, lockdep_assert_held(&t->sighand->siglock); result = TRACE_SIGNAL_IGNORED; + + if (!task_can_queue_signal(t, type)) + goto ret; + if (!prepare_signal(sig, t, force)) goto ret; @@ -1980,11 +2023,25 @@ static inline struct task_struct *posixtimer_get_target(struct k_itimer *tmr) struct task_struct *t = pid_task(tmr->it_pid, tmr->it_pid_type); if (t && tmr->it_pid_type != PIDTYPE_PID && - same_thread_group(t, current) && !current->exit_state) + same_thread_group(t, current) && !(current->flags & PF_EXITING)) t = current; return t; } +/* + * Find the target task for the POSIX timer signal and prevent that a + * PIDTYPE_PID signal is queued on a task which has PF_EXITING set. + */ +static inline struct task_struct *posixtimer_get_unignore_target(struct k_itimer *tmr) +{ + struct task_struct *t = posixtimer_get_target(tmr); + + if (t && task_can_queue_signal(t, tmr->it_pid_type)) + return t; + + return NULL; +} + void posixtimer_send_sigqueue(struct k_itimer *tmr) { struct sigqueue *q = &tmr->sigq; @@ -2002,6 +2059,9 @@ void posixtimer_send_sigqueue(struct k_itimer *tmr) if (!likely(lock_task_sighand(t, &flags))) return; + if (!task_can_queue_signal(t, tmr->it_pid_type)) + goto unlock; + /* * Update @tmr::sigqueue_seq for posix timer signals with sighand * locked to prevent a race against dequeue_signal(). @@ -2093,6 +2153,7 @@ void posixtimer_send_sigqueue(struct k_itimer *tmr) result = TRACE_SIGNAL_DELIVERED; out: trace_signal_generate(sig, &q->info, t, tmr->it_pid_type != PIDTYPE_PID, result); +unlock: unlock_task_sighand(t, &flags); } @@ -2148,7 +2209,7 @@ static void posixtimer_sig_unignore(struct task_struct *tsk, int sig) * has exited by now, drop the reference count. */ guard(rcu)(); - target = posixtimer_get_target(tmr); + target = posixtimer_get_unignore_target(tmr); if (target) posixtimer_queue_sigqueue(&tmr->sigq, target, tmr->it_pid_type); else @@ -3132,42 +3193,36 @@ static void retarget_shared_pending(struct task_struct *tsk, sigset_t *which) void exit_signals(struct task_struct *tsk) { + LIST_HEAD(sigq_list); int group_stop = 0; - sigset_t unblocked; /* * @tsk is about to have PF_EXITING set - lock out users which - * expect stable threadgroup. + * expect a stable threadgroup. */ cgroup_threadgroup_change_begin(tsk); - if (thread_group_empty(tsk) || (tsk->signal->flags & SIGNAL_GROUP_EXIT)) { + scoped_guard(spinlock_irq, &tsk->sighand->siglock) { tsk->flags |= PF_EXITING; - cgroup_threadgroup_change_end(tsk); - return; - } - spin_lock_irq(&tsk->sighand->siglock); - /* - * From now this task is not visible for group-wide signals, - * see wants_signal(), do_signal_stop(). - */ - tsk->flags |= PF_EXITING; + sigqueue_dequeue_pending(&tsk->pending, &sigq_list); - cgroup_threadgroup_change_end(tsk); + if (task_sigpending(tsk) && !thread_group_empty(tsk) && + !(tsk->signal->flags & SIGNAL_GROUP_EXIT)) { + sigset_t unblocked = tsk->blocked; - if (!task_sigpending(tsk)) - goto out; + signotset(&unblocked); + retarget_shared_pending(tsk, &unblocked); - unblocked = tsk->blocked; - signotset(&unblocked); - retarget_shared_pending(tsk, &unblocked); + if (unlikely(tsk->jobctl & JOBCTL_STOP_PENDING) && + task_participate_group_stop(tsk)) + group_stop = CLD_STOPPED; + } + } - if (unlikely(tsk->jobctl & JOBCTL_STOP_PENDING) && - task_participate_group_stop(tsk)) - group_stop = CLD_STOPPED; -out: - spin_unlock_irq(&tsk->sighand->siglock); + cgroup_threadgroup_change_end(tsk); + + flush_sigqueue_list(&sigq_list); /* * If group stop has completed, deliver the notification. This diff --git a/kernel/time/jiffies.c b/kernel/time/jiffies.c index 213ae1d6a014..80c354811538 100644 --- a/kernel/time/jiffies.c +++ b/kernel/time/jiffies.c @@ -136,6 +136,8 @@ static int sysctl_k2u_int_conv_userhz(bool *negp, ulong *u_ptr, const int *k_ptr static ulong sysctl_msecs_to_jiffies(const ulong val) { + if (val > jiffies_to_msecs(MAX_JIFFY_OFFSET)) + return MAX_JIFFY_OFFSET; return msecs_to_jiffies(val); } @@ -181,7 +183,7 @@ static int do_proc_int_conv_ms_jiffies_minmax(bool *negp, ulong *u_ptr, int *k_ptr, int dir, const struct ctl_table *tbl) { - return proc_int_conv(negp, u_ptr, k_ptr, dir, tbl, false, + return proc_int_conv(negp, u_ptr, k_ptr, dir, tbl, true, sysctl_u2k_int_conv_ms, sysctl_k2u_int_conv_ms); } @@ -195,10 +197,10 @@ static int sysctl_k2u_ulong_conv_ms(ulong *u_ptr, const ulong *k_ptr) return proc_ulong_k2u_conv_kop(u_ptr, k_ptr, sysctl_jiffies_to_msecs); } -static int do_proc_ulong_conv_ms_jiffies(bool *negp, ulong *u_ptr, ulong *k_ptr, - int dir, const struct ctl_table *tbl) +static int do_proc_ulong_conv_ms_jiffies_minmax(bool *negp, ulong *u_ptr, ulong *k_ptr, + int dir, const struct ctl_table *tbl) { - return proc_ulong_conv(u_ptr, k_ptr, dir, tbl, false, + return proc_ulong_conv(u_ptr, k_ptr, dir, tbl, true, sysctl_u2k_ulong_conv_ms, sysctl_k2u_ulong_conv_ms); } @@ -229,8 +231,8 @@ static int do_proc_int_conv_ms_jiffies_minmax(bool *negp, ulong *u_ptr, return -ENOSYS; } -static int do_proc_ulong_conv_ms_jiffies(bool *negp, ulong *u_ptr, ulong *k_ptr, - int dir, const struct ctl_table *tbl) +static int do_proc_ulong_conv_ms_jiffies_minmax(bool *negp, ulong *u_ptr, ulong *k_ptr, + int dir, const struct ctl_table *tbl) { return -ENOSYS; } @@ -333,7 +335,7 @@ int proc_doulongvec_ms_jiffies_minmax(const struct ctl_table *table, int dir, void *buffer, size_t *lenp, loff_t *ppos) { return proc_doulongvec_conv(table, dir, buffer, lenp, ppos, - do_proc_ulong_conv_ms_jiffies); + do_proc_ulong_conv_ms_jiffies_minmax); } EXPORT_SYMBOL(proc_doulongvec_ms_jiffies_minmax); diff --git a/kernel/time/posix-cpu-timers.c b/kernel/time/posix-cpu-timers.c index d73d31c7994f..0bf4fcd969c8 100644 --- a/kernel/time/posix-cpu-timers.c +++ b/kernel/time/posix-cpu-timers.c @@ -408,6 +408,7 @@ static int posix_cpu_timer_create(struct k_itimer *new_timer) new_timer->kclock = &clock_posix_cpu; timerqueue_init(&new_timer->it.cpu.node); + INIT_LIST_HEAD(&new_timer->it.cpu.elist); new_timer->it.cpu.pid = get_pid(pid); rcu_read_unlock(); return 0; @@ -566,6 +567,24 @@ static struct task_struct *timer_lock_sighand(struct k_itimer *timer, unsigned l } /* + * If the timer is queued on the expiry list, then it cannot be dequeued because + * the firing list is not protected by sighand->lock. The delivery path is + * waiting for the timer lock. So go back, unlock and retry. + */ +static bool posix_cpu_timer_on_expiry_list(struct k_itimer *timer) +{ + if (list_empty(&timer->it.cpu.elist)) + return false; + + /* + * Prevent signal delivery as there is no point in delivering a signal + * which is made obsolete right away. + */ + timer->it.cpu.firing = false; + return true; +} + +/* * Clean up a CPU-clock timer that is about to be destroyed. * This is called from timer deletion with the timer already locked. * If we return TIMER_RETRY, it's necessary to release the timer's lock @@ -580,18 +599,10 @@ static int posix_cpu_timer_del(struct k_itimer *timer) p = timer_lock_sighand(timer, &flags); if (likely(p)) { - if (timer->it.cpu.firing) { - /* - * Prevent signal delivery. The timer cannot be dequeued - * because it is on the firing list which is not protected - * by sighand->lock. The delivery path is waiting for - * the timer lock. So go back, unlock and retry. - */ - timer->it.cpu.firing = false; + if (posix_cpu_timer_on_expiry_list(timer)) ret = TIMER_RETRY; - } else { + else disarm_timer(timer, p); - } unlock_task_sighand(p, &flags); } @@ -731,14 +742,7 @@ static int posix_cpu_timer_set(struct k_itimer *timer, int timer_flags, /* Retrieve the current expiry time before disarming the timer */ old_expires = cpu_timer_getexpires(ctmr); - if (unlikely(timer->it.cpu.firing)) { - /* - * Prevent signal delivery. The timer cannot be dequeued - * because it is on the firing list which is not protected - * by sighand->lock. The delivery path is waiting for - * the timer lock. So go back, unlock and retry. - */ - timer->it.cpu.firing = false; + if (posix_cpu_timer_on_expiry_list(timer)) { ret = TIMER_RETRY; } else { cpu_timer_dequeue(ctmr); diff --git a/lib/alloc_tag.c b/lib/alloc_tag.c deleted file mode 100644 index e5b218176c5a..000000000000 --- a/lib/alloc_tag.c +++ /dev/null @@ -1,1029 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-only -#include <linux/alloc_tag.h> -#include <linux/execmem.h> -#include <linux/fs.h> -#include <linux/gfp.h> -#include <linux/kallsyms.h> -#include <linux/module.h> -#include <linux/page_ext.h> -#include <linux/pgalloc_tag.h> -#include <linux/proc_fs.h> -#include <linux/rcupdate.h> -#include <linux/seq_buf.h> -#include <linux/seq_file.h> -#include <linux/string_choices.h> -#include <linux/vmalloc.h> -#include <linux/kmemleak.h> - -#define ALLOCINFO_FILE_NAME "allocinfo" -#define MODULE_ALLOC_TAG_VMAP_SIZE (100000UL * sizeof(struct alloc_tag)) -#define SECTION_START(NAME) (CODETAG_SECTION_START_PREFIX NAME) -#define SECTION_STOP(NAME) (CODETAG_SECTION_STOP_PREFIX NAME) - -#ifdef CONFIG_MEM_ALLOC_PROFILING_ENABLED_BY_DEFAULT -static bool mem_profiling_support = true; -#else -static bool mem_profiling_support; -#endif - -/* - * Memory allocation profiling is permanently disabled and cannot be enabled. - * Must be called after setup_early_mem_profiling(). - */ -bool mem_alloc_profiling_permanently_disabled(void) -{ - return !mem_profiling_support; -} - -static struct codetag_type *alloc_tag_cttype; - -#ifdef CONFIG_ARCH_MODULE_NEEDS_WEAK_PER_CPU -DEFINE_PER_CPU(struct alloc_tag_counters, _shared_alloc_tag); -EXPORT_SYMBOL(_shared_alloc_tag); -#endif - -DEFINE_STATIC_KEY_MAYBE(CONFIG_MEM_ALLOC_PROFILING_ENABLED_BY_DEFAULT, - mem_alloc_profiling_key); -EXPORT_SYMBOL(mem_alloc_profiling_key); - -DEFINE_STATIC_KEY_FALSE(mem_profiling_compressed); - -struct alloc_tag_kernel_section kernel_tags = { NULL, 0 }; -unsigned long alloc_tag_ref_mask; -int alloc_tag_ref_offs; - -struct allocinfo_private { - struct codetag_iterator iter; - struct codetag_iterator reported_iter; - bool print_header; -}; - -static void *allocinfo_start(struct seq_file *m, loff_t *pos) -{ - struct allocinfo_private *priv; - loff_t node = *pos; - - priv = (struct allocinfo_private *)m->private; - codetag_lock_module_list(alloc_tag_cttype); - if (node == 0) { - priv->print_header = true; - priv->iter = codetag_get_ct_iter(alloc_tag_cttype); - } else { - priv->iter = priv->reported_iter; - } - codetag_next_ct(&priv->iter); - return priv->iter.ct ? priv : NULL; -} - -static void *allocinfo_next(struct seq_file *m, void *arg, loff_t *pos) -{ - struct allocinfo_private *priv = (struct allocinfo_private *)arg; - struct codetag *ct; - - priv->reported_iter = priv->iter; - ct = codetag_next_ct(&priv->iter); - (*pos)++; - if (!ct) - return NULL; - - return priv; -} - -static void allocinfo_stop(struct seq_file *m, void *arg) -{ - codetag_unlock_module_list(alloc_tag_cttype); -} - -static void print_allocinfo_header(struct seq_buf *buf) -{ - /* Output format version, so we can change it. */ - seq_buf_printf(buf, "allocinfo - version: 2.0\n"); - seq_buf_printf(buf, "# <size> <calls> <tag info>\n"); -} - -static void alloc_tag_to_text(struct seq_buf *out, struct codetag *ct) -{ - struct alloc_tag *tag = ct_to_alloc_tag(ct); - struct alloc_tag_counters counter = alloc_tag_read(tag); - s64 bytes = counter.bytes; - - seq_buf_printf(out, "%12lli %8llu ", bytes, counter.calls); - codetag_to_text(out, ct); - if (unlikely(alloc_tag_is_inaccurate(tag))) - seq_buf_printf(out, " accurate:no"); - seq_buf_putc(out, ' '); - seq_buf_putc(out, '\n'); -} - -static int allocinfo_show(struct seq_file *m, void *arg) -{ - struct allocinfo_private *priv = (struct allocinfo_private *)arg; - char *bufp; - size_t n = seq_get_buf(m, &bufp); - struct seq_buf buf; - - seq_buf_init(&buf, bufp, n); - if (priv->print_header) { - print_allocinfo_header(&buf); - priv->print_header = false; - } - alloc_tag_to_text(&buf, priv->iter.ct); - seq_commit(m, seq_buf_used(&buf)); - return 0; -} - -static const struct seq_operations allocinfo_seq_op = { - .start = allocinfo_start, - .next = allocinfo_next, - .stop = allocinfo_stop, - .show = allocinfo_show, -}; - -size_t alloc_tag_top_users(struct codetag_bytes *tags, size_t count, bool can_sleep) -{ - struct codetag_iterator iter; - struct codetag *ct; - struct codetag_bytes n; - unsigned int i, nr = 0; - - if (IS_ERR_OR_NULL(alloc_tag_cttype)) - return 0; - - if (can_sleep) - codetag_lock_module_list(alloc_tag_cttype); - else if (!codetag_trylock_module_list(alloc_tag_cttype)) - return 0; - - iter = codetag_get_ct_iter(alloc_tag_cttype); - while ((ct = codetag_next_ct(&iter))) { - struct alloc_tag_counters counter = alloc_tag_read(ct_to_alloc_tag(ct)); - - n.ct = ct; - n.bytes = counter.bytes; - - for (i = 0; i < nr; i++) - if (n.bytes > tags[i].bytes) - break; - - if (i < count) { - nr -= nr == count; - memmove(&tags[i + 1], - &tags[i], - sizeof(tags[0]) * (nr - i)); - nr++; - tags[i] = n; - } - } - - codetag_unlock_module_list(alloc_tag_cttype); - - return nr; -} - -void pgalloc_tag_split(struct folio *folio, int old_order, int new_order) -{ - int i; - struct alloc_tag *tag; - unsigned int nr_pages = 1 << new_order; - - if (!mem_alloc_profiling_enabled()) - return; - - tag = __pgalloc_tag_get(&folio->page); - if (!tag) - return; - - for (i = nr_pages; i < (1 << old_order); i += nr_pages) { - union pgtag_ref_handle handle; - union codetag_ref ref; - - if (get_page_tag_ref(folio_page(folio, i), &ref, &handle)) { - /* Set new reference to point to the original tag */ - alloc_tag_ref_set(&ref, tag); - update_page_tag_ref(handle, &ref); - put_page_tag_ref(handle); - } - } -} - -void pgalloc_tag_swap(struct folio *new, struct folio *old) -{ - union pgtag_ref_handle handle_old, handle_new; - union codetag_ref ref_old, ref_new; - struct alloc_tag *tag_old, *tag_new; - - if (!mem_alloc_profiling_enabled()) - return; - - tag_old = __pgalloc_tag_get(&old->page); - if (!tag_old) - return; - tag_new = __pgalloc_tag_get(&new->page); - if (!tag_new) - return; - - if (!get_page_tag_ref(&old->page, &ref_old, &handle_old)) - return; - if (!get_page_tag_ref(&new->page, &ref_new, &handle_new)) { - put_page_tag_ref(handle_old); - return; - } - - /* - * Clear tag references to avoid debug warning when using - * __alloc_tag_ref_set() with non-empty reference. - */ - set_codetag_empty(&ref_old); - set_codetag_empty(&ref_new); - - /* swap tags */ - __alloc_tag_ref_set(&ref_old, tag_new); - update_page_tag_ref(handle_old, &ref_old); - __alloc_tag_ref_set(&ref_new, tag_old); - update_page_tag_ref(handle_new, &ref_new); - - put_page_tag_ref(handle_old); - put_page_tag_ref(handle_new); -} - -static void shutdown_mem_profiling(bool remove_file) -{ - if (mem_alloc_profiling_enabled()) - static_branch_disable(&mem_alloc_profiling_key); - - if (!mem_profiling_support) - return; - - if (remove_file) - remove_proc_entry(ALLOCINFO_FILE_NAME, NULL); - mem_profiling_support = false; -} - -void __init alloc_tag_sec_init(void) -{ - struct alloc_tag *last_codetag; - - if (!mem_profiling_support) - return; - - if (!static_key_enabled(&mem_profiling_compressed)) - return; - - kernel_tags.first_tag = (struct alloc_tag *)kallsyms_lookup_name( - SECTION_START(ALLOC_TAG_SECTION_NAME)); - last_codetag = (struct alloc_tag *)kallsyms_lookup_name( - SECTION_STOP(ALLOC_TAG_SECTION_NAME)); - kernel_tags.count = last_codetag - kernel_tags.first_tag; - - /* Check if kernel tags fit into page flags */ - if (kernel_tags.count > (1UL << NR_UNUSED_PAGEFLAG_BITS)) { - shutdown_mem_profiling(false); /* allocinfo file does not exist yet */ - pr_err("%lu allocation tags cannot be references using %d available page flag bits. Memory allocation profiling is disabled!\n", - kernel_tags.count, NR_UNUSED_PAGEFLAG_BITS); - return; - } - - alloc_tag_ref_offs = (LRU_REFS_PGOFF - NR_UNUSED_PAGEFLAG_BITS); - alloc_tag_ref_mask = ((1UL << NR_UNUSED_PAGEFLAG_BITS) - 1); - pr_debug("Memory allocation profiling compression is using %d page flag bits!\n", - NR_UNUSED_PAGEFLAG_BITS); -} - -#ifdef CONFIG_MODULES - -static struct maple_tree mod_area_mt = MTREE_INIT(mod_area_mt, MT_FLAGS_ALLOC_RANGE); -static struct vm_struct *vm_module_tags; -/* A dummy object used to indicate an unloaded module */ -static struct module unloaded_mod; -/* A dummy object used to indicate a module prepended area */ -static struct module prepend_mod; - -struct alloc_tag_module_section module_tags; - -static inline unsigned long alloc_tag_align(unsigned long val) -{ - if (!static_key_enabled(&mem_profiling_compressed)) { - /* No alignment requirements when we are not indexing the tags */ - return val; - } - - if (val % sizeof(struct alloc_tag) == 0) - return val; - return ((val / sizeof(struct alloc_tag)) + 1) * sizeof(struct alloc_tag); -} - -static bool ensure_alignment(unsigned long align, unsigned int *prepend) -{ - if (!static_key_enabled(&mem_profiling_compressed)) { - /* No alignment requirements when we are not indexing the tags */ - return true; - } - - /* - * If alloc_tag size is not a multiple of required alignment, tag - * indexing does not work. - */ - if (!IS_ALIGNED(sizeof(struct alloc_tag), align)) - return false; - - /* Ensure prepend consumes multiple of alloc_tag-sized blocks */ - if (*prepend) - *prepend = alloc_tag_align(*prepend); - - return true; -} - -static inline bool tags_addressable(void) -{ - unsigned long tag_idx_count; - - if (!static_key_enabled(&mem_profiling_compressed)) - return true; /* with page_ext tags are always addressable */ - - tag_idx_count = CODETAG_ID_FIRST + kernel_tags.count + - module_tags.size / sizeof(struct alloc_tag); - - return tag_idx_count < (1UL << NR_UNUSED_PAGEFLAG_BITS); -} - -static bool needs_section_mem(struct module *mod, unsigned long size) -{ - if (!mem_profiling_support) - return false; - - return size >= sizeof(struct alloc_tag); -} - -static bool clean_unused_counters(struct alloc_tag *start_tag, - struct alloc_tag *end_tag) -{ - struct alloc_tag *tag; - bool ret = true; - - for (tag = start_tag; tag <= end_tag; tag++) { - struct alloc_tag_counters counter; - - if (!tag->counters) - continue; - - counter = alloc_tag_read(tag); - if (!counter.bytes) { - free_percpu(tag->counters); - tag->counters = NULL; - } else { - ret = false; - } - } - - return ret; -} - -/* Called with mod_area_mt locked */ -static void clean_unused_module_areas_locked(void) -{ - MA_STATE(mas, &mod_area_mt, 0, module_tags.size); - struct module *val; - - mas_for_each(&mas, val, module_tags.size) { - struct alloc_tag *start_tag; - struct alloc_tag *end_tag; - - if (val != &unloaded_mod) - continue; - - /* Release area if all tags are unused */ - start_tag = (struct alloc_tag *)(module_tags.start_addr + mas.index); - end_tag = (struct alloc_tag *)(module_tags.start_addr + mas.last); - if (clean_unused_counters(start_tag, end_tag)) - mas_erase(&mas); - } -} - -/* Called with mod_area_mt locked */ -static bool find_aligned_area(struct ma_state *mas, unsigned long section_size, - unsigned long size, unsigned int prepend, unsigned long align) -{ - bool cleanup_done = false; - -repeat: - /* Try finding exact size and hope the start is aligned */ - if (!mas_empty_area(mas, 0, section_size - 1, prepend + size)) { - if (IS_ALIGNED(mas->index + prepend, align)) - return true; - - /* Try finding larger area to align later */ - mas_reset(mas); - if (!mas_empty_area(mas, 0, section_size - 1, - size + prepend + align - 1)) - return true; - } - - /* No free area, try cleanup stale data and repeat the search once */ - if (!cleanup_done) { - clean_unused_module_areas_locked(); - cleanup_done = true; - mas_reset(mas); - goto repeat; - } - - return false; -} - -static int vm_module_tags_populate(void) -{ - unsigned long phys_end = ALIGN_DOWN(module_tags.start_addr, PAGE_SIZE) + - (vm_module_tags->nr_pages << PAGE_SHIFT); - unsigned long new_end = module_tags.start_addr + module_tags.size; - - if (phys_end < new_end) { - struct page **next_page = vm_module_tags->pages + vm_module_tags->nr_pages; - unsigned long old_shadow_end = ALIGN(phys_end, MODULE_ALIGN); - unsigned long new_shadow_end = ALIGN(new_end, MODULE_ALIGN); - unsigned long more_pages; - unsigned long nr = 0; - - more_pages = ALIGN(new_end - phys_end, PAGE_SIZE) >> PAGE_SHIFT; - while (nr < more_pages) { - unsigned long allocated; - - allocated = alloc_pages_bulk_node(GFP_KERNEL | __GFP_NOWARN, - NUMA_NO_NODE, more_pages - nr, next_page + nr); - - if (!allocated) - break; - nr += allocated; - } - - if (nr < more_pages || - vmap_pages_range(phys_end, phys_end + (nr << PAGE_SHIFT), PAGE_KERNEL, - next_page, PAGE_SHIFT) < 0) { - release_pages_arg arg = { .pages = next_page }; - - /* Clean up and error out */ - release_pages(arg, nr); - return -ENOMEM; - } - - vm_module_tags->nr_pages += nr; - - /* - * Kasan allocates 1 byte of shadow for every 8 bytes of data. - * When kasan_alloc_module_shadow allocates shadow memory, - * its unit of allocation is a page. - * Therefore, here we need to align to MODULE_ALIGN. - */ - if (old_shadow_end < new_shadow_end) - kasan_alloc_module_shadow((void *)old_shadow_end, - new_shadow_end - old_shadow_end, - GFP_KERNEL); - } - - /* - * Mark the pages as accessible, now that they are mapped. - * With hardware tag-based KASAN, marking is skipped for - * non-VM_ALLOC mappings, see __kasan_unpoison_vmalloc(). - */ - kasan_unpoison_vmalloc((void *)module_tags.start_addr, - new_end - module_tags.start_addr, - KASAN_VMALLOC_PROT_NORMAL); - - return 0; -} - -static void *reserve_module_tags(struct module *mod, unsigned long size, - unsigned int prepend, unsigned long align) -{ - unsigned long section_size = module_tags.end_addr - module_tags.start_addr; - MA_STATE(mas, &mod_area_mt, 0, section_size - 1); - unsigned long offset; - void *ret = NULL; - - /* If no tags return error */ - if (size < sizeof(struct alloc_tag)) - return ERR_PTR(-EINVAL); - - /* - * align is always power of 2, so we can use IS_ALIGNED and ALIGN. - * align 0 or 1 means no alignment, to simplify set to 1. - */ - if (!align) - align = 1; - - if (!ensure_alignment(align, &prepend)) { - shutdown_mem_profiling(true); - pr_err("%s: alignment %lu is incompatible with allocation tag indexing. Memory allocation profiling is disabled!\n", - mod->name, align); - return ERR_PTR(-EINVAL); - } - - mas_lock(&mas); - if (!find_aligned_area(&mas, section_size, size, prepend, align)) { - ret = ERR_PTR(-ENOMEM); - goto unlock; - } - - /* Mark found area as reserved */ - offset = mas.index; - offset += prepend; - offset = ALIGN(offset, align); - if (offset != mas.index) { - unsigned long pad_start = mas.index; - - mas.last = offset - 1; - mas_store(&mas, &prepend_mod); - if (mas_is_err(&mas)) { - ret = ERR_PTR(xa_err(mas.node)); - goto unlock; - } - mas.index = offset; - mas.last = offset + size - 1; - mas_store(&mas, mod); - if (mas_is_err(&mas)) { - mas.index = pad_start; - mas_erase(&mas); - ret = ERR_PTR(xa_err(mas.node)); - } - } else { - mas.last = offset + size - 1; - mas_store(&mas, mod); - if (mas_is_err(&mas)) - ret = ERR_PTR(xa_err(mas.node)); - } -unlock: - mas_unlock(&mas); - - if (IS_ERR(ret)) - return ret; - - if (module_tags.size < offset + size) { - int grow_res; - - module_tags.size = offset + size; - if (mem_alloc_profiling_enabled() && !tags_addressable()) { - shutdown_mem_profiling(true); - pr_warn("With module %s there are too many tags to fit in %d page flag bits. Memory allocation profiling is disabled!\n", - mod->name, NR_UNUSED_PAGEFLAG_BITS); - } - - grow_res = vm_module_tags_populate(); - if (grow_res) { - shutdown_mem_profiling(true); - pr_err("Failed to allocate memory for allocation tags in the module %s. Memory allocation profiling is disabled!\n", - mod->name); - return ERR_PTR(grow_res); - } - } - - return (struct alloc_tag *)(module_tags.start_addr + offset); -} - -static void release_module_tags(struct module *mod, bool used) -{ - MA_STATE(mas, &mod_area_mt, module_tags.size, module_tags.size); - struct alloc_tag *start_tag; - struct alloc_tag *end_tag; - struct module *val; - - mas_lock(&mas); - mas_for_each_rev(&mas, val, 0) - if (val == mod) - break; - - if (!val) /* module not found */ - goto out; - - if (!used) - goto release_area; - - start_tag = (struct alloc_tag *)(module_tags.start_addr + mas.index); - end_tag = (struct alloc_tag *)(module_tags.start_addr + mas.last); - if (!clean_unused_counters(start_tag, end_tag)) { - struct alloc_tag *tag; - - for (tag = start_tag; tag <= end_tag; tag++) { - struct alloc_tag_counters counter; - - if (!tag->counters) - continue; - - counter = alloc_tag_read(tag); - pr_info("%s:%u module %s func:%s has %llu allocated at module unload\n", - tag->ct.filename, tag->ct.lineno, tag->ct.modname, - tag->ct.function, counter.bytes); - } - } else { - used = false; - } -release_area: - mas_store(&mas, used ? &unloaded_mod : NULL); - val = mas_prev_range(&mas, 0); - if (val == &prepend_mod) - mas_store(&mas, NULL); -out: - mas_unlock(&mas); -} - -static int load_module(struct module *mod, struct codetag *start, struct codetag *stop) -{ - /* Allocate module alloc_tag percpu counters */ - struct alloc_tag *start_tag; - struct alloc_tag *stop_tag; - struct alloc_tag *tag; - - /* percpu counters for core allocations are already statically allocated */ - if (!mod) - return 0; - - start_tag = ct_to_alloc_tag(start); - stop_tag = ct_to_alloc_tag(stop); - for (tag = start_tag; tag < stop_tag; tag++) { - WARN_ON(tag->counters); - tag->counters = alloc_percpu(struct alloc_tag_counters); - if (!tag->counters) { - while (--tag >= start_tag) { - free_percpu(tag->counters); - tag->counters = NULL; - } - pr_err("Failed to allocate memory for allocation tag percpu counters in the module %s\n", - mod->name); - return -ENOMEM; - } - - /* - * Avoid a kmemleak false positive. The pointer to the counters is stored - * in the alloc_tag section of the module and cannot be directly accessed. - */ - kmemleak_ignore_percpu(tag->counters); - } - return 0; -} - -static void replace_module(struct module *mod, struct module *new_mod) -{ - MA_STATE(mas, &mod_area_mt, 0, module_tags.size); - struct module *val; - - mas_lock(&mas); - mas_for_each(&mas, val, module_tags.size) { - if (val != mod) - continue; - - mas_store_gfp(&mas, new_mod, GFP_KERNEL); - break; - } - mas_unlock(&mas); -} - -static int __init alloc_mod_tags_mem(void) -{ - /* Map space to copy allocation tags */ - vm_module_tags = execmem_vmap(MODULE_ALLOC_TAG_VMAP_SIZE); - if (!vm_module_tags) { - pr_err("Failed to map %lu bytes for module allocation tags\n", - MODULE_ALLOC_TAG_VMAP_SIZE); - module_tags.start_addr = 0; - return -ENOMEM; - } - - vm_module_tags->pages = kmalloc_objs(struct page *, - get_vm_area_size(vm_module_tags) >> PAGE_SHIFT, - GFP_KERNEL | __GFP_ZERO); - if (!vm_module_tags->pages) { - free_vm_area(vm_module_tags); - return -ENOMEM; - } - - module_tags.start_addr = (unsigned long)vm_module_tags->addr; - module_tags.end_addr = module_tags.start_addr + MODULE_ALLOC_TAG_VMAP_SIZE; - /* Ensure the base is alloc_tag aligned when required for indexing */ - module_tags.start_addr = alloc_tag_align(module_tags.start_addr); - - return 0; -} - -static void __init free_mod_tags_mem(void) -{ - release_pages_arg arg = { .pages = vm_module_tags->pages }; - - module_tags.start_addr = 0; - release_pages(arg, vm_module_tags->nr_pages); - kfree(vm_module_tags->pages); - free_vm_area(vm_module_tags); -} - -#else /* CONFIG_MODULES */ - -static inline int alloc_mod_tags_mem(void) { return 0; } -static inline void free_mod_tags_mem(void) {} - -#endif /* CONFIG_MODULES */ - -/* See: Documentation/mm/allocation-profiling.rst */ -static int __init setup_early_mem_profiling(char *str) -{ - bool compressed = false; - bool enable; - - if (!str || !str[0]) - return -EINVAL; - - if (!strncmp(str, "never", 5)) { - enable = false; - mem_profiling_support = false; - pr_info("Memory allocation profiling is disabled!\n"); - } else { - char *token = strsep(&str, ","); - - if (kstrtobool(token, &enable)) - return -EINVAL; - - if (str) { - - if (strcmp(str, "compressed")) - return -EINVAL; - - compressed = true; - } - mem_profiling_support = true; - pr_info("Memory allocation profiling is enabled %s compression and is turned %s!\n", - compressed ? "with" : "without", str_on_off(enable)); - } - - if (enable != mem_alloc_profiling_enabled()) { - if (enable) - static_branch_enable(&mem_alloc_profiling_key); - else - static_branch_disable(&mem_alloc_profiling_key); - } - if (compressed != static_key_enabled(&mem_profiling_compressed)) { - if (compressed) - static_branch_enable(&mem_profiling_compressed); - else - static_branch_disable(&mem_profiling_compressed); - } - - return 0; -} -early_param("sysctl.vm.mem_profiling", setup_early_mem_profiling); - -static __init bool need_page_alloc_tagging(void) -{ - if (static_key_enabled(&mem_profiling_compressed)) - return false; - - return mem_profiling_support; -} - -#ifdef CONFIG_MEM_ALLOC_PROFILING_DEBUG -/* - * Track page allocations before page_ext is initialized. - * Some pages are allocated before page_ext becomes available, leaving - * their codetag uninitialized. Track these early PFNs so we can clear - * their codetag refs later to avoid warnings when they are freed. - * - * Each page is cast to a pfn_pool: the first few bytes hold metadata - * (next pointer and slot count), the remainder stores PFNs. - */ -struct pfn_pool { - struct pfn_pool *next; - atomic_t count; - unsigned long pfns[]; -}; - -#define PFN_POOL_SIZE ((PAGE_SIZE - offsetof(struct pfn_pool, pfns)) / \ - sizeof(unsigned long)) - -/* - * Skip early PFN recording for a page allocation. Reuses the - * %__GFP_NO_OBJ_EXT bit. Used by __alloc_tag_add_early_pfn() to avoid - * recursion when allocating pages for the early PFN tracking list - * itself. - * - * Codetags of the pages allocated with __GFP_NO_CODETAG should be - * cleared (via clear_page_tag_ref()) before freeing the pages to prevent - * alloc_tag_sub_check() from triggering a warning. - */ -#define __GFP_NO_CODETAG __GFP_NO_OBJ_EXT - -static struct pfn_pool *current_pfn_pool __initdata; - -static void __init __alloc_tag_add_early_pfn(unsigned long pfn) -{ - struct pfn_pool *pool; - int idx; - - do { - pool = READ_ONCE(current_pfn_pool); - if (!pool || atomic_read(&pool->count) >= PFN_POOL_SIZE) { - struct page *new_page = alloc_page(__GFP_HIGH | __GFP_NO_CODETAG); - struct pfn_pool *new; - - if (!new_page) { - pr_warn_once("early PFN tracking page allocation failed\n"); - return; - } - new = page_address(new_page); - new->next = pool; - atomic_set(&new->count, 0); - if (cmpxchg(¤t_pfn_pool, pool, new) != pool) { - clear_page_tag_ref(new_page); - __free_page(new_page); - continue; - } - pool = new; - } - idx = atomic_read(&pool->count); - if (idx >= PFN_POOL_SIZE) - continue; - if (atomic_cmpxchg(&pool->count, idx, idx + 1) == idx) - break; - } while (1); - - pool->pfns[idx] = pfn; -} - -typedef void alloc_tag_add_func(unsigned long pfn); -static alloc_tag_add_func __rcu *alloc_tag_add_early_pfn_ptr __refdata = - RCU_INITIALIZER(__alloc_tag_add_early_pfn); - -void alloc_tag_add_early_pfn(unsigned long pfn, gfp_t gfp_flags) -{ - alloc_tag_add_func *alloc_tag_add; - - if (static_key_enabled(&mem_profiling_compressed)) - return; - - /* Skip allocations for the tracking list itself to avoid recursion. */ - if (gfp_flags & __GFP_NO_CODETAG) - return; - - rcu_read_lock(); - alloc_tag_add = rcu_dereference(alloc_tag_add_early_pfn_ptr); - if (alloc_tag_add) - alloc_tag_add(pfn); - rcu_read_unlock(); -} - -static void __init clear_early_alloc_pfn_tag_refs(void) -{ - struct pfn_pool *pool, *next; - struct page *page; - int i; - - if (static_key_enabled(&mem_profiling_compressed)) - return; - - rcu_assign_pointer(alloc_tag_add_early_pfn_ptr, NULL); - /* Make sure we are not racing with __alloc_tag_add_early_pfn() */ - synchronize_rcu(); - - for (pool = current_pfn_pool; pool; pool = next) { - int nr_pfns = atomic_read(&pool->count); - - for (i = 0; i < nr_pfns; i++) { - unsigned long pfn = pool->pfns[i]; - - if (pfn_valid(pfn)) { - union pgtag_ref_handle handle; - union codetag_ref ref; - - if (get_page_tag_ref(pfn_to_page(pfn), &ref, &handle)) { - /* - * An early-allocated page could be freed and reallocated - * after its page_ext is initialized but before we clear it. - * In that case, it already has a valid tag set. - * We should not overwrite that valid tag - * with CODETAG_EMPTY. - * - * Note: there is still a small race window between checking - * ref.ct and calling set_codetag_empty(). We accept this - * race as it's unlikely and the extra complexity of atomic - * cmpxchg is not worth it for this debug-only code path. - */ - if (ref.ct) { - put_page_tag_ref(handle); - continue; - } - - set_codetag_empty(&ref); - update_page_tag_ref(handle, &ref); - put_page_tag_ref(handle); - } - } - } - - next = pool->next; - page = virt_to_page(pool); - clear_page_tag_ref(page); - __free_page(page); - } -} -#else /* !CONFIG_MEM_ALLOC_PROFILING_DEBUG */ -static inline void __init clear_early_alloc_pfn_tag_refs(void) {} -#endif /* CONFIG_MEM_ALLOC_PROFILING_DEBUG */ - -static __init void init_page_alloc_tagging(void) -{ - clear_early_alloc_pfn_tag_refs(); -} - -struct page_ext_operations page_alloc_tagging_ops = { - .size = sizeof(union codetag_ref), - .need = need_page_alloc_tagging, - .init = init_page_alloc_tagging, -}; -EXPORT_SYMBOL(page_alloc_tagging_ops); - -#ifdef CONFIG_SYSCTL -/* - * Not using proc_do_static_key() directly to prevent enabling profiling - * after it was shut down. - */ -static int proc_mem_profiling_handler(const struct ctl_table *table, int write, - void *buffer, size_t *lenp, loff_t *ppos) -{ - if (write) { - /* - * Call from do_sysctl_args() which is a no-op since the same - * value was already set by setup_early_mem_profiling. - * Return success to avoid warnings from do_sysctl_args(). - */ - if (!current->mm) - return 0; - -#ifdef CONFIG_MEM_ALLOC_PROFILING_DEBUG - /* User can't toggle profiling while debugging */ - return -EACCES; -#endif - if (!mem_profiling_support) - return -EINVAL; - } - - return proc_do_static_key(table, write, buffer, lenp, ppos); -} - - -static const struct ctl_table memory_allocation_profiling_sysctls[] = { - { - .procname = "mem_profiling", - .data = &mem_alloc_profiling_key, - .mode = 0644, - .proc_handler = proc_mem_profiling_handler, - }, -}; - -static void __init sysctl_init(void) -{ - register_sysctl_init("vm", memory_allocation_profiling_sysctls); -} -#else /* CONFIG_SYSCTL */ -static inline void sysctl_init(void) {} -#endif /* CONFIG_SYSCTL */ - -static int __init alloc_tag_init(void) -{ - const struct codetag_type_desc desc = { - .section = ALLOC_TAG_SECTION_NAME, - .tag_size = sizeof(struct alloc_tag), -#ifdef CONFIG_MODULES - .needs_section_mem = needs_section_mem, - .alloc_section_mem = reserve_module_tags, - .free_section_mem = release_module_tags, - .module_load = load_module, - .module_replaced = replace_module, -#endif - }; - int res; - - sysctl_init(); - - if (!mem_profiling_support) { - pr_info("Memory allocation profiling is not supported!\n"); - return 0; - } - - if (!proc_create_seq_private(ALLOCINFO_FILE_NAME, 0400, NULL, &allocinfo_seq_op, - sizeof(struct allocinfo_private), NULL)) { - pr_err("Failed to create %s file\n", ALLOCINFO_FILE_NAME); - shutdown_mem_profiling(false); - return -ENOMEM; - } - - res = alloc_mod_tags_mem(); - if (res) { - pr_err("Failed to reserve address space for module tags, errno = %d\n", res); - shutdown_mem_profiling(true); - return res; - } - - alloc_tag_cttype = codetag_register_type(&desc); - if (IS_ERR(alloc_tag_cttype)) { - pr_err("Allocation tags registration failed, errno = %pe\n", alloc_tag_cttype); - free_mod_tags_mem(); - shutdown_mem_profiling(true); - return PTR_ERR(alloc_tag_cttype); - } - - return 0; -} -module_init(alloc_tag_init); diff --git a/mm/filemap.c b/mm/filemap.c index 6afec636881f..00fd89cf6f55 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -1616,7 +1616,7 @@ static void filemap_end_dropbehind(struct folio *folio) return; if (!folio_test_clear_dropbehind(folio)) return; - if (mapping) + if (mapping && !folio_mapped(folio)) folio_unmap_invalidate(mapping, folio, 0); } diff --git a/mm/folio.c b/mm/folio.c index c02dcea9c03c..50a6dbe55998 100644 --- a/mm/folio.c +++ b/mm/folio.c @@ -33,6 +33,7 @@ #include <linux/page_idle.h> #include <linux/local_lock.h> #include <linux/buffer_head.h> +#include <linux/kvm_types.h> #include "internal.h" #include "page_alloc.h" @@ -926,6 +927,7 @@ void lru_cache_drain_for_folio(const struct folio *folio, *drained = LRU_CACHE_DRAINED_ALL; } } +EXPORT_SYMBOL_FOR_KVM(lru_cache_drain_for_folio); atomic_t lru_disable_count = ATOMIC_INIT(0); diff --git a/mm/huge_memory.c b/mm/huge_memory.c index afbb5974bd22..1e5d68acf62a 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -92,7 +92,7 @@ unsigned long huge_anon_orders_madvise __read_mostly; unsigned long huge_anon_orders_inherit __read_mostly; static bool anon_orders_configured __initdata; -static inline bool file_thp_enabled(struct vm_area_struct *vma) +static inline bool file_thp_enabled(const struct vm_area_struct *vma) { struct inode *inode; @@ -118,6 +118,67 @@ static bool vma_is_special_huge(const struct vm_area_struct *vma) return vma_test_any(vma, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); } +static bool vma_file_bypass_thp_tuneables(const struct vm_area_struct *vma, + enum tva_type type) +{ + const bool has_huge_fault = vma->vm_ops->huge_fault; + + /* MADV_COLLAPSE ignores tuneables. */ + if (type == TVA_FORCED_COLLAPSE) + return true; + /* Huge PFN mappings are uncompactable so the policy doesn't apply. */ + if (vma_test(vma, VMA_PFNMAP_BIT) && has_huge_fault) + return true; + return false; +} + +static bool vma_file_allow_thp_tuneables(vm_flags_t vm_flags) +{ + /* THP=always? */ + if (hugepage_global_always()) + return true; + /* THP=madvise and marked MADV_HUGEPAGE? */ + if (hugepage_global_enabled() && (vm_flags & VM_HUGEPAGE)) + return true; + return false; +} + +static bool vma_file_check_thp_tuneables(const struct vm_area_struct *vma, + vm_flags_t vm_flags, enum tva_type type) +{ + return vma_file_bypass_thp_tuneables(vma, type) || + vma_file_allow_thp_tuneables(vm_flags); +} + +static bool vma_can_map_huge_file(const struct vm_area_struct *vma, + vm_flags_t vm_flags, enum tva_type type) +{ + const bool has_huge_fault = vma->vm_ops->huge_fault; + + /* + * Enforce THP collapse requirements as necessary. Anonymous vmas + * were already handled in thp_vma_allowable_orders(). + */ + if (!vma_file_check_thp_tuneables(vma, vm_flags, type)) + return false; + + switch (type) { + case TVA_PAGEFAULT: + /* + * Trust that ->huge_fault() handlers know what they are doing + * in fault path. + */ + return has_huge_fault; + case TVA_SMAPS: + if (has_huge_fault) + return true; + fallthrough; + default: + /* Only regular file is valid in collapse path. */ + return file_thp_enabled(vma); + } +} + unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, vm_flags_t vm_flags, enum tva_type type, @@ -190,27 +251,8 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, vma, vma_start_pgoff(vma), 0, forced_collapse); - if (!vma_is_anonymous(vma)) { - /* - * Enforce THP collapse requirements as necessary. Anonymous vmas - * were already handled in thp_vma_allowable_orders(). - */ - if (!forced_collapse && - (!hugepage_global_enabled() || (!(vm_flags & VM_HUGEPAGE) && - !hugepage_global_always()))) - return 0; - - /* - * Trust that ->huge_fault() handlers know what they are doing - * in fault path. - */ - if (((in_pf || smaps)) && vma->vm_ops->huge_fault) - return orders; - /* Only regular file is valid in collapse path */ - if (((!in_pf || smaps)) && file_thp_enabled(vma)) - return orders; - return 0; - } + if (!vma_is_anonymous(vma)) + return vma_can_map_huge_file(vma, vm_flags, type) ? orders : 0; if (vma_is_temporary_stack(vma)) return 0; diff --git a/mm/memblock.c b/mm/memblock.c index 9ce86349a29f..021db49eb7fc 100644 --- a/mm/memblock.c +++ b/mm/memblock.c @@ -2908,14 +2908,18 @@ static int memblock_debug_show(struct seq_file *m, void *private) else seq_printf(m, "%4c ", 'x'); if (reg->flags) { - for (j = 0; j < count; j++) { - if (reg->flags & (1U << j)) { - seq_printf(m, "%s\n", flagname[j]); - break; - } + unsigned int flags = reg->flags; + bool first = true; + + for (j = 0; flags; j++, flags >>= 1) { + if (!(flags & 1)) + continue; + if (!first) + seq_putc(m, '|'); + seq_puts(m, j < count ? flagname[j] : "UNKNOWN"); + first = false; } - if (j == count) - seq_printf(m, "%s\n", "UNKNOWN"); + seq_putc(m, '\n'); } else { seq_printf(m, "%s\n", "NONE"); } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1271d390b617..856a7d07586c 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -3158,7 +3158,7 @@ static int obj_cgroup_charge_pages(struct obj_cgroup *objcg, gfp_t gfp, memcg = get_mem_cgroup_from_objcg(objcg); - ret = try_charge_memcg(memcg, gfp, nr_pages); + ret = try_charge(memcg, gfp, nr_pages); if (ret) goto out; diff --git a/mm/mlock.c b/mm/mlock.c index efa6716e4dfb..39215a3eab1f 100644 --- a/mm/mlock.c +++ b/mm/mlock.c @@ -141,7 +141,7 @@ static struct lruvec *__munlock_folio(struct folio *folio, struct lruvec *lruvec munlock: if (folio_test_clear_mlocked(folio)) { - __zone_stat_mod_folio(folio, NR_MLOCK, -nr_pages); + zone_stat_mod_folio(folio, NR_MLOCK, -nr_pages); if (isolated || !folio_test_unevictable(folio)) __count_vm_events(UNEVICTABLE_PGMUNLOCKED, nr_pages); else diff --git a/mm/mremap.c b/mm/mremap.c index 2b4b523a86b8..7c368440fafe 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -1355,12 +1355,11 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, if (vma_is_anonymous(vma) && !vma->vm_file) vma_set_pgoff(vma, pgoff_unfaulted); } - - /* Because we won't unmap we don't need to touch locked_vm. */ } static unsigned long move_vma(struct vma_remap_struct *vrm) { + const bool is_dontunmap = vrm->flags & MREMAP_DONTUNMAP; struct mm_struct *mm = current->mm; struct vm_area_struct *new_vma; unsigned long hiwater_vm; @@ -1401,10 +1400,10 @@ static unsigned long move_vma(struct vma_remap_struct *vrm) */ hiwater_vm = mm->hiwater_vm; - vrm_stat_account(vrm, vrm->new_len); - if (unlikely(!err && (vrm->flags & MREMAP_DONTUNMAP))) + if (unlikely(is_dontunmap && !err)) dontunmap_complete(vrm, new_vma); - else + vrm_stat_account(vrm, vrm->new_len); + if (!is_dontunmap || err) unmap_source_vma(vrm); mm->hiwater_vm = hiwater_vm; diff --git a/mm/shrinker.c b/mm/shrinker.c index a70aab124a0e..7ec2a9704f6f 100644 --- a/mm/shrinker.c +++ b/mm/shrinker.c @@ -227,6 +227,8 @@ static int shrinker_memcg_alloc(struct shrinker *shrinker) { int id; + shrinker->id = -1; + if (mem_cgroup_disabled()) return -ENOSYS; if (mem_cgroup_kmem_disabled() && !(shrinker->flags & SHRINKER_NONSLAB)) diff --git a/mm/swapfile.c b/mm/swapfile.c index 53bf01d5f7f1..601979b97f95 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -156,7 +156,7 @@ static struct swap_info_struct *swap_entry_to_info(swp_entry_t entry) * This bit will be set if the device is not on the plist and not * usable, will be cleared if the device is on the plist. */ -#define SWAP_USAGE_OFFLIST_BIT (1UL << (BITS_PER_TYPE(atomic_t) - 2)) +#define SWAP_USAGE_OFFLIST_BIT (1UL << (BITS_PER_TYPE(atomic_long_t) - 2)) #define SWAP_USAGE_COUNTER_MASK (~SWAP_USAGE_OFFLIST_BIT) static long swap_usage_in_pages(struct swap_info_struct *si) { @@ -2859,10 +2859,12 @@ static unsigned long __mmap_region(struct file *file, unsigned long addr, map.check_ksm_early = can_set_ksm_flags_early(&map); error = __mmap_setup(&map, &desc, uf); - if (!error && have_mmap_prepare) - error = call_mmap_prepare(&map, &desc); if (error) goto abort_munmap; + if (have_mmap_prepare) + error = call_mmap_prepare(&map, &desc); + if (error) + goto unacct_error; if (map.check_ksm_early) update_ksm_flags(&map); diff --git a/net/atm/pppoatm.c b/net/atm/pppoatm.c index 6da52d12df68..5214786e61d1 100644 --- a/net/atm/pppoatm.c +++ b/net/atm/pppoatm.c @@ -292,10 +292,13 @@ static int pppoatm_send(struct ppp_channel *chan, struct sk_buff *skb) struct atm_vcc *vcc; int ret; + if (!pskb_may_pull(skb, 1)) { + kfree_skb(skb); + return DROP_PACKET; + } + ATM_SKB(skb)->vcc = pvcc->atmvcc; pr_debug("(skb=0x%p, vcc=0x%p)\n", skb, pvcc->atmvcc); - if (skb->data[0] == '\0' && (pvcc->flags & SC_COMP_PROT)) - (void) skb_pull(skb, 1); vcc = ATM_SKB(skb)->vcc; bh_lock_sock(sk_atm(vcc)); @@ -317,23 +320,13 @@ static int pppoatm_send(struct ppp_channel *chan, struct sk_buff *skb) switch (pvcc->encaps) { /* LLC encapsulation needed */ case e_llc: - if (skb_headroom(skb) < LLC_LEN) { - struct sk_buff *n; - n = skb_realloc_headroom(skb, LLC_LEN); - if (n != NULL && - !pppoatm_may_send(pvcc, n->truesize)) { - kfree_skb(n); - goto nospace; - } - consume_skb(skb); - skb = n; - if (skb == NULL) { - bh_unlock_sock(sk_atm(vcc)); - return DROP_PACKET; - } - } else if (!pppoatm_may_send(pvcc, skb->truesize)) + if (skb_cow_head(skb, LLC_LEN)) { + bh_unlock_sock(sk_atm(vcc)); + kfree_skb(skb); + return DROP_PACKET; + } + if (!pppoatm_may_send(pvcc, skb->truesize)) goto nospace; - memcpy(skb_push(skb, LLC_LEN), pppllc, LLC_LEN); break; case e_vc: if (!pppoatm_may_send(pvcc, skb->truesize)) @@ -346,6 +339,12 @@ static int pppoatm_send(struct ppp_channel *chan, struct sk_buff *skb) return 1; } + if (skb->data[0] == '\0' && (pvcc->flags & SC_COMP_PROT)) + skb_pull(skb, 1); + + if (pvcc->encaps == e_llc) + memcpy(skb_push(skb, LLC_LEN), pppllc, LLC_LEN); + atm_account_tx(vcc, skb); pr_debug("atm_skb(%p)->vcc(%p)->dev(%p)\n", skb, ATM_SKB(skb)->vcc, ATM_SKB(skb)->vcc->dev); @@ -355,13 +354,6 @@ static int pppoatm_send(struct ppp_channel *chan, struct sk_buff *skb) return ret; nospace: bh_unlock_sock(sk_atm(vcc)); - /* - * We don't have space to send this SKB now, but we might have - * already applied SC_COMP_PROT compression, so may need to undo - */ - if ((pvcc->flags & SC_COMP_PROT) && skb_headroom(skb) > 0 && - skb->data[-1] == '\0') - (void) skb_push(skb, 1); return 0; } diff --git a/net/bluetooth/coredump.c b/net/bluetooth/coredump.c index 5bee863bd6d2..71fc8dab4004 100644 --- a/net/bluetooth/coredump.c +++ b/net/bluetooth/coredump.c @@ -104,6 +104,22 @@ static void hci_devcd_free(struct hci_dev *hdev) hci_devcd_reset(hdev); } +void hci_devcd_shutdown(struct hci_dev *hdev) +{ + unsigned long flags; + + spin_lock_irqsave(&hdev->dump.dump_q.lock, flags); + hdev->dump.supported = false; + spin_unlock_irqrestore(&hdev->dump.dump_q.lock, flags); + + disable_work_sync(&hdev->dump.dump_rx); + disable_delayed_work_sync(&hdev->dump.dump_timeout); + + hci_dev_lock(hdev); + hci_devcd_free(hdev); + hci_dev_unlock(hdev); +} + /* Call with hci_dev_lock only. */ static int hci_devcd_alloc(struct hci_dev *hdev, u32 size) { @@ -442,7 +458,29 @@ EXPORT_SYMBOL(hci_devcd_register); static inline bool hci_devcd_enabled(struct hci_dev *hdev) { - return hdev->dump.supported; + return READ_ONCE(hdev->dump.supported); +} + +static int hci_devcd_queue(struct hci_dev *hdev, struct sk_buff *skb) +{ + unsigned long flags; + int err = 0; + + spin_lock_irqsave(&hdev->dump.dump_q.lock, flags); + if (!hdev->dump.supported) + err = -EOPNOTSUPP; + else + __skb_queue_tail(&hdev->dump.dump_q, skb); + spin_unlock_irqrestore(&hdev->dump.dump_q.lock, flags); + + if (err) { + kfree_skb(skb); + return err; + } + + queue_work(hdev->workqueue, &hdev->dump.dump_rx); + + return 0; } int hci_devcd_init(struct hci_dev *hdev, u32 dump_size) @@ -459,10 +497,7 @@ int hci_devcd_init(struct hci_dev *hdev, u32 dump_size) hci_dmp_cb(skb)->pkt_type = HCI_DEVCOREDUMP_PKT_INIT; put_unaligned_le32(dump_size, skb_put(skb, 4)); - skb_queue_tail(&hdev->dump.dump_q, skb); - queue_work(hdev->workqueue, &hdev->dump.dump_rx); - - return 0; + return hci_devcd_queue(hdev, skb); } EXPORT_SYMBOL(hci_devcd_init); @@ -478,10 +513,7 @@ int hci_devcd_append(struct hci_dev *hdev, struct sk_buff *skb) hci_dmp_cb(skb)->pkt_type = HCI_DEVCOREDUMP_PKT_SKB; - skb_queue_tail(&hdev->dump.dump_q, skb); - queue_work(hdev->workqueue, &hdev->dump.dump_rx); - - return 0; + return hci_devcd_queue(hdev, skb); } EXPORT_SYMBOL(hci_devcd_append); @@ -503,10 +535,7 @@ int hci_devcd_append_pattern(struct hci_dev *hdev, u8 pattern, u32 len) hci_dmp_cb(skb)->pkt_type = HCI_DEVCOREDUMP_PKT_PATTERN; skb_put_data(skb, &p, sizeof(p)); - skb_queue_tail(&hdev->dump.dump_q, skb); - queue_work(hdev->workqueue, &hdev->dump.dump_rx); - - return 0; + return hci_devcd_queue(hdev, skb); } EXPORT_SYMBOL(hci_devcd_append_pattern); @@ -523,10 +552,7 @@ int hci_devcd_complete(struct hci_dev *hdev) hci_dmp_cb(skb)->pkt_type = HCI_DEVCOREDUMP_PKT_COMPLETE; - skb_queue_tail(&hdev->dump.dump_q, skb); - queue_work(hdev->workqueue, &hdev->dump.dump_rx); - - return 0; + return hci_devcd_queue(hdev, skb); } EXPORT_SYMBOL(hci_devcd_complete); @@ -543,10 +569,7 @@ int hci_devcd_abort(struct hci_dev *hdev) hci_dmp_cb(skb)->pkt_type = HCI_DEVCOREDUMP_PKT_ABORT; - skb_queue_tail(&hdev->dump.dump_q, skb); - queue_work(hdev->workqueue, &hdev->dump.dump_rx); - - return 0; + return hci_devcd_queue(hdev, skb); } EXPORT_SYMBOL(hci_devcd_abort); diff --git a/net/bluetooth/eir.c b/net/bluetooth/eir.c index a55696820b22..ee0136bfae40 100644 --- a/net/bluetooth/eir.c +++ b/net/bluetooth/eir.c @@ -373,7 +373,15 @@ void *eir_get_service_data(u8 *eir, size_t eir_len, u16 uuid, size_t *len) size_t dlen; while ((eir = eir_get_data(eir, eir_len, EIR_SERVICE_DATA, &dlen))) { - u16 value = get_unaligned_le16(eir); + u16 value; + + if (dlen < sizeof(value)) { + eir += dlen; + eir_len = eir_end - eir; + continue; + } + + value = get_unaligned_le16(eir); if (uuid == value) { if (len) diff --git a/net/bluetooth/hci_codec.c b/net/bluetooth/hci_codec.c index 5bc5003c387c..7a7e813dcdda 100644 --- a/net/bluetooth/hci_codec.c +++ b/net/bluetooth/hci_codec.c @@ -145,11 +145,12 @@ void hci_read_supported_codecs(struct hci_dev *hdev) skb_pull(skb, sizeof(rp->status)); - std_codecs = (void *)skb->data; + std_codecs = skb_pull_data(skb, sizeof(*std_codecs)); + if (!std_codecs) + goto error; /* validate codecs length before accessing */ - if (skb->len < flex_array_size(std_codecs, codec, std_codecs->num) - + sizeof(std_codecs->num)) + if (skb->len < flex_array_size(std_codecs, codec, std_codecs->num)) goto error; /* enumerate codec capabilities of standard codecs */ @@ -161,15 +162,14 @@ void hci_read_supported_codecs(struct hci_dev *hdev) LOCAL_CODEC_ACL_MASK | LOCAL_CODEC_SCO_MASK, &caps); } - skb_pull(skb, flex_array_size(std_codecs, codec, std_codecs->num) - + sizeof(std_codecs->num)); + skb_pull(skb, flex_array_size(std_codecs, codec, std_codecs->num)); - vnd_codecs = (void *)skb->data; + vnd_codecs = skb_pull_data(skb, sizeof(*vnd_codecs)); + if (!vnd_codecs) + goto error; /* validate vendor codecs length before accessing */ - if (skb->len < - flex_array_size(vnd_codecs, codec, vnd_codecs->num) - + sizeof(vnd_codecs->num)) + if (skb->len < flex_array_size(vnd_codecs, codec, vnd_codecs->num)) goto error; /* enumerate vendor codec capabilities */ @@ -214,11 +214,12 @@ void hci_read_supported_codecs_v2(struct hci_dev *hdev) skb_pull(skb, sizeof(rp->status)); - std_codecs = (void *)skb->data; + std_codecs = skb_pull_data(skb, sizeof(*std_codecs)); + if (!std_codecs) + goto error; /* check for payload data length before accessing */ - if (skb->len < flex_array_size(std_codecs, codec, std_codecs->num) - + sizeof(std_codecs->num)) + if (skb->len < flex_array_size(std_codecs, codec, std_codecs->num)) goto error; memset(&caps, 0, sizeof(caps)); @@ -229,15 +230,14 @@ void hci_read_supported_codecs_v2(struct hci_dev *hdev) &caps); } - skb_pull(skb, flex_array_size(std_codecs, codec, std_codecs->num) - + sizeof(std_codecs->num)); + skb_pull(skb, flex_array_size(std_codecs, codec, std_codecs->num)); - vnd_codecs = (void *)skb->data; + vnd_codecs = skb_pull_data(skb, sizeof(*vnd_codecs)); + if (!vnd_codecs) + goto error; /* check for payload data length before accessing */ - if (skb->len < - flex_array_size(vnd_codecs, codec, vnd_codecs->num) - + sizeof(vnd_codecs->num)) + if (skb->len < flex_array_size(vnd_codecs, codec, vnd_codecs->num)) goto error; for (i = 0; i < vnd_codecs->num; i++) { diff --git a/net/bluetooth/hci_conn.c b/net/bluetooth/hci_conn.c index 8de98af2fb58..fa72cf8aaa7a 100644 --- a/net/bluetooth/hci_conn.c +++ b/net/bluetooth/hci_conn.c @@ -1023,6 +1023,19 @@ static struct hci_conn *__hci_conn_add(struct hci_dev *hdev, int type, if (!hdev->le_mtu && hdev->acl_mtu < HCI_MIN_LE_MTU) return ERR_PTR(-ECONNREFUSED); irk = hci_get_irk(hdev, dst, dst_type); + /* An identity address only reaches a peer advertising an RPA + * if the controller translates it. Unless address resolution + * is enabled and this peer is programmed into the resolving + * list, keep the RPA the peer is on air with; + * le_conn_complete_evt() resolves it back once the link is + * up. + */ + if (irk && + (!hci_dev_test_flag(hdev, HCI_LL_RPA_RESOLUTION) || + !hci_bdaddr_list_lookup_with_irk(&hdev->le_resolv_list, + &irk->bdaddr, + irk->addr_type))) + irk = NULL; break; case SCO_LINK: case ESCO_LINK: @@ -1505,7 +1518,15 @@ struct hci_conn *hci_connect_le(struct hci_dev *hdev, bdaddr_t *dst, } if (conn) { + /* dst may just have been swapped for the peer's RPA above, and + * dst_type describes dst -- it has to travel with it. Leaving + * the identity type behind makes the pair describe a peer that + * does not exist, and nothing downstream repairs it: + * hci_bdaddr_is_rpa() tests the type before the address, so + * the RPA is never treated as one. + */ bacpy(&conn->dst, dst); + conn->dst_type = dst_type; } else { conn = hci_conn_add_unset(hdev, LE_LINK, dst, dst_type, role); if (IS_ERR(conn)) diff --git a/net/bluetooth/hci_core.c b/net/bluetooth/hci_core.c index d7355c73f93e..d183efaf9063 100644 --- a/net/bluetooth/hci_core.c +++ b/net/bluetooth/hci_core.c @@ -2673,6 +2673,7 @@ void hci_unregister_dev(struct hci_dev *hdev) disable_work_sync(&hdev->error_reset); disable_delayed_work_sync(&hdev->cmd_timer); disable_delayed_work_sync(&hdev->ncmd_timer); + hci_devcd_shutdown(hdev); hci_cmd_sync_clear(hdev); @@ -3236,6 +3237,17 @@ static void hci_queue_acl(struct hci_chan *chan, struct sk_buff_head *queue, bt_dev_dbg(hdev, "chan %p queued %d", chan, skb_queue_len(queue)); } +/* Queue hdev->tx_work, unless hdev->workqueue is being drained by + * hci_dev_close_sync(), which would otherwise WARN and drop the work. + */ +static void hci_sched_tx(struct hci_dev *hdev) +{ + rcu_read_lock(); + if (!hci_dev_test_flag(hdev, HCI_CMD_DRAIN_WORKQUEUE)) + queue_work(hdev->workqueue, &hdev->tx_work); + rcu_read_unlock(); +} + void hci_send_acl(struct hci_chan *chan, struct sk_buff *skb, __u16 flags) { struct hci_dev *hdev = chan->conn->hdev; @@ -3244,7 +3256,7 @@ void hci_send_acl(struct hci_chan *chan, struct sk_buff *skb, __u16 flags) hci_queue_acl(chan, &chan->data_q, skb, flags); - queue_work(hdev->workqueue, &hdev->tx_work); + hci_sched_tx(hdev); } /* Send SCO data */ @@ -3269,7 +3281,7 @@ void hci_send_sco(struct hci_conn *conn, struct sk_buff *skb) bt_dev_dbg(hdev, "hcon %p queued %d", conn, skb_queue_len(&conn->data_q)); - queue_work(hdev->workqueue, &hdev->tx_work); + hci_sched_tx(hdev); } /* Send ISO data */ @@ -3340,7 +3352,7 @@ void hci_send_iso(struct hci_conn *conn, struct sk_buff *skb) hci_queue_iso(conn, &conn->data_q, skb); - queue_work(hdev->workqueue, &hdev->tx_work); + hci_sched_tx(hdev); } /* ---- HCI TX task (outgoing data) ---- */ diff --git a/net/bluetooth/hci_sync.c b/net/bluetooth/hci_sync.c index 2a651a4d60e6..74e2b04c84b2 100644 --- a/net/bluetooth/hci_sync.c +++ b/net/bluetooth/hci_sync.c @@ -5673,7 +5673,9 @@ int hci_dev_close_sync(struct hci_dev *hdev) memset(hdev->eir, 0, sizeof(hdev->eir)); memset(hdev->dev_class, 0, sizeof(hdev->dev_class)); bacpy(&hdev->random_addr, BDADDR_ANY); + hci_dev_lock(hdev); hci_codec_list_clear(&hdev->local_codecs); + hci_dev_unlock(hdev); hci_dev_put(hdev); return err; diff --git a/net/bluetooth/iso.c b/net/bluetooth/iso.c index 75bfd5938b2e..eb99653f33f9 100644 --- a/net/bluetooth/iso.c +++ b/net/bluetooth/iso.c @@ -819,19 +819,24 @@ static void iso_sock_destruct(struct sock *sk) skb_queue_purge(&sk->sk_error_queue); } -static void iso_sock_cleanup_listen(struct sock *parent) +/* Close not yet accepted channels */ +static void iso_sock_flush_accept_q(struct sock *parent) { struct sock *sk; - BT_DBG("parent %p", parent); - - /* Close not yet accepted channels */ while ((sk = bt_accept_dequeue(parent, NULL))) { iso_sock_close(sk); iso_sock_kill(sk); /* Drop the reference handed back by bt_accept_dequeue(). */ sock_put(sk); } +} + +static void iso_sock_cleanup_listen(struct sock *parent) +{ + BT_DBG("parent %p", parent); + + iso_sock_flush_accept_q(parent); /* If listening socket has a hcon, properly disconnect it */ if (iso_pi(parent)->conn && iso_pi(parent)->conn->hcon) { @@ -1737,6 +1742,13 @@ static int iso_sock_recvmsg(struct socket *sock, struct msghdr *msg, switch (sk->sk_state) { case BT_CONNECT2: if (test_bit(BT_SK_PA_SYNC, &pi->flags)) { + /* Move to BT_LISTEN before requesting the BIG + * sync: the BIS connections are matched to a + * parent socket in BT_LISTEN state, and they + * may be notified before the request returns. + */ + sk->sk_state = BT_LISTEN; + release_sock(sk); err = iso_conn_big_sync(sk); lock_sock(sk); @@ -1745,12 +1757,20 @@ static int iso_sock_recvmsg(struct socket *sock, struct msghdr *msg, * connection may have been torn down * meanwhile and iso_chan_del() may have * already moved the socket to BT_CLOSED. - * Only move on to BT_LISTEN if the BIG sync - * was actually started and nothing else has - * changed the state. + * Only move back if the BIG sync could not be + * started and nothing else has changed the + * state. */ - if (!err && sk->sk_state == BT_CONNECT2) - sk->sk_state = BT_LISTEN; + if (err && sk->sk_state == BT_LISTEN) { + /* Discard any child socket that may + * have been queued while the socket + * was in BT_LISTEN, as the cleanup of + * BT_CONNECT2 doesn't drain the + * accept queue. + */ + iso_sock_flush_accept_q(sk); + sk->sk_state = BT_CONNECT2; + } } else { iso_conn_defer_accept(pi->conn->hcon); sk->sk_state = BT_CONFIG; @@ -1760,12 +1780,22 @@ static int iso_sock_recvmsg(struct socket *sock, struct msghdr *msg, break; case BT_CONNECTED: if (test_bit(BT_SK_PA_SYNC, &iso_pi(sk)->flags)) { + /* As above, the BIS connections may be + * notified before the request returns. + */ + sk->sk_state = BT_LISTEN; + release_sock(sk); err = iso_conn_big_sync(sk); lock_sock(sk); - if (!err && sk->sk_state == BT_CONNECTED) - sk->sk_state = BT_LISTEN; + if (err && sk->sk_state == BT_LISTEN) { + /* As above, don't leave any child + * socket behind in the accept queue. + */ + iso_sock_flush_accept_q(sk); + sk->sk_state = BT_CONNECTED; + } early_ret = true; } @@ -2289,6 +2319,7 @@ static void iso_conn_ready(struct iso_conn *conn) BTPROTO_ISO, GFP_ATOMIC, 0); if (!sk) { release_sock(parent); + sock_put(parent); return; } diff --git a/net/bluetooth/rfcomm/sock.c b/net/bluetooth/rfcomm/sock.c index 958081adb9b5..e2486bc11cbc 100644 --- a/net/bluetooth/rfcomm/sock.c +++ b/net/bluetooth/rfcomm/sock.c @@ -242,9 +242,7 @@ static void __rfcomm_sock_close(struct sock *sk) */ static void rfcomm_sock_close(struct sock *sk) { - lock_sock(sk); __rfcomm_sock_close(sk); - release_sock(sk); } static void rfcomm_sock_init(struct sock *sk, struct sock *parent) @@ -905,6 +903,7 @@ static int rfcomm_sock_compat_ioctl(struct socket *sock, unsigned int cmd, unsig static int rfcomm_sock_shutdown(struct socket *sock, int how) { struct sock *sk = sock->sk; + bool cleanup_listen = false; int err = 0; BT_DBG("sock %p, sk %p", sock, sk); @@ -915,9 +914,17 @@ static int rfcomm_sock_shutdown(struct socket *sock, int how) lock_sock(sk); if (!sk->sk_shutdown) { sk->sk_shutdown = SHUTDOWN_MASK; + if (sk->sk_state == BT_LISTEN) { + /* Block new children before cleaning up without sk lock. */ + sk->sk_state = BT_CLOSED; + cleanup_listen = true; + } release_sock(sk); - __rfcomm_sock_close(sk); + if (cleanup_listen) + rfcomm_sock_cleanup_listen(sk); + else + __rfcomm_sock_close(sk); lock_sock(sk); if (sock_flag(sk, SOCK_LINGER) && sk->sk_lingertime && diff --git a/net/bridge/br_mst.c b/net/bridge/br_mst.c index 43a300ae6bfa..1654efd3045b 100644 --- a/net/bridge/br_mst.c +++ b/net/bridge/br_mst.c @@ -107,21 +107,24 @@ int br_mst_set_state(struct net_bridge_port *p, u16 msti, u8 state, struct net_bridge_vlan *v; int err = 0; - rcu_read_lock(); - vg = nbp_vlan_group_rcu(p); - if (!vg) - goto out; - /* MSTI 0 (CST) state changes are notified via the regular - * SWITCHDEV_ATTR_ID_PORT_STP_STATE. + * SWITCHDEV_ATTR_ID_PORT_STP_STATE. All other MSTIs are handled via + * netlink with RTNL held */ if (msti) { + ASSERT_RTNL(); + err = switchdev_port_attr_set(p->dev, &attr, extack); if (err && err != -EOPNOTSUPP) goto out; + err = 0; } - err = 0; + rcu_read_lock(); + vg = nbp_vlan_group_rcu(p); + if (!vg) + goto out_rcu_unlock; + list_for_each_entry_rcu(v, &vg->vlan_list, vlist) { if (v->brvlan->msti != msti) continue; @@ -129,8 +132,9 @@ int br_mst_set_state(struct net_bridge_port *p, u16 msti, u8 state, br_mst_vlan_set_state(vg, v, state); } -out: +out_rcu_unlock: rcu_read_unlock(); +out: return err; } diff --git a/net/bridge/br_vlan.c b/net/bridge/br_vlan.c index 1e0e436629ec..92b3cb621a26 100644 --- a/net/bridge/br_vlan.c +++ b/net/bridge/br_vlan.c @@ -387,12 +387,12 @@ out_filt: goto out; } -static int __vlan_del(struct net_bridge_vlan *v) +static void __vlan_del(struct net_bridge_vlan *v) { struct net_bridge_vlan *masterv = v; struct net_bridge_vlan_group *vg; struct net_bridge_port *p = NULL; - int err = 0; + int err; if (br_vlan_is_master(v)) { vg = br_vlan_group(v->br); @@ -406,12 +406,16 @@ static int __vlan_del(struct net_bridge_vlan *v) if (p) { err = __vlan_vid_del(p->dev, p->br, v); if (err) - goto out; + br_warn(p->br, + "port %u(%s) failed to delete vlan %u from switchdev: %pe\n", + (unsigned int)p->port_no, p->dev->name, + v->vid, ERR_PTR(err)); } else { err = br_switchdev_port_vlan_del(v->br->dev, v->vid); if (err && err != -EOPNOTSUPP) - goto out; - err = 0; + br_warn(v->br, + "failed to delete bridge vlan %u from switchdev: %pe\n", + v->vid, ERR_PTR(err)); } if (br_vlan_should_use(v)) { @@ -431,8 +435,6 @@ static int __vlan_del(struct net_bridge_vlan *v) } br_vlan_put_master(masterv); -out: - return err; } static void __vlan_group_free(struct net_bridge_vlan_group *vg) @@ -449,7 +451,6 @@ static void __vlan_flush(const struct net_bridge *br, { struct net_bridge_vlan *vlan, *tmp; u16 v_start = 0, v_end = 0; - int err; __vlan_delete_pvid(vg, vg->pvid); list_for_each_entry_safe(vlan, tmp, &vg->vlan_list, vlist) { @@ -463,13 +464,7 @@ static void __vlan_flush(const struct net_bridge *br, } v_end = vlan->vid; - err = __vlan_del(vlan); - if (err) { - br_err(br, - "port %u(%s) failed to delete vlan %d: %pe\n", - (unsigned int) p->port_no, p->dev->name, - vlan->vid, ERR_PTR(err)); - } + __vlan_del(vlan); } /* notify about the last/whole vlan range */ @@ -837,8 +832,9 @@ int br_vlan_delete(struct net_bridge *br, u16 vid) br_fdb_delete_by_port(br, NULL, vid, 0); vlan_tunnel_info_del(vg, v); + __vlan_del(v); - return __vlan_del(v); + return 0; } void br_vlan_flush(struct net_bridge *br) @@ -1368,8 +1364,9 @@ int nbp_vlan_delete(struct net_bridge_port *port, u16 vid) return -ENOENT; br_fdb_find_delete_local(port->br, port, port->dev->dev_addr, vid); br_fdb_delete_by_port(port->br, port, vid, 0); + __vlan_del(v); - return __vlan_del(v); + return 0; } void nbp_vlan_flush(struct net_bridge_port *port) diff --git a/net/core/dev.c b/net/core/dev.c index ecfbd72d5d1a..c67900354fa6 100644 --- a/net/core/dev.c +++ b/net/core/dev.c @@ -789,7 +789,7 @@ int dev_fill_forward_path(struct net_device_path_ctx *ctx, goto err_out; stack->num_paths++; - if (WARN_ON_ONCE(last_dev == ctx->dev)) + if (last_dev == ctx->dev) goto err_out; } diff --git a/net/core/drop_monitor.c b/net/core/drop_monitor.c index abaf108ac4db..edc660778408 100644 --- a/net/core/drop_monitor.c +++ b/net/core/drop_monitor.c @@ -141,9 +141,8 @@ static struct sk_buff *reset_per_cpu_data(struct per_cpu_dm_data *data) al = sizeof(struct net_dm_alert_msg); al += dm_hit_limit * sizeof(struct net_dm_drop_point); - al += sizeof(struct nlattr); - skb = genlmsg_new(al, GFP_KERNEL); + skb = genlmsg_new(nla_total_size(al), GFP_KERNEL); if (!skb) goto err; @@ -448,7 +447,7 @@ net_dm_hw_trap_summary_probe(void *ignore, const struct devlink *devlink, if (metadata->trap_type == DEVLINK_TRAP_TYPE_CONTROL) return; - hw_data = this_cpu_ptr(&dm_hw_cpu_data); + hw_data = raw_cpu_ptr(&dm_hw_cpu_data); raw_spin_lock_irqsave(&hw_data->lock, flags); hw_entries = hw_data->hw_entries; @@ -516,7 +515,7 @@ static void net_dm_packet_trace_kfree_skb_hit(void *ignore, */ nskb->tstamp = tstamp; - data = this_cpu_ptr(&dm_cpu_data); + data = raw_cpu_ptr(&dm_cpu_data); spin_lock_irqsave(&data->drop_queue.lock, flags); if (skb_queue_len(&data->drop_queue) < net_dm_queue_len) @@ -983,7 +982,7 @@ net_dm_hw_trap_packet_probe(void *ignore, const struct devlink *devlink, NET_DM_SKB_CB(nskb)->hw_metadata = n_hw_metadata; nskb->tstamp = tstamp; - hw_data = this_cpu_ptr(&dm_hw_cpu_data); + hw_data = raw_cpu_ptr(&dm_hw_cpu_data); spin_lock_irqsave(&hw_data->drop_queue.lock, flags); if (skb_queue_len(&hw_data->drop_queue) < net_dm_queue_len) @@ -1083,7 +1082,7 @@ err_module_put: struct per_cpu_dm_data *hw_data = &per_cpu(dm_hw_cpu_data, cpu); struct sk_buff *skb; - timer_delete_sync(&hw_data->send_timer); + timer_shutdown_sync(&hw_data->send_timer); cancel_work_sync(&hw_data->dm_alert_work); while ((skb = __skb_dequeue(&hw_data->drop_queue))) { struct devlink_trap_metadata *hw_metadata; @@ -1117,7 +1116,7 @@ static void net_dm_hw_monitor_stop(struct netlink_ext_ack *extack) struct per_cpu_dm_data *hw_data = &per_cpu(dm_hw_cpu_data, cpu); struct sk_buff *skb; - timer_delete_sync(&hw_data->send_timer); + timer_shutdown_sync(&hw_data->send_timer); cancel_work_sync(&hw_data->dm_alert_work); while ((skb = __skb_dequeue(&hw_data->drop_queue))) { struct devlink_trap_metadata *hw_metadata; @@ -1173,12 +1172,13 @@ static int net_dm_trace_on_set(struct netlink_ext_ack *extack) err_unregister_trace: unregister_trace_kfree_skb(ops->kfree_skb_probe, NULL); + tracepoint_synchronize_unregister(); err_module_put: for_each_possible_cpu(cpu) { struct per_cpu_dm_data *data = &per_cpu(dm_cpu_data, cpu); struct sk_buff *skb; - timer_delete_sync(&data->send_timer); + timer_shutdown_sync(&data->send_timer); cancel_work_sync(&data->dm_alert_work); while ((skb = __skb_dequeue(&data->drop_queue))) consume_skb(skb); @@ -1206,7 +1206,7 @@ static void net_dm_trace_off_set(void) struct per_cpu_dm_data *data = &per_cpu(dm_cpu_data, cpu); struct sk_buff *skb; - timer_delete_sync(&data->send_timer); + timer_shutdown_sync(&data->send_timer); cancel_work_sync(&data->dm_alert_work); while ((skb = __skb_dequeue(&data->drop_queue))) consume_skb(skb); diff --git a/net/core/neighbour.c b/net/core/neighbour.c index 1349c0eedb64..7448320f7ad5 100644 --- a/net/core/neighbour.c +++ b/net/core/neighbour.c @@ -2359,6 +2359,13 @@ static const struct nla_policy nl_neightbl_policy[NDTA_MAX+1] = { [NDTA_PARMS] = { .type = NLA_NESTED }, }; +#define NTBL_PARM_MS_MAX (24 * 60 * 60 * MSEC_PER_SEC) + +static const struct netlink_range_validation nl_ntbl_parm_ms_range = { + .min = 1, + .max = NTBL_PARM_MS_MAX, +}; + static const struct nla_policy nl_ntbl_parm_policy[NDTPA_MAX+1] = { [NDTPA_IFINDEX] = { .type = NLA_U32 }, [NDTPA_QUEUE_LEN] = { .type = NLA_U32 }, @@ -2375,7 +2382,8 @@ static const struct nla_policy nl_ntbl_parm_policy[NDTPA_MAX+1] = { [NDTPA_ANYCAST_DELAY] = { .type = NLA_U64 }, [NDTPA_PROXY_DELAY] = { .type = NLA_U64 }, [NDTPA_LOCKTIME] = { .type = NLA_U64 }, - [NDTPA_INTERVAL_PROBE_TIME_MS] = { .type = NLA_U64, .min = 1 }, + [NDTPA_INTERVAL_PROBE_TIME_MS] = NLA_POLICY_FULL_RANGE(NLA_U64, + &nl_ntbl_parm_ms_range), }; static int neightbl_set(struct sk_buff *skb, struct nlmsghdr *nlh, @@ -2579,9 +2587,10 @@ static int neightbl_dump_info(struct sk_buff *skb, struct netlink_callback *cb) { const struct nlmsghdr *nlh = cb->nlh; struct net *net = sock_net(skb->sk); + int default_skip = cb->args[2]; + int neigh_skip = cb->args[1]; int family, tidx, nidx = 0; int tbl_skip = cb->args[0]; - int neigh_skip = cb->args[1]; struct neigh_table *tbl; if (cb->strict_check) { @@ -2605,17 +2614,21 @@ static int neightbl_dump_info(struct sk_buff *skb, struct netlink_callback *cb) if (tidx < tbl_skip || (family && tbl->family != family)) continue; - if (neightbl_fill_info(skb, tbl, NETLINK_CB(cb->skb).portid, + if (!default_skip && + neightbl_fill_info(skb, tbl, NETLINK_CB(cb->skb).portid, nlh->nlmsg_seq, RTM_NEWNEIGHTBL, NLM_F_MULTI) < 0) break; - nidx = 0; - p = list_next_entry(&tbl->parms, list); - list_for_each_entry_from_rcu(p, &tbl->parms_list, list) { + default_skip = 1; + + list_for_each_entry_rcu(p, &tbl->parms_list, list) { if (!net_eq(neigh_parms_net(p), net)) continue; + if (!p->dev || p->dev == blackhole_netdev) + continue; + if (nidx < neigh_skip) goto next; @@ -2630,12 +2643,15 @@ static int neightbl_dump_info(struct sk_buff *skb, struct netlink_callback *cb) } neigh_skip = 0; + nidx = 0; + default_skip = 0; } out: rcu_read_unlock(); cb->args[0] = tidx; cb->args[1] = nidx; + cb->args[2] = default_skip; return skb->len; } @@ -3669,12 +3685,13 @@ static int neigh_proc_dointvec_ms_jiffies_positive(const struct ctl_table *ctl, void *buffer, size_t *lenp, loff_t *ppos) { struct ctl_table tmp = *ctl; - int ret; + int ret, min, max; - int min = msecs_to_jiffies(1); + min = msecs_to_jiffies(1); + max = msecs_to_jiffies(NTBL_PARM_MS_MAX); tmp.extra1 = &min; - tmp.extra2 = NULL; + tmp.extra2 = &max; ret = proc_dointvec_ms_jiffies_minmax(&tmp, write, buffer, lenp, ppos); neigh_proc_update(ctl, write); diff --git a/net/core/skbuff.c b/net/core/skbuff.c index cc3b4b70288b..609f2c7f4a47 100644 --- a/net/core/skbuff.c +++ b/net/core/skbuff.c @@ -6832,6 +6832,34 @@ failure: } EXPORT_SYMBOL(alloc_skb_with_frags); +/* pskb_carve_inside_header() and pskb_carve_inside_nonlinear() + * remove the first bytes of a packet and reallocate skb->head. + * + * Whatever headers were present before the operation are gone, + * we must not leave stale offsets, otherwise users of this skb + * (skb_dump(), drop_monitor, taps, ...) would read or pull garbage. + */ +static void skb_carve_reset_headers(struct sk_buff *skb) +{ + skb_unset_mac_header(skb); + skb_unset_transport_header(skb); + skb_reset_network_header(skb); + skb->mac_len = 0; + + /* Inner offsets have no "unset" marker, zero them so that + * skb_inner_network_header_was_set() becomes false and no + * consumer mistakes them for a real (and long gone) header. + */ + skb->inner_mac_header = 0; + skb->inner_network_header = 0; + skb->inner_transport_header = 0; + skb->inner_protocol = 0; + skb->encapsulation = 0; + + if (skb->ip_summed == CHECKSUM_PARTIAL) + skb->ip_summed = CHECKSUM_NONE; +} + /* carve out the first off bytes from skb when off < headlen */ static int pskb_carve_inside_header(struct sk_buff *skb, const u32 off, const int headlen, gfp_t gfp_mask) @@ -6887,7 +6915,7 @@ static int pskb_carve_inside_header(struct sk_buff *skb, const u32 off, skb->head_frag = 0; skb_set_end_offset(skb, size); skb_set_tail_pointer(skb, skb_headlen(skb)); - skb_headers_offset_update(skb, 0); + skb_carve_reset_headers(skb); skb->cloned = 0; skb->hdr_len = 0; skb->nohdr = 0; @@ -7027,7 +7055,7 @@ static int pskb_carve_inside_nonlinear(struct sk_buff *skb, const u32 off, skb->data = data; skb_set_end_offset(skb, size); skb_reset_tail_pointer(skb); - skb_headers_offset_update(skb, 0); + skb_carve_reset_headers(skb); skb->cloned = 0; skb->hdr_len = 0; skb->nohdr = 0; diff --git a/net/core/sock.c b/net/core/sock.c index fa60b7494c58..d23333bb4f3f 100644 --- a/net/core/sock.c +++ b/net/core/sock.c @@ -142,6 +142,7 @@ #include <trace/events/sock.h> +#include <net/psp.h> #include <net/tcp.h> #include <net/busy_poll.h> #include <net/phonet/phonet.h> @@ -2670,6 +2671,12 @@ void sk_setup_caps(struct sock *sk, struct dst_entry *dst) } EXPORT_SYMBOL_GPL(sk_setup_caps); +bool sk_has_decrypt_user(const struct sock *sk) +{ + return psp_sk_assoc(sk) || + (sk_is_inet(sk) && inet_csk_has_ulp(sk)); /* for tls */ +} + /* * Simple resource managers for sockets. */ @@ -3911,7 +3918,14 @@ int sock_gettstamp(struct socket *sock, void __user *userstamp, struct sock *sk = sock->sk; struct timespec64 ts; - sock_enable_timestamp(sk, SOCK_TIMESTAMP); + /* sk->sk_flags must only be changed under the socket lock, + * because sock_set_flag() uses non atomic operations. + */ + if (!sock_flag(sk, SOCK_TIMESTAMP)) { + lock_sock(sk); + sock_enable_timestamp(sk, SOCK_TIMESTAMP); + release_sock(sk); + } ts = ktime_to_timespec64(sock_read_timestamp(sk)); if (ts.tv_sec == -1) return -ENOENT; diff --git a/net/ipv4/esp4.c b/net/ipv4/esp4.c index a6c18aea7498..e76db5817e78 100644 --- a/net/ipv4/esp4.c +++ b/net/ipv4/esp4.c @@ -441,6 +441,12 @@ int esp_output_head(struct xfrm_state *x, struct sk_buff *skb, struct esp_info * esp->inplace = false; + /* Take real page refs and clear SKBFL_MANAGED_FRAG_REFS before + * we mutate the frag array, so the per-frag unref stays balanced + * for zerocopy managed frags (see __ip_append_data()). + */ + skb_zcopy_downgrade_managed(skb); + allocsize = ALIGN(tailen, L1_CACHE_BYTES); spin_lock_bh(&x->lock); diff --git a/net/ipv4/icmp.c b/net/ipv4/icmp.c index 0caedfc7ca92..90c0e22c29be 100644 --- a/net/ipv4/icmp.c +++ b/net/ipv4/icmp.c @@ -581,16 +581,19 @@ static struct rtable *icmp_route_lookup(struct net *net, struct flowi4 *fl4, skb_dstref_restore(skb_in, orefdst); /* - * At this point, fl4_dec.daddr should NOT be local (we - * checked fl4_dec.saddr above). However, a race condition - * may occur if the address is added to the interface - * concurrently. In that case, ip_route_input() returns a - * LOCAL route with dst.output=ip_rt_bug, which must not - * be used for output. + * fl4_dec.daddr is not expected to be local here, but it can be + * added to an interface concurrently, in which case + * ip_route_input() returns a LOCAL route. It can also fail to + * build a forwarding route towards fl4_dec.daddr, for example, + * when forwarding is disabled, and return an UNREACHABLE route. + * Both cases will result in a route with dst.output=ip_rt_bug, + * which must not be used for output. */ - if (!err && rt2 && rt2->rt_type == RTN_LOCAL) { + if (!err && rt2 && rt2->rt_type == RTN_LOCAL) net_warn_ratelimited("detected local route for %pI4 during ICMP sending, src %pI4\n", &fl4_dec.daddr, &fl4_dec.saddr); + if (!err && rt2 && + (rt2->rt_type == RTN_LOCAL || rt2->rt_type == RTN_UNREACHABLE)) { dst_release(&rt2->dst); err = -EINVAL; } diff --git a/net/ipv4/ip_tunnel_core.c b/net/ipv4/ip_tunnel_core.c index 5168d546ea2f..bab42b9e277f 100644 --- a/net/ipv4/ip_tunnel_core.c +++ b/net/ipv4/ip_tunnel_core.c @@ -680,8 +680,14 @@ static int ip_tun_get_optlen(struct nlattr *attr, } static int ip_tun_set_opts(struct nlattr *attr, struct ip_tunnel_info *info, - struct netlink_ext_ack *extack) + int opts_len, struct netlink_ext_ack *extack) { + /* `options_len` is the __counted_by() annotation of the `options` + * flexible array, it must be initialized before parsing writes + * into it. + */ + info->options_len = opts_len; + return ip_tun_parse_opts(attr, info, extack); } @@ -712,7 +718,8 @@ static int ip_tun_build_state(struct net *net, struct nlattr *attr, tun_info = lwt_tun_info(new_state); - err = ip_tun_set_opts(tb[LWTUNNEL_IP_OPTS], tun_info, extack); + err = ip_tun_set_opts(tb[LWTUNNEL_IP_OPTS], tun_info, opt_len, + extack); if (err < 0) { lwtstate_free(new_state); return err; @@ -753,7 +760,6 @@ static int ip_tun_build_state(struct net *net, struct nlattr *attr, } tun_info->mode = IP_TUNNEL_INFO_TX; - tun_info->options_len = opt_len; *ts = new_state; @@ -1006,7 +1012,8 @@ static int ip6_tun_build_state(struct net *net, struct nlattr *attr, tun_info = lwt_tun_info(new_state); - err = ip_tun_set_opts(tb[LWTUNNEL_IP6_OPTS], tun_info, extack); + err = ip_tun_set_opts(tb[LWTUNNEL_IP6_OPTS], tun_info, opt_len, + extack); if (err < 0) { lwtstate_free(new_state); return err; @@ -1040,7 +1047,6 @@ static int ip6_tun_build_state(struct net *net, struct nlattr *attr, } tun_info->mode = IP_TUNNEL_INFO_TX | IP_TUNNEL_INFO_IPV6; - tun_info->options_len = opt_len; *ts = new_state; diff --git a/net/ipv4/sysctl_net_ipv4.c b/net/ipv4/sysctl_net_ipv4.c index 2f0363bca2a8..e3760daa3470 100644 --- a/net/ipv4/sysctl_net_ipv4.c +++ b/net/ipv4/sysctl_net_ipv4.c @@ -51,6 +51,8 @@ static int tcp_ecn_mode_max = 5; static u32 icmp_errors_extension_mask_all = GENMASK_U8(ICMP_ERR_EXT_COUNT - 1, 0); +static int tcp_min_rcvbuf = 4096; + /* obsolete */ static int sysctl_tcp_low_latency __read_mostly; @@ -1462,7 +1464,7 @@ static const struct ctl_table ipv4_net_table[] = { .maxlen = sizeof(init_net.ipv4.sysctl_tcp_rmem), .mode = 0644, .proc_handler = proc_dointvec_minmax, - .extra1 = SYSCTL_ONE, + .extra1 = &tcp_min_rcvbuf, }, { .procname = "tcp_comp_sack_delay_ns", diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c index 0f60a1dbf927..92bc60716f33 100644 --- a/net/ipv4/tcp_input.c +++ b/net/ipv4/tcp_input.c @@ -6490,6 +6490,7 @@ reset: * or pure receivers (this means either the sequence number or the ack * value must stay constant) * - Unexpected TCP option. + * - ACK sequence number is outside [SND.UNA, SND.NXT]. * * When these conditions are not satisfied it drops into a standard * receive procedure patterned after RFC793 to handle all cases. @@ -6539,7 +6540,7 @@ void tcp_rcv_established(struct sock *sk, struct sk_buff *skb) if ((tcp_flag_word(th) & TCP_HP_BITS) == tp->pred_flags && TCP_SKB_CB(skb)->seq == tp->rcv_nxt && - !after(TCP_SKB_CB(skb)->ack_seq, tp->snd_nxt)) { + between(TCP_SKB_CB(skb)->ack_seq, tp->snd_una, tp->snd_nxt)) { int tcp_header_len = tp->tcp_header_len; s32 delta = 0; int flag = 0; diff --git a/net/ipv4/tcp_ulp.c b/net/ipv4/tcp_ulp.c index 2aa442128630..b58045df101e 100644 --- a/net/ipv4/tcp_ulp.c +++ b/net/ipv4/tcp_ulp.c @@ -136,6 +136,10 @@ static int __tcp_set_ulp(struct sock *sk, const struct tcp_ulp_ops *ulp_ops) if (icsk->icsk_ulp_ops) goto out_err; + err = -EINVAL; + if (sk_has_decrypt_user(sk)) + goto out_err; + if (sk->sk_socket) clear_bit(SOCK_SUPPORT_ZC, &sk->sk_socket->flags); diff --git a/net/ipv6/esp6.c b/net/ipv6/esp6.c index 72ec0d7d1120..b1c9b36f76dc 100644 --- a/net/ipv6/esp6.c +++ b/net/ipv6/esp6.c @@ -471,6 +471,12 @@ int esp6_output_head(struct xfrm_state *x, struct sk_buff *skb, struct esp_info esp->inplace = false; + /* Take real page refs and clear SKBFL_MANAGED_FRAG_REFS before + * we mutate the frag array, so the per-frag unref stays balanced + * for zerocopy managed frags (see __ip_append_data()). + */ + skb_zcopy_downgrade_managed(skb); + allocsize = ALIGN(tailen, L1_CACHE_BYTES); spin_lock_bh(&x->lock); diff --git a/net/ipv6/seg6_local.c b/net/ipv6/seg6_local.c index 7b5212220185..d1070aec7b72 100644 --- a/net/ipv6/seg6_local.c +++ b/net/ipv6/seg6_local.c @@ -257,10 +257,13 @@ static bool decap_and_validate(struct sk_buff *skb, int proto) return false; if (proto == IPPROTO_IPIP) { + bool l3slave = ipv6_l3mdev_skb(IP6CB(skb)->flags); int iif = IP6CB(skb)->iif; memset(IPCB(skb), 0, sizeof(*IPCB(skb))); IPCB(skb)->iif = iif; + if (l3slave) + IPCB(skb)->flags |= IPSKB_L3SLAVE; } else if (proto == IPPROTO_IPV6) { bool l3slave = ipv6_l3mdev_skb(IP6CB(skb)->flags); int iif = IP6CB(skb)->iif; diff --git a/net/ipv6/tcp_ipv6.c b/net/ipv6/tcp_ipv6.c index df9c29eb5c1f..7fa4ed2fd4f1 100644 --- a/net/ipv6/tcp_ipv6.c +++ b/net/ipv6/tcp_ipv6.c @@ -1604,7 +1604,8 @@ int tcp_v6_do_rcv(struct sock *sk, struct sk_buff *skb) by tcp. Feel free to propose better solution. --ANK (980728) */ - if (np->rxopt.all && sk->sk_state != TCP_LISTEN) + if (np->rxopt.all && + !((1 << sk->sk_state) & (TCPF_LISTEN | TCPF_CLOSE))) opt_skb = skb_clone_and_charge_r(skb, sk); if (sk->sk_state == TCP_ESTABLISHED) { /* Fast path */ diff --git a/net/ipv6/xfrm6_output.c b/net/ipv6/xfrm6_output.c index 512bdaf13699..44b221a09a0c 100644 --- a/net/ipv6/xfrm6_output.c +++ b/net/ipv6/xfrm6_output.c @@ -19,7 +19,10 @@ void xfrm6_local_rxpmtu(struct sk_buff *skb, u32 mtu) { struct flowi6 fl6; - struct sock *sk = skb->sk; + struct sock *sk = skb_to_full_sk(skb); + + if (!sk) + return; fl6.flowi6_oif = sk->sk_bound_dev_if; fl6.daddr = ipv6_hdr(skb)->daddr; @@ -31,7 +34,10 @@ void xfrm6_local_error(struct sk_buff *skb, u32 mtu) { struct flowi6 fl6; const struct ipv6hdr *hdr; - struct sock *sk = skb->sk; + struct sock *sk = skb_to_full_sk(skb); + + if (!sk) + return; hdr = skb->encapsulation ? inner_ipv6_hdr(skb) : ipv6_hdr(skb); fl6.fl6_dport = inet_sk(sk)->inet_dport; diff --git a/net/mac80211/cfg.c b/net/mac80211/cfg.c index 23f4f9ec86d0..a1753335eb9d 100644 --- a/net/mac80211/cfg.c +++ b/net/mac80211/cfg.c @@ -115,6 +115,10 @@ static int ieee80211_set_mon_options(struct ieee80211_sub_if_data *sdata, return -EBUSY; } + /* TXQs are reserved in ieee80211_if_add() and cannot be added later */ + if ((params->flags & MONITOR_FLAG_ACTIVE) && !sdata->vif.txq) + return -EOPNOTSUPP; + /* validate whether MU-MIMO can be configured */ if (!ieee80211_hw_check(&local->hw, WANT_MONITOR_VIF) && !ieee80211_hw_check(&local->hw, NO_VIRTUAL_MONITOR) && @@ -1929,6 +1933,9 @@ static int ieee80211_start_ap(struct wiphy *wiphy, struct net_device *dev, return 0; error: + link_conf->enable_beacon = false; + link_conf->beacon_int = prev_beacon_int; + sdata->vif.cfg.ssid_len = 0; ieee80211_link_release_channel(link); return err; @@ -3320,7 +3327,11 @@ static int ieee80211_join_mesh(struct wiphy *wiphy, struct net_device *dev, if (err) return err; - return ieee80211_start_mesh(sdata); + err = ieee80211_start_mesh(sdata); + if (err) + ieee80211_link_release_channel(&sdata->deflink); + + return err; } static int ieee80211_leave_mesh(struct wiphy *wiphy, struct net_device *dev) @@ -3475,7 +3486,7 @@ static int ieee80211_set_txq_params(struct wiphy *wiphy, static int ieee80211_suspend(struct wiphy *wiphy, struct cfg80211_wowlan *wowlan) { - return __ieee80211_suspend(wiphy_priv(wiphy), wowlan); + return __ieee80211_suspend(wiphy_priv(wiphy), wowlan, false); } static int ieee80211_resume(struct wiphy *wiphy) @@ -4113,6 +4124,9 @@ static int ieee80211_set_bitrate_mask(struct wiphy *wiphy, if (!ieee80211_sdata_running(sdata)) return -ENETDOWN; + if (!(sdata->flags & IEEE80211_SDATA_IN_DRIVER)) + return -ENETDOWN; + /* * If active validate the setting and reject it if it doesn't leave * at least one basic rate usable, since we really have to be able diff --git a/net/mac80211/debugfs.c b/net/mac80211/debugfs.c index 105653a16b68..e38631d7cfb4 100644 --- a/net/mac80211/debugfs.c +++ b/net/mac80211/debugfs.c @@ -384,7 +384,7 @@ static ssize_t reset_write(struct file *file, const char __user *user_buf, rtnl_lock(); wiphy_lock(local->hw.wiphy); - __ieee80211_suspend(&local->hw, NULL); + __ieee80211_suspend(&local->hw, NULL, true); ret = __ieee80211_resume(&local->hw); wiphy_unlock(local->hw.wiphy); diff --git a/net/mac80211/debugfs_netdev.c b/net/mac80211/debugfs_netdev.c index f3c6a41e4911..6aba22493670 100644 --- a/net/mac80211/debugfs_netdev.c +++ b/net/mac80211/debugfs_netdev.c @@ -657,6 +657,9 @@ static ssize_t ieee80211_if_fmt_tsf( struct ieee80211_local *local = sdata->local; u64 tsf; + if (!ieee80211_sdata_running((struct ieee80211_sub_if_data *)sdata)) + return -ENETDOWN; + tsf = drv_get_tsf(local, (struct ieee80211_sub_if_data *)sdata); return scnprintf(buf, buflen, "0x%016llx\n", (unsigned long long) tsf); @@ -670,6 +673,9 @@ static ssize_t ieee80211_if_parse_tsf( int ret; int tsf_is_delta = 0; + if (!ieee80211_sdata_running(sdata)) + return -ENETDOWN; + if (strncmp(buf, "reset", 5) == 0) { if (local->ops->reset_tsf) { drv_reset_tsf(local, sdata); @@ -729,6 +735,9 @@ static ssize_t ieee80211_if_parse_active_links(struct ieee80211_sub_if_data *sda if (kstrtou16(buf, 0, &active_links) || !active_links) return -EINVAL; + if (!ieee80211_sdata_running(sdata)) + return -ENETDOWN; + return ieee80211_set_active_links(&sdata->vif, active_links) ?: buflen; } IEEE80211_IF_FILE_RW(active_links); diff --git a/net/mac80211/ieee80211_i.h b/net/mac80211/ieee80211_i.h index 5761e9621491..d05f59467399 100644 --- a/net/mac80211/ieee80211_i.h +++ b/net/mac80211/ieee80211_i.h @@ -2432,7 +2432,7 @@ int ieee80211_reconfig(struct ieee80211_local *local); void ieee80211_stop_device(struct ieee80211_local *local, bool suspend); int __ieee80211_suspend(struct ieee80211_hw *hw, - struct cfg80211_wowlan *wowlan); + struct cfg80211_wowlan *wowlan, bool reset); static inline int __ieee80211_resume(struct ieee80211_hw *hw) { diff --git a/net/mac80211/iface.c b/net/mac80211/iface.c index 43460a705a6b..889c32fd8de1 100644 --- a/net/mac80211/iface.c +++ b/net/mac80211/iface.c @@ -616,6 +616,8 @@ static void ieee80211_do_stop(struct ieee80211_sub_if_data *sdata, bool going_do RCU_INIT_POINTER(sdata->vif.bss_conf.chanctx_conf, NULL); /* see comment in the default case below */ ieee80211_free_keys(sdata, true); + /* increased by AP value on ifup, so reset on ifdown */ + sdata->crypto_tx_tailroom_needed_cnt = 0; /* no need to tell driver */ break; case NL80211_IFTYPE_MONITOR: @@ -924,9 +926,33 @@ static void ieee80211_teardown_sdata(struct ieee80211_sub_if_data *sdata) } } +/* + * The netdev can be unregistered without mac80211 doing it, e.g. by the netdev + * core when cfg80211 couldn't move it out of a network namespace that's being + * destroyed. Drop it from the interface list either way. + */ +static void ieee80211_unlist_sdata(struct ieee80211_sub_if_data *sdata) +{ + struct ieee80211_local *local = sdata->local; + struct ieee80211_sub_if_data *iter; + + ASSERT_RTNL(); + + list_for_each_entry(iter, &local->interfaces, list) { + if (iter != sdata) + continue; + guard(mutex)(&local->iflist_mtx); + list_del_rcu(&sdata->list); + return; + } +} + static void ieee80211_uninit(struct net_device *dev) { - ieee80211_teardown_sdata(IEEE80211_DEV_TO_SUB_IF(dev)); + struct ieee80211_sub_if_data *sdata = IEEE80211_DEV_TO_SUB_IF(dev); + + ieee80211_unlist_sdata(sdata); + ieee80211_teardown_sdata(sdata); } static int ieee80211_netdev_setup_tc(struct net_device *dev, @@ -935,6 +961,9 @@ static int ieee80211_netdev_setup_tc(struct net_device *dev, struct ieee80211_sub_if_data *sdata = IEEE80211_DEV_TO_SUB_IF(dev); struct ieee80211_local *local = sdata->local; + if (sdata->vif.type == NL80211_IFTYPE_AP_VLAN) + return -EOPNOTSUPP; + return drv_net_setup_tc(local, sdata, dev, type, type_data); } @@ -964,7 +993,7 @@ static u16 ieee80211_monitor_select_queue(struct net_device *dev, /* reset flags and info before parsing radiotap header */ memset(info, 0, sizeof(*info)); - if (!ieee80211_parse_tx_radiotap(skb, dev)) + if (!ieee80211_parse_tx_radiotap(skb, dev, NULL)) return 0; /* doesn't matter, frame will be dropped */ len_rthdr = ieee80211_get_radiotap_len(skb->data); @@ -1603,8 +1632,12 @@ int ieee80211_do_open(struct wireless_dev *wdev, bool coming_up) err_del_interface: drv_remove_interface(local, sdata); err_stop: - if (!local->open_count) + if (!local->open_count) { + ieee80211_led_radio(local, false); + ieee80211_mod_tpt_led_trig(local, 0, + IEEE80211_TPT_LEDTRIG_FL_RADIO); drv_stop(local, false); + } if (sdata->vif.type == NL80211_IFTYPE_NAN_DATA) RCU_INIT_POINTER(sdata->u.nan_data.nmi, NULL); if (sdata->vif.type == NL80211_IFTYPE_AP_VLAN) diff --git a/net/mac80211/main.c b/net/mac80211/main.c index a59837b9f480..6408e8464338 100644 --- a/net/mac80211/main.c +++ b/net/mac80211/main.c @@ -1453,6 +1453,10 @@ int ieee80211_register_hw(struct ieee80211_hw *hw) sizeof(struct ieee80211_he_mcs_nss_supp) + IEEE80211_HE_PPE_THRES_MAX_LEN; + if (local->hw.wiphy->bands[NL80211_BAND_6GHZ]) + local->scan_ies_len += + 3 + sizeof(struct ieee80211_he_6ghz_capa); + if (supp_eht) local->scan_ies_len += 3 + sizeof(struct ieee80211_eht_cap_elem) + diff --git a/net/mac80211/mesh.c b/net/mac80211/mesh.c index d4507e4e6ec1..8f8814125375 100644 --- a/net/mac80211/mesh.c +++ b/net/mac80211/mesh.c @@ -1196,6 +1196,21 @@ int ieee80211_start_mesh(struct ieee80211_sub_if_data *sdata) return 0; } +static void ieee80211_mesh_reset_csa(struct ieee80211_sub_if_data *sdata) +{ + struct ieee80211_if_mesh *ifmsh = &sdata->u.mesh; + struct mesh_csa_settings *csa; + + /* Reset the TTL value and Initiator flag */ + ifmsh->csa_role = IEEE80211_MESH_CSA_ROLE_NONE; + ifmsh->chsw_ttl = 0; + + /* Remove the CSA and MCSP elements from the beacon */ + csa = sdata_dereference(ifmsh->csa, sdata); + RCU_INIT_POINTER(ifmsh->csa, NULL); + kfree_rcu(csa, rcu_head); +} + void ieee80211_stop_mesh(struct ieee80211_sub_if_data *sdata) { struct ieee80211_local *local = sdata->local; @@ -1204,6 +1219,11 @@ void ieee80211_stop_mesh(struct ieee80211_sub_if_data *sdata) netif_carrier_off(sdata->dev); + /* abort any running channel switch */ + sdata->vif.bss_conf.csa_active = false; + ieee80211_mesh_reset_csa(sdata); + ieee80211_vif_unblock_queues_csa(sdata); + /* flush STAs and mpaths on this iface */ sta_info_flush(sdata, -1); ieee80211_free_keys(sdata, true); @@ -1510,19 +1530,10 @@ free: int ieee80211_mesh_finish_csa(struct ieee80211_sub_if_data *sdata, u64 *changed) { - struct ieee80211_if_mesh *ifmsh = &sdata->u.mesh; - struct mesh_csa_settings *tmp_csa_settings; - int ret = 0; + int ret; - /* Reset the TTL value and Initiator flag */ - ifmsh->csa_role = IEEE80211_MESH_CSA_ROLE_NONE; - ifmsh->chsw_ttl = 0; + ieee80211_mesh_reset_csa(sdata); - /* Remove the CSA and MCSP elements from the beacon */ - tmp_csa_settings = sdata_dereference(ifmsh->csa, sdata); - RCU_INIT_POINTER(ifmsh->csa, NULL); - if (tmp_csa_settings) - kfree_rcu(tmp_csa_settings, rcu_head); ret = ieee80211_mesh_rebuild_beacon(sdata); if (ret) return -EINVAL; @@ -1555,7 +1566,6 @@ int ieee80211_mesh_csa_beacon(struct ieee80211_sub_if_data *sdata, ret = ieee80211_mesh_rebuild_beacon(sdata); if (ret) { - tmp_csa_settings = rcu_dereference(ifmsh->csa); RCU_INIT_POINTER(ifmsh->csa, NULL); kfree_rcu(tmp_csa_settings, rcu_head); return ret; diff --git a/net/mac80211/offchannel.c b/net/mac80211/offchannel.c index 7acef80d5f1f..e30767853c36 100644 --- a/net/mac80211/offchannel.c +++ b/net/mac80211/offchannel.c @@ -460,6 +460,13 @@ static void __ieee80211_roc_work(struct ieee80211_local *local) return; if (!roc->started) { + /* + * The work can be started by a previous ROC work, but a scan + * can get between things; scan finish will retrigger us. + */ + if (local->scanning) + return; + WARN_ON(!local->emulate_chanctx); _ieee80211_start_next_roc(local); } else { diff --git a/net/mac80211/pm.c b/net/mac80211/pm.c index 5a508d99e84f..f63676c44853 100644 --- a/net/mac80211/pm.c +++ b/net/mac80211/pm.c @@ -18,7 +18,8 @@ static void ieee80211_sched_scan_cancel(struct ieee80211_local *local) cfg80211_sched_scan_stopped_locked(local->hw.wiphy, 0); } -int __ieee80211_suspend(struct ieee80211_hw *hw, struct cfg80211_wowlan *wowlan) +int __ieee80211_suspend(struct ieee80211_hw *hw, struct cfg80211_wowlan *wowlan, + bool reset) { struct ieee80211_local *local = hw_to_local(hw); struct ieee80211_sub_if_data *sdata; @@ -166,9 +167,10 @@ int __ieee80211_suspend(struct ieee80211_hw *hw, struct cfg80211_wowlan *wowlan) /* * We disconnected on all interfaces before suspend, all channel - * contexts should be released. + * contexts should be released, but on 'reset' debugfs that's + * not true so don't check there. */ - WARN_ON(!list_empty(&local->chanctx_list)); + WARN_ON(!reset && !list_empty(&local->chanctx_list)); /* stop hardware - this must stop RX */ ieee80211_stop_device(local, true); diff --git a/net/mac80211/rate.c b/net/mac80211/rate.c index 64768abb0a5f..e910f03af777 100644 --- a/net/mac80211/rate.c +++ b/net/mac80211/rate.c @@ -372,6 +372,14 @@ static void __rate_control_send_low(struct ieee80211_hw *hw, u32 rate_flags = 0; int i; + /* + * Frames that shouldn't use the rate mask could be anything, + * even on a different band, so don't take the sta into account + * to avoid ending up without rates. + */ + if (info->control.flags & IEEE80211_TX_CTRL_DONT_USE_RATE_MASK) + sta = NULL; + if (sband->band == NL80211_BAND_S1GHZ) { info->control.rates[0].flags |= IEEE80211_TX_RC_S1G_MCS; info->control.rates[0].idx = 0; diff --git a/net/mac80211/scan.c b/net/mac80211/scan.c index eeff230bd909..8e950ef6d1ea 100644 --- a/net/mac80211/scan.c +++ b/net/mac80211/scan.c @@ -1242,7 +1242,7 @@ int ieee80211_request_ibss_scan(struct ieee80211_sub_if_data *sdata, } } - if (WARN_ON_ONCE(n_ch == 0)) + if (n_ch == 0) return -EINVAL; local->int_scan_req->n_channels = n_ch; diff --git a/net/mac80211/tdls.c b/net/mac80211/tdls.c index dc2f662fe4c4..f663d28d9209 100644 --- a/net/mac80211/tdls.c +++ b/net/mac80211/tdls.c @@ -1142,6 +1142,7 @@ ieee80211_tdls_mgmt_setup(struct wiphy *wiphy, struct net_device *dev, struct ieee80211_local *local = sdata->local; enum ieee80211_smps_mode smps_mode = sdata->deflink.u.mgd.driver_smps_mode; + struct sta_info *sta; int ret; /* don't support setup with forced SMPS mode that's not off */ @@ -1168,14 +1169,10 @@ ieee80211_tdls_mgmt_setup(struct wiphy *wiphy, struct net_device *dev, * Allow error packets to be sent - sometimes we don't even add a STA * before failing the setup. */ - if (status_code == 0) { - rcu_read_lock(); - if (!sta_info_get(sdata, peer)) { - rcu_read_unlock(); - ret = -ENOLINK; - goto out_unlock; - } - rcu_read_unlock(); + sta = sta_info_get(sdata, peer); + if ((status_code == 0 && !sta) || (sta && !sta->sta.tdls)) { + ret = -ENOLINK; + goto out_unlock; } ieee80211_flush_queues(local, sdata, false); @@ -1284,6 +1281,24 @@ int ieee80211_tdls_mgmt(struct wiphy *wiphy, struct net_device *dev, peer_capability, initiator, extra_ies, extra_ies_len); break; + case WLAN_TDLS_SETUP_CONFIRM: { + struct sta_info *sta; + + sta = sta_info_get(sdata, peer); + if (!sta || !sta->sta.tdls) { + ret = -ENOLINK; + break; + } + + ret = ieee80211_tdls_prep_mgmt_packet(wiphy, dev, peer, + link_id, action_code, + dialog_token, + status_code, + peer_capability, + initiator, extra_ies, + extra_ies_len, 0, NULL); + break; + } case WLAN_TDLS_DISCOVERY_REQUEST: /* * Protect the discovery so we can hear the TDLS discovery @@ -1292,7 +1307,6 @@ int ieee80211_tdls_mgmt(struct wiphy *wiphy, struct net_device *dev, */ drv_mgd_protect_tdls_discover(sdata->local, sdata, link_id); fallthrough; - case WLAN_TDLS_SETUP_CONFIRM: case WLAN_PUB_ACTION_TDLS_DISCOVER_RES: /* no special handling */ ret = ieee80211_tdls_prep_mgmt_packet(wiphy, dev, peer, @@ -1442,6 +1456,10 @@ int ieee80211_tdls_oper(struct wiphy *wiphy, struct net_device *dev, */ tdls_dbg(sdata, "TDLS oper %d peer %pM\n", oper, peer); + sta = sta_info_get(sdata, peer); + if (!sta || !sta->sta.tdls) + return -ENOLINK; + switch (oper) { case NL80211_TDLS_ENABLE_LINK: if (sdata->vif.bss_conf.csa_active) { @@ -1449,10 +1467,6 @@ int ieee80211_tdls_oper(struct wiphy *wiphy, struct net_device *dev, return -EBUSY; } - sta = sta_info_get(sdata, peer); - if (!sta || !sta->sta.tdls) - return -ENOLINK; - iee80211_tdls_recalc_chanctx(sdata, sta); iee80211_tdls_recalc_ht_protection(sdata, sta); diff --git a/net/mac80211/tx.c b/net/mac80211/tx.c index 3a1e2c9e1565..814399989b5e 100644 --- a/net/mac80211/tx.c +++ b/net/mac80211/tx.c @@ -744,10 +744,12 @@ ieee80211_tx_h_rate_ctrl(struct ieee80211_tx_data *tx) assoc = test_sta_flag(tx->sta, WLAN_STA_ASSOC); /* - * Lets not bother rate control if we're associated and cannot - * talk to the sta. This should not happen. + * Lets not bother rate control if we're associated and cannot talk to + * the sta. This should not happen - except for frames that aren't + * really for the peer to start with and already ignore rates. */ - if (WARN(test_bit(SCAN_SW_SCANNING, &tx->local->scanning) && assoc && + if (!(info->control.flags & IEEE80211_TX_CTRL_DONT_USE_RATE_MASK) && + WARN(test_bit(SCAN_SW_SCANNING, &tx->local->scanning) && assoc && !rate_usable_index_exists(sband, &tx->sta->sta), "%s: Dropped data frame as no usable bitrate found while " "scanning and associated. Target station: " @@ -2103,8 +2105,29 @@ static bool ieee80211_validate_radiotap_len(struct sk_buff *skb) return true; } +static bool ieee80211_rate_bw_usable(u16 rate_flags, + const struct cfg80211_chan_def *chandef) +{ + int width; + + if (!chandef) + return true; + + if (rate_flags & IEEE80211_TX_RC_160_MHZ_WIDTH) + width = 160; + else if (rate_flags & IEEE80211_TX_RC_80_MHZ_WIDTH) + width = 80; + else if (rate_flags & IEEE80211_TX_RC_40_MHZ_WIDTH) + width = 40; + else + return true; + + return width <= cfg80211_chandef_get_width(chandef); +} + bool ieee80211_parse_tx_radiotap(struct sk_buff *skb, - struct net_device *dev) + struct net_device *dev, + const struct cfg80211_chan_def *chandef) { struct ieee80211_local *local = wdev_priv(dev->ieee80211_ptr); struct ieee80211_radiotap_iterator iterator; @@ -2278,6 +2301,9 @@ bool ieee80211_parse_tx_radiotap(struct sk_buff *skb, struct ieee80211_supported_band *sband = local->hw.wiphy->bands[info->band]; + if (!ieee80211_rate_bw_usable(rate_flags, chandef)) + return false; + info->control.flags |= IEEE80211_TX_CTRL_RATE_INJECT; for (i = 0; i < IEEE80211_TX_MAX_RATES; i++) { @@ -2477,7 +2503,7 @@ netdev_tx_t ieee80211_monitor_start_xmit(struct sk_buff *skb, * selected chandef above to accurately set injection rates and * retransmissions. */ - if (!ieee80211_parse_tx_radiotap(skb, dev)) + if (!ieee80211_parse_tx_radiotap(skb, dev, chandef)) goto fail_rcu; /* remove the injection radiotap header */ @@ -2955,10 +2981,23 @@ static struct sk_buff *ieee80211_build_hdr(struct ieee80211_sub_if_data *sdata, */ skb = skb_share_check(skb, GFP_ATOMIC); if (unlikely(!skb)) { - ret = -ENOMEM; - goto free; + /* skb_share_check() already freed the skb */ + if (info_id) + ieee80211_remove_ack_skb(local, info_id); + return ERR_PTR(-ENOMEM); } + /* set this up so failure paths can clean up ack skb */ + info = IEEE80211_SKB_CB(skb); + memset(info, 0, sizeof(*info)); + + info->flags = info_flags; + if (info_id) { + info->status_data = info_id; + info->status_data_idr = 1; + } + info->band = band; + hdr.frame_control = fc; hdr.duration_id = 0; hdr.seq_ctrl = 0; @@ -2997,10 +3036,8 @@ static struct sk_buff *ieee80211_build_hdr(struct ieee80211_sub_if_data *sdata, head_need += local->tx_headroom; head_need = max_t(int, 0, head_need); if (ieee80211_skb_resize(sdata, skb, head_need, ENCRYPT_DATA)) { - ieee80211_free_txskb(&local->hw, skb); - skb = NULL; ret = -ENOMEM; - goto free; + goto free_txskb; } } @@ -3027,16 +3064,6 @@ static struct sk_buff *ieee80211_build_hdr(struct ieee80211_sub_if_data *sdata, skb_reset_mac_header(skb); - info = IEEE80211_SKB_CB(skb); - memset(info, 0, sizeof(*info)); - - info->flags = info_flags; - if (info_id) { - info->status_data = info_id; - info->status_data_idr = 1; - } - info->band = band; - if (likely(!cookie)) { ctrl_flags |= u32_encode_bits(link_id, IEEE80211_TX_CTRL_MLO_LINK); @@ -3060,16 +3087,17 @@ static struct sk_buff *ieee80211_build_hdr(struct ieee80211_sub_if_data *sdata, pre_conf_link_id, link_id); #endif ret = -EINVAL; - goto free; + goto free_txskb; } } info->control.flags = ctrl_flags; return skb; + free_txskb: + ieee80211_free_txskb(&local->hw, skb); + return ERR_PTR(ret); free: - if (info_id) - ieee80211_remove_ack_skb(local, info_id); kfree_skb(skb); return ERR_PTR(ret); } @@ -5089,11 +5117,19 @@ static void ieee80211_beacon_add_tim_pvb(struct ps_data *ps, */ static void ieee80211_s1g_beacon_add_tim_pvb(struct ps_data *ps, struct sk_buff *skb, - bool mcast_traffic) + bool mcast_traffic, + bool ucast_traffic) { int blk; /* + * if no unicast and multicast traffic don't emit a bitmap control + * or pvb + */ + if (!mcast_traffic && !ucast_traffic) + return; + + /* * Emit a bitmap control block with a page slice number of 31 and a * page index of 0 which indicates as per IEEE80211-2024 9.4.2.5.1 * that the entire page (2048 bits) indicated by the page index @@ -5101,6 +5137,10 @@ static void ieee80211_s1g_beacon_add_tim_pvb(struct ps_data *ps, */ skb_put_u8(skb, mcast_traffic | (31 << 1)); + /* If there's no unicast traffic we don't need to include a PVB. */ + if (!ucast_traffic) + return; + /* Emit an encoded block for each non-zero sub-block */ for (blk = 0; blk < IEEE80211_MAX_SUPPORTED_S1G_TIM_BLOCKS; blk++) { u8 blk_bmap = 0; @@ -5182,25 +5222,16 @@ static void __ieee80211_beacon_add_tim(struct ieee80211_sub_if_data *sdata, ps->dtim_bc_mc = mcast_traffic; - if (have_bits) { - if (s1g) - ieee80211_s1g_beacon_add_tim_pvb(ps, skb, - mcast_traffic); - else - ieee80211_beacon_add_tim_pvb(ps, skb, mcast_traffic); + if (s1g) { + ieee80211_s1g_beacon_add_tim_pvb(ps, skb, mcast_traffic, + have_bits); + } else if (have_bits) { + ieee80211_beacon_add_tim_pvb(ps, skb, mcast_traffic); } else { - /* - * If there is no buffered unicast traffic for an S1G - * interface, we can exclude the bitmap control. This is in - * contrast to other phy types as they do include the bitmap - * control and pvb even when there is no buffered traffic. - */ - if (!s1g) { - /* Bitmap control */ - skb_put_u8(skb, mcast_traffic); - /* Part Virt Bitmap */ - skb_put_u8(skb, 0); - } + /* Bitmap control */ + skb_put_u8(skb, mcast_traffic); + /* Part Virt Bitmap */ + skb_put_u8(skb, 0); } tim->datalen = skb_tail_pointer(skb) - tim->data; diff --git a/net/mptcp/protocol.c b/net/mptcp/protocol.c index 0098e2830931..e89a69ab927c 100644 --- a/net/mptcp/protocol.c +++ b/net/mptcp/protocol.c @@ -856,12 +856,12 @@ static bool __mptcp_move_skbs_from_subflow(struct mptcp_sock *msk, mptcp_dss_corruption(msk, ssk); } } else { + sk_eat_skb(ssk, skb); + if (unlikely(!fin)) { DEBUG_NET_WARN_ON_ONCE(1); mptcp_dss_corruption(msk, ssk); } - - sk_eat_skb(ssk, skb); } WRITE_ONCE(tp->copied_seq, seq); @@ -1664,7 +1664,9 @@ struct sock *mptcp_subflow_get_send(struct mptcp_sock *msk) static void mptcp_push_release(struct sock *ssk, struct mptcp_sendmsg_info *info) { - tcp_push(ssk, 0, info->mss_now, tcp_sk(ssk)->nonagle, info->size_goal); + if (info->mss_now) + tcp_push(ssk, 0, info->mss_now, tcp_sk(ssk)->nonagle, + info->size_goal); release_sock(ssk); } @@ -1852,7 +1854,8 @@ static void __mptcp_subflow_push_pending(struct sock *sk, struct sock *ssk, bool ret = __subflow_push_pending(sk, ssk, &info); if (ret <= 0) keep_pushing = false; - copied += ret; + else + copied += ret; } mptcp_for_each_subflow(msk, subflow) { @@ -2459,7 +2462,7 @@ static int mptcp_recvmsg(struct sock *sk, struct msghdr *msg, size_t len, mptcp_cleanup_rbuf(msk, copied); err = sk_wait_data(sk, &timeo, last); if (err < 0) { - err = copied ? : err; + copied = copied ? : err; goto out_err; } } diff --git a/net/mptcp/protocol.h b/net/mptcp/protocol.h index 2b4c27426477..0384d6a023f9 100644 --- a/net/mptcp/protocol.h +++ b/net/mptcp/protocol.h @@ -585,7 +585,8 @@ struct mptcp_subflow_context { is_mptfo : 1, /* subflow is doing TFO */ close_event_done : 1, /* has done the post-closed part */ mpc_drop : 1, /* the MPC option has been dropped in a rtx */ - __unused : 9; + resetting : 1, /* subflow is resetting */ + __unused : 8; bool data_avail; bool scheduled; bool pm_listener; /* a listener managed by the kernel PM? */ diff --git a/net/mptcp/subflow.c b/net/mptcp/subflow.c index 01db7edce18a..f0a6725d2c37 100644 --- a/net/mptcp/subflow.c +++ b/net/mptcp/subflow.c @@ -438,6 +438,10 @@ void mptcp_subflow_reset(struct sock *ssk) /* must hold: tcp_done() could drop last reference on parent */ sock_hold(sk); + subflow->resetting = 1; + + /* No need to delay the actual close for to-be discarded data. */ + __skb_queue_purge(&ssk->sk_receive_queue); mptcp_send_active_reset_reason(ssk); tcp_done(ssk); if (!test_and_set_bit(MPTCP_WORK_CLOSE_SUBFLOW, &mptcp_sk(sk)->flags)) @@ -1883,6 +1887,13 @@ static void subflow_state_change(struct sock *sk) __subflow_state_change(sk); + /* Rx queue processing is unneeded, error reporting will take place at + * __mptcp_close_ssk() time and subflow reset can't happen in case of + * fallback: subflow_sched_work_if_closed() would be a no-op. + */ + if (subflow->resetting) + return; + /* as recvmsg() does not acquire the subflow socket for ssk selection * a fin packet carrying a DSS can be unnoticed if we don't trigger * the data available machinery here. diff --git a/net/netfilter/nf_flow_table_core.c b/net/netfilter/nf_flow_table_core.c index 03241d4bfd5e..934c6151f558 100644 --- a/net/netfilter/nf_flow_table_core.c +++ b/net/netfilter/nf_flow_table_core.c @@ -258,6 +258,14 @@ static void flow_offload_route_release(struct flow_offload *flow) nft_flow_dst_release(flow, FLOW_OFFLOAD_DIR_REPLY); } +static void flow_offload_free_rcu(struct rcu_head *rcu_head) +{ + struct flow_offload *flow = container_of(rcu_head, struct flow_offload, rcu_head); + + nf_ct_put(flow->ct); + kfree(flow); +} + void flow_offload_free(struct flow_offload *flow) { switch (flow->type) { @@ -267,8 +275,7 @@ void flow_offload_free(struct flow_offload *flow) default: break; } - nf_ct_put(flow->ct); - kfree_rcu(flow, rcu_head); + call_rcu(&flow->rcu_head, flow_offload_free_rcu); } EXPORT_SYMBOL_GPL(flow_offload_free); @@ -854,6 +861,7 @@ out_pernet: static void __exit nf_flow_table_module_exit(void) { + rcu_barrier(); nf_flow_table_offload_exit(); unregister_pernet_subsys(&nf_flow_table_net_ops); kmem_cache_destroy(flow_offload_cachep); diff --git a/net/netfilter/nf_nat_core.c b/net/netfilter/nf_nat_core.c index 8ac326e1eb5b..a4858c2b2d65 100644 --- a/net/netfilter/nf_nat_core.c +++ b/net/netfilter/nf_nat_core.c @@ -1224,31 +1224,45 @@ int nf_nat_register_fn(struct net *net, u8 pf, const struct nf_hook_ops *ops, } ret = nf_register_net_hooks(net, nat_ops, ops_count); - if (ret < 0) { - mutex_unlock(&nf_nat_proto_mutex); - for (i = 0; i < ops_count; i++) { - priv = nat_ops[i].priv; - kfree_rcu(priv, rcu_head); - } - kfree_rcu(nat_ops, rcu); - return ret; - } - - nat_proto_net->nat_hook_ops = nat_ops; + if (ret < 0) + goto err_free_hooks; + } else { + nat_ops = nat_proto_net->nat_hook_ops; } - nat_ops = nat_proto_net->nat_hook_ops; priv = nat_ops[hooknum].priv; if (WARN_ON_ONCE(!priv)) { - mutex_unlock(&nf_nat_proto_mutex); - return -EOPNOTSUPP; + ret = -EOPNOTSUPP; + goto err_unregister_hooks; } ret = nf_hook_entries_insert_raw(&priv->entries, ops); - if (ret == 0) - nat_proto_net->users++; + if (ret) + goto err_unregister_hooks; + + if (!nat_proto_net->nat_hook_ops) + nat_proto_net->nat_hook_ops = nat_ops; + + nat_proto_net->users++; mutex_unlock(&nf_nat_proto_mutex); + + return 0; + +err_unregister_hooks: + if (nat_proto_net->nat_hook_ops) { + mutex_unlock(&nf_nat_proto_mutex); + return ret; + } + nf_unregister_net_hooks(net, nat_ops, ops_count); +err_free_hooks: + mutex_unlock(&nf_nat_proto_mutex); + for (i = 0; i < ops_count; i++) { + priv = nat_ops[i].priv; + kfree_rcu(priv, rcu_head); + } + kfree_rcu(nat_ops, rcu); + return ret; } diff --git a/net/netfilter/nf_tables_api.c b/net/netfilter/nf_tables_api.c index 31fbd5a28937..c0b754a2d45b 100644 --- a/net/netfilter/nf_tables_api.c +++ b/net/netfilter/nf_tables_api.c @@ -2440,11 +2440,14 @@ err_hook_free: } static struct nft_hook *nft_hook_list_find(struct list_head *hook_list, - const struct nft_hook *this) + const struct nft_hook *this, + bool strict) { struct nft_hook *hook; list_for_each_entry(hook, hook_list, list) { + if (strict && hook->ifnamelen != this->ifnamelen) + continue; if (!strncmp(hook->ifname, this->ifname, min(hook->ifnamelen, this->ifnamelen))) { if (hook->flags & NFT_HOOK_REMOVE) @@ -2486,7 +2489,7 @@ static int nf_tables_parse_netdev_hooks(struct net *net, err = PTR_ERR(hook); goto err_hook; } - if (nft_hook_list_find(hook_list, hook)) { + if (nft_hook_list_find(hook_list, hook, false)) { NL_SET_BAD_ATTR(extack, tmp); nft_netdev_hook_free(hook); err = -EEXIST; @@ -2943,7 +2946,7 @@ static int nf_tables_updchain(struct nft_ctx *ctx, u8 genmask, u8 policy, ops->hook = basechain->ops.hook; } - if (nft_hook_list_find(&basechain->hook_list, h)) { + if (nft_hook_list_find(&basechain->hook_list, h, false)) { list_del(&h->list); nft_netdev_hook_free(h); continue; @@ -2956,7 +2959,8 @@ static int nf_tables_updchain(struct nft_ctx *ctx, u8 genmask, u8 policy, !nft_trans_chain_update(trans)) continue; - if (nft_hook_list_find(&nft_trans_chain_hooks(trans), h)) { + if (nft_hook_list_find(&nft_trans_chain_hooks(trans), + h, false)) { nft_chain_release_hook(&hook); return -EEXIST; } @@ -3257,7 +3261,7 @@ static int nft_delchain_hook(struct nft_ctx *ctx, return err; list_for_each_entry(this, &chain_hook.list, list) { - hook = nft_hook_list_find(&basechain->hook_list, this); + hook = nft_hook_list_find(&basechain->hook_list, this, true); if (!hook) { err = -ENOENT; goto err_chain_del_hook; @@ -9053,7 +9057,7 @@ static int nft_register_flowtable_net_hooks(struct net *net, if (!nft_is_active_next(net, ft)) continue; - if (nft_hook_list_find(&ft->hook_list, hook)) { + if (nft_hook_list_find(&ft->hook_list, hook, false)) { err = -EEXIST; goto err_unregister_net_hooks; } @@ -9130,7 +9134,7 @@ static int nft_flowtable_update(struct nft_ctx *ctx, const struct nlmsghdr *nlh, return err; list_for_each_entry_safe(hook, next, &flowtable_hook.list, list) { - if (nft_hook_list_find(&flowtable->hook_list, hook)) { + if (nft_hook_list_find(&flowtable->hook_list, hook, false)) { list_del(&hook->list); nft_netdev_hook_free(hook); continue; @@ -9143,7 +9147,7 @@ static int nft_flowtable_update(struct nft_ctx *ctx, const struct nlmsghdr *nlh, !nft_trans_flowtable_update(trans)) continue; - if (nft_hook_list_find(&nft_trans_flowtable_hooks(trans), hook)) { + if (nft_hook_list_find(&nft_trans_flowtable_hooks(trans), hook, false)) { err = -EEXIST; goto err_flowtable_update_hook; } @@ -9363,7 +9367,7 @@ static int nft_delflowtable_hook(struct nft_ctx *ctx, return err; list_for_each_entry(this, &flowtable_hook.list, list) { - hook = nft_hook_list_find(&flowtable->hook_list, this); + hook = nft_hook_list_find(&flowtable->hook_list, this, true); if (!hook) { err = -ENOENT; goto err_flowtable_del_hook; diff --git a/net/netfilter/nft_nat.c b/net/netfilter/nft_nat.c index e32cd9fbc7c2..cdbd800cac96 100644 --- a/net/netfilter/nft_nat.c +++ b/net/netfilter/nft_nat.c @@ -64,8 +64,8 @@ static void nft_nat_setup_netmap(struct nf_nat_range2 *range, const struct nft_pktinfo *pkt, const struct nft_nat *priv) { + union nf_inet_addr new_addr = {}; struct sk_buff *skb = pkt->skb; - union nf_inet_addr new_addr; __be32 netmask; int i, len = 0; diff --git a/net/netlink/af_netlink.c b/net/netlink/af_netlink.c index e6b1d9758c9c..9fdf964224ab 100644 --- a/net/netlink/af_netlink.c +++ b/net/netlink/af_netlink.c @@ -922,9 +922,9 @@ netlink_update_subscriptions(struct sock *sk, unsigned int subscriptions) static int netlink_realloc_groups(struct sock *sk) { + unsigned long *new_groups, *old_groups = NULL; struct netlink_sock *nlk = nlk_sk(sk); unsigned int groups; - unsigned long *new_groups; int err = 0; netlink_table_grab(); @@ -938,18 +938,37 @@ static int netlink_realloc_groups(struct sock *sk) if (nlk->ngroups >= groups) goto out_unlock; - new_groups = krealloc(nlk->groups, NLGRPSZ(groups), GFP_ATOMIC); - if (new_groups == NULL) { + /* Can not use krealloc(), because the old buffer might be freed + * immediately, while lockless readers (netlink diag dump and + * /proc/net/netlink) can still be looking at it. + */ + new_groups = kzalloc(NLGRPSZ(groups), GFP_ATOMIC); + if (!new_groups) { err = -ENOMEM; goto out_unlock; } - memset((char *)new_groups + NLGRPSZ(nlk->ngroups), 0, - NLGRPSZ(groups) - NLGRPSZ(nlk->ngroups)); + old_groups = nlk->groups; + if (old_groups) + memcpy(new_groups, old_groups, NLGRPSZ(nlk->ngroups)); + + /* Publish the new bitmap and its content: pairs with the address + * dependency in lockless readers, which can pick up the new pointer + * while still seeing the old (smaller) nlk->ngroups. + */ + smp_store_release(&nlk->groups, new_groups); + + /* Then publish the new size: pairs with smp_load_acquire() from + * lockless readers, so that they can not read NLGRPSZ(new ngroups) + * bytes from the old buffer. + */ + smp_store_release(&nlk->ngroups, groups); - nlk->groups = new_groups; - nlk->ngroups = groups; out_unlock: netlink_table_ungrab(); + + if (old_groups) + kfree_rcu_mightsleep(old_groups); + return err; } @@ -2705,12 +2724,19 @@ static int netlink_native_seq_show(struct seq_file *seq, void *v) } else { struct sock *s = v; struct netlink_sock *nlk = nlk_sk(s); + const unsigned long *groups; + + /* Lockless read : netlink_realloc_groups() can change + * nlk->groups under us. The old buffer is freed after an + * RCU grace period, and this walk is RCU protected. + */ + groups = READ_ONCE(nlk->groups); seq_printf(seq, "%pK %-3d %-10u %08x %-8d %-8d %-5d %-8d %-8u %-8llu\n", s, s->sk_protocol, nlk->portid, - nlk->groups ? (u32)nlk->groups[0] : 0, + groups ? (u32)groups[0] : 0, sk_rmem_alloc_get(s), sk_wmem_alloc_get(s), READ_ONCE(nlk->cb_running), diff --git a/net/netlink/diag.c b/net/netlink/diag.c index 0b3e021bd0ed..7979bd9b2606 100644 --- a/net/netlink/diag.c +++ b/net/netlink/diag.c @@ -12,12 +12,24 @@ static int sk_diag_dump_groups(struct sock *sk, struct sk_buff *nlskb) { struct netlink_sock *nlk = nlk_sk(sk); - - if (nlk->groups == NULL) + unsigned long *groups; + unsigned int ngroups; + + /* Hashed sockets are dumped from the rhashtable walk, which only + * holds rcu_read_lock(), while netlink_realloc_groups() can replace + * nlk->groups and nlk->ngroups at any time. + * + * Read nlk->ngroups first : this pairs with smp_store_release() + * from netlink_realloc_groups(), so that we can not use the new + * (bigger) size with the old (smaller) buffer. The old buffer is + * freed after an RCU grace period. + */ + ngroups = smp_load_acquire(&nlk->ngroups); + groups = READ_ONCE(nlk->groups); + if (!groups) return 0; - return nla_put(nlskb, NETLINK_DIAG_GROUPS, NLGRPSZ(nlk->ngroups), - nlk->groups); + return nla_put(nlskb, NETLINK_DIAG_GROUPS, NLGRPSZ(ngroups), groups); } static int sk_diag_put_flags(struct sock *sk, struct sk_buff *skb) diff --git a/net/openvswitch/conntrack.c b/net/openvswitch/conntrack.c index 27115967e5d9..0f433688e17b 100644 --- a/net/openvswitch/conntrack.c +++ b/net/openvswitch/conntrack.c @@ -366,7 +366,7 @@ static struct nf_conn_labels *ovs_ct_get_conn_labels(struct nf_conn *ct) struct nf_conn_labels *cl; cl = nf_ct_labels_find(ct); - if (!cl) { + if (!cl && !nf_ct_is_confirmed(ct)) { nf_ct_labels_ext_add(ct); cl = nf_ct_labels_find(ct); } diff --git a/net/packet/af_packet.c b/net/packet/af_packet.c index 76bde7906d49..50cae32ae269 100644 --- a/net/packet/af_packet.c +++ b/net/packet/af_packet.c @@ -2384,7 +2384,9 @@ static int tpacket_rcv(struct sk_buff *skb, struct net_device *dev, virtio_net_hdr_from_skb(skb, h.raw + macoff - sizeof(struct virtio_net_hdr), vio_le(), true, 0)) { - if (po->tp_version == TPACKET_V3) + if (po->tp_version <= TPACKET_V2) + __clear_bit(slot_id, po->rx_ring.rx_owner_map); + else prb_clear_blk_fill_status(&po->rx_ring); goto drop_n_account; } diff --git a/net/packet/internal.h b/net/packet/internal.h index b76e645cd78d..f5c8cd0eed2c 100644 --- a/net/packet/internal.h +++ b/net/packet/internal.h @@ -21,7 +21,7 @@ struct tpacket_kbdq_core { unsigned int hdrlen; unsigned char reset_pending_on_curr_blk; unsigned short kactive_blk_num; - unsigned short blk_sizeof_priv; + unsigned int blk_sizeof_priv; unsigned short version; diff --git a/net/psp/psp_sock.c b/net/psp/psp_sock.c index 1a2a6b7516b0..e9b53eedf8db 100644 --- a/net/psp/psp_sock.c +++ b/net/psp/psp_sock.c @@ -143,6 +143,10 @@ int psp_sock_assoc_set_rx(struct sock *sk, struct psp_assoc *pas, NL_SET_ERR_MSG(extack, "Socket already has PSP state"); err = -EBUSY; goto exit_unlock; + } else if (sk_has_decrypt_user(sk)) { + NL_SET_ERR_MSG(extack, "Socket has incompatible state"); + err = -EINVAL; + goto exit_unlock; } refcount_inc(&pas->refcnt); diff --git a/net/qrtr/af_qrtr.c b/net/qrtr/af_qrtr.c index 78347c937af7..e7b3647424b8 100644 --- a/net/qrtr/af_qrtr.c +++ b/net/qrtr/af_qrtr.c @@ -623,6 +623,19 @@ static void qrtr_hello_work(struct work_struct *work) qrtr_port_put(ctrl); } +/* Trigger the HELLO handshake after the remote has been reset, eg on resume */ +void qrtr_endpoint_hello(struct qrtr_endpoint *ep) +{ + struct qrtr_node *node = ep->node; + + mutex_lock(&node->ep_lock); + node->hello_sent = false; + mutex_unlock(&node->ep_lock); + + schedule_delayed_work(&node->say_hello, 0); +} +EXPORT_SYMBOL_GPL(qrtr_endpoint_hello); + /** * qrtr_endpoint_register() - register a new endpoint * @ep: endpoint to register diff --git a/net/qrtr/mhi.c b/net/qrtr/mhi.c index 3990da1a65dc..e9a4bb92ce76 100644 --- a/net/qrtr/mhi.c +++ b/net/qrtr/mhi.c @@ -183,6 +183,7 @@ static int __maybe_unused qcom_mhi_qrtr_pm_suspend_late(struct device *dev) static int __maybe_unused qcom_mhi_qrtr_pm_resume_early(struct device *dev) { struct mhi_device *mhi_dev = container_of(dev, struct mhi_device, dev); + struct qrtr_mhi_dev *qdev = dev_get_drvdata(dev); enum mhi_state state; int rc; @@ -201,7 +202,13 @@ static int __maybe_unused qcom_mhi_qrtr_pm_resume_early(struct device *dev) return rc; } - return qcom_mhi_qrtr_queue_dl_buffers(mhi_dev); + rc = qcom_mhi_qrtr_queue_dl_buffers(mhi_dev); + if (rc) + return rc; + + qrtr_endpoint_hello(&qdev->ep); + + return 0; } static const struct dev_pm_ops qcom_mhi_qrtr_pm_ops = { diff --git a/net/qrtr/qrtr.h b/net/qrtr/qrtr.h index 3f2d28696062..de2de69a6199 100644 --- a/net/qrtr/qrtr.h +++ b/net/qrtr/qrtr.h @@ -27,6 +27,8 @@ int qrtr_endpoint_register(struct qrtr_endpoint *ep, unsigned int nid); void qrtr_endpoint_unregister(struct qrtr_endpoint *ep); +void qrtr_endpoint_hello(struct qrtr_endpoint *ep); + int qrtr_endpoint_post(struct qrtr_endpoint *ep, const void *data, size_t len); int qrtr_ns_init(void); diff --git a/net/rds/ib_cm.c b/net/rds/ib_cm.c index 4feb0edc360c..6e3110a04ae6 100644 --- a/net/rds/ib_cm.c +++ b/net/rds/ib_cm.c @@ -115,7 +115,7 @@ void rds_ib_cm_connect_complete(struct rds_connection *conn, struct rdma_cm_even &conn->c_laddr, &conn->c_faddr, RDS_PROTOCOL_MAJOR(conn->c_version), RDS_PROTOCOL_MINOR(conn->c_version)); - rds_conn_destroy(conn); + rds_conn_drop(conn); return; } } diff --git a/net/sched/act_api.c b/net/sched/act_api.c index 19501dc99464..3f653721c45f 100644 --- a/net/sched/act_api.c +++ b/net/sched/act_api.c @@ -1218,11 +1218,16 @@ static int tcf_action_put(struct tc_action *p) static void tcf_action_put_many(struct tc_action *actions[]) { - struct tc_action *a; int i; - tcf_act_for_each_action(i, a, actions) { - const struct tc_action_ops *ops = a->ops; + /* Deletion may have cleared entries before failing. */ + for (i = 0; i < TCA_ACT_MAX_PRIO; i++) { + struct tc_action *a = actions[i]; + const struct tc_action_ops *ops; + + if (!a) + continue; + ops = a->ops; if (tcf_action_put(a)) module_put(ops->owner); } diff --git a/net/sched/sch_hhf.c b/net/sched/sch_hhf.c index fc72f825fbd9..5dec1ed969ad 100644 --- a/net/sched/sch_hhf.c +++ b/net/sched/sch_hhf.c @@ -527,7 +527,7 @@ static void hhf_destroy(struct Qdisc *sch) static const struct nla_policy hhf_policy[TCA_HHF_MAX + 1] = { [TCA_HHF_BACKLOG_LIMIT] = { .type = NLA_U32 }, [TCA_HHF_QUANTUM] = { .type = NLA_U32 }, - [TCA_HHF_HH_FLOWS_LIMIT] = { .type = NLA_U32 }, + [TCA_HHF_HH_FLOWS_LIMIT] = NLA_POLICY_MAX(NLA_U32, 2 * HH_FLOWS_CNT), [TCA_HHF_RESET_TIMEOUT] = { .type = NLA_U32 }, [TCA_HHF_ADMIT_BYTES] = { .type = NLA_U32 }, [TCA_HHF_EVICT_TIMEOUT] = { .type = NLA_U32 }, @@ -546,7 +546,7 @@ static int hhf_change(struct Qdisc *sch, struct nlattr *opt, u32 new_hhf_non_hh_weight = q->hhf_non_hh_weight; err = nla_parse_nested_deprecated(tb, TCA_HHF_MAX, opt, hhf_policy, - NULL); + extack); if (err < 0) return err; @@ -624,6 +624,9 @@ static int hhf_init(struct Qdisc *sch, struct nlattr *opt, q->hhf_evict_timeout = HZ; /* 1 sec */ q->hhf_non_hh_weight = 2; + /* Cap max active HHs at twice len of hh_flows table. */ + q->hh_flows_limit = 2 * HH_FLOWS_CNT; + if (opt) { int err = hhf_change(sch, opt, extack); @@ -639,8 +642,6 @@ static int hhf_init(struct Qdisc *sch, struct nlattr *opt, for (i = 0; i < HH_FLOWS_CNT; i++) INIT_LIST_HEAD(&q->hh_flows[i]); - /* Cap max active HHs at twice len of hh_flows table. */ - q->hh_flows_limit = 2 * HH_FLOWS_CNT; q->hh_flows_overlimit = 0; q->hh_flows_total_cnt = 0; q->hh_flows_current_cnt = 0; diff --git a/net/unix/garbage.c b/net/unix/garbage.c index 9fcaaf55cba5..da774f56ca64 100644 --- a/net/unix/garbage.c +++ b/net/unix/garbage.c @@ -374,7 +374,7 @@ static bool unix_vertex_dead(struct unix_vertex *vertex) static LIST_HEAD(unix_visited_vertices); static unsigned long unix_vertex_grouped_index = UNIX_VERTEX_INDEX_MARK2; -static bool unix_scc_dead(struct list_head *scc, bool fast) +static bool unix_scc_dead(struct list_head *scc) { struct unix_vertex *vertex; bool scc_dead = true; @@ -386,10 +386,6 @@ static bool unix_scc_dead(struct list_head *scc, bool fast) /* Don't restart DFS from this vertex. */ list_move_tail(&vertex->entry, &unix_visited_vertices); - /* Mark vertex as off-stack for __unix_walk_scc(). */ - if (!fast) - vertex->index = unix_vertex_grouped_index; - if (scc_dead) scc_dead = unix_vertex_dead(vertex); } @@ -521,6 +517,7 @@ prev_vertex: } if (vertex->index == vertex->scc_index) { + struct unix_vertex *v; struct list_head scc; /* SCC finalised. @@ -530,7 +527,13 @@ prev_vertex: */ __list_cut_position(&scc, &vertex_stack, &vertex->scc_entry); - if (unix_scc_dead(&scc, false)) { + list_for_each_entry_reverse(v, &scc, scc_entry) { + /* Mark vertex as off-stack and assign a unique ID. */ + v->index = unix_vertex_grouped_index; + v->scc_index = vertex->scc_index; + } + + if (unix_scc_dead(&scc)) { unix_collect_skb(&scc, hitlist); } else { if (unix_vertex_max_scc_index < vertex->scc_index) @@ -588,7 +591,7 @@ static void unix_walk_scc_fast(struct sk_buff_head *hitlist) vertex = list_first_entry(&unix_unvisited_vertices, typeof(*vertex), entry); list_add(&scc, &vertex->scc_entry); - if (unix_scc_dead(&scc, true)) { + if (unix_scc_dead(&scc)) { cyclic_sccs--; unix_collect_skb(&scc, hitlist); } diff --git a/net/wireless/core.c b/net/wireless/core.c index 3032993ba5dc..cde3ca85494d 100644 --- a/net/wireless/core.c +++ b/net/wireless/core.c @@ -153,68 +153,93 @@ int cfg80211_dev_rename(struct cfg80211_registered_device *rdev, return 0; } -int cfg80211_switch_netns(struct cfg80211_registered_device *rdev, - struct net *net) +static int cfg80211_switch_wdev_netns(struct wireless_dev *wdev, + struct net *net) { - struct wireless_dev *wdev; - int err = 0; + int err; - if (!(rdev->wiphy.flags & WIPHY_FLAG_NETNS_OK)) - return -EOPNOTSUPP; + if (!wdev->netdev) + return 0; - list_for_each_entry(wdev, &rdev->wiphy.wdev_list, list) { - if (!wdev->netdev) + wdev->netdev->netns_immutable = false; + err = dev_change_net_namespace(wdev->netdev, net, "wlan%d"); + wdev->netdev->netns_immutable = true; + + return err; +} + +static int __cfg80211_switch_netns(struct cfg80211_registered_device *rdev, + struct net *net, bool force) +{ + struct net *old_net = wiphy_net(&rdev->wiphy); + struct wireless_dev *wdev, *tmp; + int err = 0; + + list_for_each_entry_safe(wdev, tmp, &rdev->wiphy.wdev_list, list) { + err = cfg80211_switch_wdev_netns(wdev, net); + if (!err) continue; - wdev->netdev->netns_immutable = false; - err = dev_change_net_namespace(wdev->netdev, net, "wlan%d"); - if (err) - break; - wdev->netdev->netns_immutable = true; + if (!force) + goto undo; + /* remove interfaces that fail to allow wiphy switching */ + dev_close(wdev->netdev); + scoped_guard(wiphy, &rdev->wiphy) + cfg80211_unregister_wdev(wdev); + err = 0; } - if (err) { - /* failed -- clean up to old netns */ - net = wiphy_net(&rdev->wiphy); - - list_for_each_entry_continue_reverse(wdev, - &rdev->wiphy.wdev_list, - list) { + scoped_guard(wiphy, &rdev->wiphy) { + list_for_each_entry(wdev, &rdev->wiphy.wdev_list, list) { if (!wdev->netdev) continue; - wdev->netdev->netns_immutable = false; - err = dev_change_net_namespace(wdev->netdev, net, - "wlan%d"); - WARN_ON(err); - wdev->netdev->netns_immutable = true; + nl80211_notify_iface(rdev, wdev, + NL80211_CMD_DEL_INTERFACE); } - return err; - } + nl80211_notify_wiphy(rdev, NL80211_CMD_DEL_WIPHY); - guard(wiphy)(&rdev->wiphy); + wiphy_net_set(&rdev->wiphy, net); - list_for_each_entry(wdev, &rdev->wiphy.wdev_list, list) { - if (!wdev->netdev) - continue; - nl80211_notify_iface(rdev, wdev, NL80211_CMD_DEL_INTERFACE); - } + /* this only fails on allocation failure */ + err = device_rename(&rdev->wiphy.dev, + dev_name(&rdev->wiphy.dev)); + if (err && !force) + wiphy_net_set(&rdev->wiphy, old_net); - nl80211_notify_wiphy(rdev, NL80211_CMD_DEL_WIPHY); + nl80211_notify_wiphy(rdev, NL80211_CMD_NEW_WIPHY); - wiphy_net_set(&rdev->wiphy, net); + list_for_each_entry(wdev, &rdev->wiphy.wdev_list, list) { + if (!wdev->netdev) + continue; + nl80211_notify_iface(rdev, wdev, + NL80211_CMD_NEW_INTERFACE); + } + } - err = device_rename(&rdev->wiphy.dev, dev_name(&rdev->wiphy.dev)); - WARN_ON(err); + if (!err || force) + return err; - nl80211_notify_wiphy(rdev, NL80211_CMD_NEW_WIPHY); + /* set to the last one to undo all of them */ + wdev = list_entry(&rdev->wiphy.wdev_list, typeof(*wdev), list); +undo: + /* + * Move back everything, if this fails again (allocation failures) + * then things get stuck in different network namespaces. + */ + list_for_each_entry_continue_reverse(wdev, &rdev->wiphy.wdev_list, + list) + WARN_ON(cfg80211_switch_wdev_netns(wdev, old_net)); - list_for_each_entry(wdev, &rdev->wiphy.wdev_list, list) { - if (!wdev->netdev) - continue; - nl80211_notify_iface(rdev, wdev, NL80211_CMD_NEW_INTERFACE); - } + return err; +} - return 0; +int cfg80211_switch_netns(struct cfg80211_registered_device *rdev, + struct net *net) +{ + if (!(rdev->wiphy.flags & WIPHY_FLAG_NETNS_OK)) + return -EOPNOTSUPP; + + return __cfg80211_switch_netns(rdev, net, false); } static void cfg80211_rfkill_poll(struct rfkill *rfkill, void *data) @@ -244,9 +269,8 @@ void cfg80211_stop_p2p_device(struct cfg80211_registered_device *rdev, rdev->opencount--; if (rdev->scan_req && rdev->scan_req->req.wdev == wdev) { - if (WARN_ON(!rdev->scan_req->notified && - (!rdev->int_scan_req || - !rdev->int_scan_req->notified))) + if (!rdev->scan_req->notified && + (!rdev->int_scan_req || !rdev->int_scan_req->notified)) rdev->scan_req->info.aborted = true; ___cfg80211_scan_done(rdev, false); } @@ -645,6 +669,8 @@ use_default_name: INIT_WORK(&rdev->destroy_work, cfg80211_destroy_iface_wk); wiphy_work_init(&rdev->sched_scan_stop_wk, cfg80211_sched_scan_stop_wk); INIT_WORK(&rdev->sched_scan_res_wk, cfg80211_sched_scan_results_wk); + wiphy_work_init(&rdev->reg_check_chans_wk, reg_leave_invalid_chans_wk); + INIT_WORK(&rdev->reg_leave_nan_wk, reg_leave_invalid_nan_wk); INIT_WORK(&rdev->propagate_radar_detect_wk, cfg80211_propagate_radar_detect_wk); INIT_WORK(&rdev->propagate_cac_done_wk, cfg80211_propagate_cac_done_wk); @@ -1344,6 +1370,7 @@ void wiphy_unregister(struct wiphy *wiphy) cancel_delayed_work_sync(&rdev->dfs_update_channels_wk); cancel_delayed_work_sync(&rdev->background_cac_done_wk); flush_work(&rdev->destroy_work); + flush_work(&rdev->reg_leave_nan_wk); flush_work(&rdev->propagate_radar_detect_wk); flush_work(&rdev->propagate_cac_done_wk); flush_work(&rdev->mgmt_registrations_update_wk); @@ -1757,9 +1784,9 @@ static int cfg80211_netdev_notifier_call(struct notifier_block *nb, wiphy_lock(&rdev->wiphy); cfg80211_update_iface_num(rdev, wdev->iftype, -1); if (rdev->scan_req && rdev->scan_req->req.wdev == wdev) { - if (WARN_ON(!rdev->scan_req->notified && - (!rdev->int_scan_req || - !rdev->int_scan_req->notified))) + if (!rdev->scan_req->notified && + (!rdev->int_scan_req || + !rdev->int_scan_req->notified)) rdev->scan_req->info.aborted = true; ___cfg80211_scan_done(rdev, false); } @@ -1867,7 +1894,7 @@ static void __net_exit cfg80211_pernet_exit(struct net *net) rtnl_lock(); for_each_rdev(rdev) { if (net_eq(wiphy_net(&rdev->wiphy), net)) - WARN_ON(cfg80211_switch_netns(rdev, &init_net)); + WARN_ON(__cfg80211_switch_netns(rdev, &init_net, true)); } rtnl_unlock(); } diff --git a/net/wireless/core.h b/net/wireless/core.h index b4610f6685dc..6138d207caf4 100644 --- a/net/wireless/core.h +++ b/net/wireless/core.h @@ -24,6 +24,16 @@ struct cfg80211_scan_request_int { struct cfg80211_scan_info info; bool notified; + /* + * set while the request is handed to the driver, i.e. between + * rdev_scan() and cfg80211_scan_done() + */ + bool driver_owns; + /* + * set when cfg80211 is done with the request but the driver still + * owns it, so that cfg80211_scan_done() knows to just free it + */ + bool stale; /* must be last - variable members */ struct cfg80211_scan_request req; }; @@ -104,6 +114,8 @@ struct cfg80211_registered_device { struct work_struct destroy_work; struct wiphy_work sched_scan_stop_wk; struct work_struct sched_scan_res_wk; + struct wiphy_work reg_check_chans_wk; + struct work_struct reg_leave_nan_wk; struct cfg80211_chan_def radar_chandef; struct work_struct propagate_radar_detect_wk; @@ -280,8 +292,7 @@ struct cfg80211_event { bool locally_generated; } dc; struct { - u8 bssid[ETH_ALEN]; - struct ieee80211_channel *channel; + struct cfg80211_bss *bss; } ij; struct { u8 peer_addr[ETH_ALEN]; @@ -344,8 +355,7 @@ int __cfg80211_join_ibss(struct cfg80211_registered_device *rdev, void cfg80211_clear_ibss(struct net_device *dev, bool nowext); int cfg80211_leave_ibss(struct cfg80211_registered_device *rdev, struct net_device *dev, bool nowext); -void __cfg80211_ibss_joined(struct net_device *dev, const u8 *bssid, - struct ieee80211_channel *channel); +void __cfg80211_ibss_joined(struct net_device *dev, struct cfg80211_bss *bss); int cfg80211_ibss_wext_join(struct cfg80211_registered_device *rdev, struct wireless_dev *wdev); diff --git a/net/wireless/ibss.c b/net/wireless/ibss.c index b1d748bdb504..7f6779d326b8 100644 --- a/net/wireless/ibss.c +++ b/net/wireless/ibss.c @@ -16,26 +16,18 @@ #include "rdev-ops.h" -void __cfg80211_ibss_joined(struct net_device *dev, const u8 *bssid, - struct ieee80211_channel *channel) +void __cfg80211_ibss_joined(struct net_device *dev, struct cfg80211_bss *bss) { struct wireless_dev *wdev = dev->ieee80211_ptr; - struct cfg80211_bss *bss; #ifdef CONFIG_CFG80211_WEXT union iwreq_data wrqu; #endif if (WARN_ON(wdev->iftype != NL80211_IFTYPE_ADHOC)) - return; + goto put_bss; if (!wdev->u.ibss.ssid_len) - return; - - bss = cfg80211_get_bss(wdev->wiphy, channel, bssid, NULL, 0, - IEEE80211_BSS_TYPE_IBSS, IEEE80211_PRIVACY_ANY); - - if (WARN_ON(!bss)) - return; + goto put_bss; if (wdev->u.ibss.current_bss) { cfg80211_unhold_bss(wdev->u.ibss.current_bss); @@ -43,17 +35,22 @@ void __cfg80211_ibss_joined(struct net_device *dev, const u8 *bssid, } cfg80211_hold_bss(bss_from_pub(bss)); + /* the reference from the event is transferred to current_bss */ wdev->u.ibss.current_bss = bss_from_pub(bss); cfg80211_upload_connect_keys(wdev); - nl80211_send_ibss_bssid(wiphy_to_rdev(wdev->wiphy), dev, bssid, + nl80211_send_ibss_bssid(wiphy_to_rdev(wdev->wiphy), dev, bss->bssid, GFP_KERNEL); #ifdef CONFIG_CFG80211_WEXT memset(&wrqu, 0, sizeof(wrqu)); - memcpy(wrqu.ap_addr.sa_data, bssid, ETH_ALEN); + memcpy(wrqu.ap_addr.sa_data, bss->bssid, ETH_ALEN); wireless_send_event(dev, SIOCGIWAP, &wrqu, NULL); #endif + return; + +put_bss: + cfg80211_put_bss(wdev->wiphy, bss); } void cfg80211_ibss_joined(struct net_device *dev, const u8 *bssid, @@ -62,6 +59,7 @@ void cfg80211_ibss_joined(struct net_device *dev, const u8 *bssid, struct wireless_dev *wdev = dev->ieee80211_ptr; struct cfg80211_registered_device *rdev = wiphy_to_rdev(wdev->wiphy); struct cfg80211_event *ev; + struct cfg80211_bss *bss; unsigned long flags; trace_cfg80211_ibss_joined(dev, bssid, channel); @@ -69,13 +67,19 @@ void cfg80211_ibss_joined(struct net_device *dev, const u8 *bssid, if (WARN_ON(!channel)) return; + bss = cfg80211_get_bss(wdev->wiphy, channel, bssid, NULL, 0, + IEEE80211_BSS_TYPE_IBSS, IEEE80211_PRIVACY_ANY); + if (WARN_ON(!bss)) + return; + ev = kzalloc_obj(*ev, gfp); - if (!ev) + if (!ev) { + cfg80211_put_bss(wdev->wiphy, bss); return; + } ev->type = EVENT_IBSS_JOINED; - memcpy(ev->ij.bssid, bssid, ETH_ALEN); - ev->ij.channel = channel; + ev->ij.bss = bss; spin_lock_irqsave(&wdev->event_lock, flags); list_add_tail(&ev->list, &wdev->event_list); diff --git a/net/wireless/nl80211.c b/net/wireless/nl80211.c index 899b6374c550..9fafd7673d85 100644 --- a/net/wireless/nl80211.c +++ b/net/wireless/nl80211.c @@ -8966,10 +8966,12 @@ int cfg80211_check_station_change(struct wiphy *wiphy, EXPORT_SYMBOL(cfg80211_check_station_change); /* - * Get vlan interface making sure it is running and on the right wiphy. + * Get vlan interface making sure it is running, on the right wiphy + * and actually belongs to the given AP/P2P_GO interface. */ static struct net_device *get_vlan(struct genl_info *info, - struct cfg80211_registered_device *rdev) + struct cfg80211_registered_device *rdev, + struct net_device *dev) { struct nlattr *vlanattr = info->attrs[NL80211_ATTR_STA_VLAN]; struct net_device *v; @@ -8999,6 +9001,12 @@ static struct net_device *get_vlan(struct genl_info *info, goto error; } + /* Check if the VLAN interface belongs to the AP interface */ + if (!dev || !ether_addr_equal(v->dev_addr, dev->dev_addr)) { + ret = -EINVAL; + goto error; + } + return v; error: dev_put(v); @@ -9296,7 +9304,7 @@ static int nl80211_set_station(struct sk_buff *skb, struct genl_info *info) if (err) return err; - params.vlan = get_vlan(info, rdev); + params.vlan = get_vlan(info, rdev, dev); if (IS_ERR(params.vlan)) return PTR_ERR(params.vlan); @@ -9328,7 +9336,7 @@ static int nl80211_set_station(struct sk_buff *skb, struct genl_info *info) static int nl80211_new_station(struct sk_buff *skb, struct genl_info *info) { struct cfg80211_registered_device *rdev = info->user_ptr[0]; - int err; + int err, link_id; struct wireless_dev *wdev = info->user_ptr[1]; struct net_device *dev = wdev->netdev; struct station_parameters params; @@ -9376,6 +9384,16 @@ static int nl80211_new_station(struct sk_buff *skb, struct genl_info *info) params.link_sta_params.link_id = nl80211_link_id_or_invalid(info->attrs); + if (wdev->valid_links) { + if (params.link_sta_params.link_id < 0) + return -EINVAL; + if (!(wdev->valid_links & BIT(params.link_sta_params.link_id))) + return -ENOLINK; + } else { + if (params.link_sta_params.link_id >= 0) + return -EINVAL; + } + if (info->attrs[NL80211_ATTR_MLD_ADDR]) { mac_addr = nla_data(info->attrs[NL80211_ATTR_MLD_ADDR]); params.link_sta_params.mld_mac = mac_addr; @@ -9556,8 +9574,12 @@ static int nl80211_new_station(struct sk_buff *skb, struct genl_info *info) switch (wdev->iftype) { case NL80211_IFTYPE_AP: - case NL80211_IFTYPE_AP_VLAN: case NL80211_IFTYPE_P2P_GO: + /* Add a new station only after the AP and link has been started */ + link_id = wdev->valid_links ? params.link_sta_params.link_id : 0; + if (!wdev->links[link_id].ap.beacon_interval) + return -ENETDOWN; + /* ignore WME attributes if iface/sta is not capable */ if (!(rdev->wiphy.flags & WIPHY_FLAG_AP_UAPSD) || !(params.sta_flags_set & BIT(NL80211_STA_FLAG_WME))) @@ -9597,11 +9619,24 @@ static int nl80211_new_station(struct sk_buff *skb, struct genl_info *info) } /* must be last in here for error handling */ - params.vlan = get_vlan(info, rdev); + params.vlan = get_vlan(info, rdev, dev); if (IS_ERR(params.vlan)) return PTR_ERR(params.vlan); break; case NL80211_IFTYPE_MESH_POINT: + /* + * Add a new station only after the mesh has been started. + * libertas doesn't implement join_mesh(); it configures the + * mesh via sysfs and joins it when the channel is set, so + * use that as the started indication instead. + */ + if (rdev->ops->libertas_set_mesh_channel) { + if (!wdev->u.mesh.chandef.chan) + return -ENETDOWN; + } else if (!wdev->u.mesh.beacon_interval) { + return -ENETDOWN; + } + /* ignore uAPSD data */ params.sta_modify_mask &= ~STATION_PARAM_APPLY_UAPSD; @@ -9649,27 +9684,10 @@ static int nl80211_new_station(struct sk_buff *skb, struct genl_info *info) /* be aware of params.vlan when changing code here */ - if (wdev->valid_links) { - if (params.link_sta_params.link_id < 0) { - err = -EINVAL; - goto out; - } - if (!(wdev->valid_links & BIT(params.link_sta_params.link_id))) { - err = -ENOLINK; - goto out; - } - } else { - if (params.link_sta_params.link_id >= 0) { - err = -EINVAL; - goto out; - } - } - params.epp_peer = nla_get_flag(info->attrs[NL80211_ATTR_EPP_PEER]); err = rdev_add_station(rdev, wdev, mac_addr, ¶ms); -out: dev_put(params.vlan); return err; } diff --git a/net/wireless/rdev-ops.h b/net/wireless/rdev-ops.h index 46849fe8d0b3..adcfd0278da3 100644 --- a/net/wireless/rdev-ops.h +++ b/net/wireless/rdev-ops.h @@ -464,7 +464,10 @@ static inline int rdev_scan(struct cfg80211_registered_device *rdev, return -EINVAL; trace_rdev_scan(&rdev->wiphy, request); + request->driver_owns = true; ret = rdev->ops->scan(&rdev->wiphy, &request->req); + if (ret) + request->driver_owns = false; trace_rdev_return_int(&rdev->wiphy, ret); return ret; } diff --git a/net/wireless/reg.c b/net/wireless/reg.c index a8336baf85dc..11665e0a7efc 100644 --- a/net/wireless/reg.c +++ b/net/wireless/reg.c @@ -2345,7 +2345,7 @@ static bool reg_wdev_chan_valid(struct wiphy *wiphy, struct wireless_dev *wdev) iftype = wdev->iftype; /* make sure the interface is active */ - if (!wdev->netdev || !netif_running(wdev->netdev)) + if (!wdev_running(wdev)) return true; /* NAN doesn't have links, handle it separately */ @@ -2446,19 +2446,52 @@ static bool reg_wdev_chan_valid(struct wiphy *wiphy, struct wireless_dev *wdev) return true; } -static void reg_leave_invalid_chans(struct wiphy *wiphy) +void reg_leave_invalid_nan_wk(struct work_struct *work) { + struct cfg80211_registered_device *rdev; struct wireless_dev *wdev; - struct cfg80211_registered_device *rdev = wiphy_to_rdev(wiphy); + + rdev = container_of(work, struct cfg80211_registered_device, + reg_leave_nan_wk); + + /* stopping NAN closes its data interfaces, which needs the RTNL */ + rtnl_lock(); list_for_each_entry(wdev, &rdev->wiphy.wdev_list, list) { bool valid; - scoped_guard(wiphy, wiphy) - valid = reg_wdev_chan_valid(wiphy, wdev); + if (wdev->iftype != NL80211_IFTYPE_NAN) + continue; + + scoped_guard(wiphy, &rdev->wiphy) + valid = reg_wdev_chan_valid(&rdev->wiphy, wdev); if (!valid) cfg80211_leave(rdev, wdev, -1); } + + rtnl_unlock(); +} + +void reg_leave_invalid_chans_wk(struct wiphy *wiphy, struct wiphy_work *work) +{ + struct cfg80211_registered_device *rdev = wiphy_to_rdev(wiphy); + struct wireless_dev *wdev; + + lockdep_assert_held(&wiphy->mtx); + + list_for_each_entry(wdev, &rdev->wiphy.wdev_list, list) { + if (reg_wdev_chan_valid(wiphy, wdev)) + continue; + + /* + * Tearing down NAN needs the RTNL for closing NAN_DATA + * interfaces, handle that separately. + */ + if (wdev->iftype == NL80211_IFTYPE_NAN) + schedule_work(&rdev->reg_leave_nan_wk); + else + cfg80211_leave_locked(rdev, wdev, -1); + } } static void reg_check_chans_work(struct work_struct *work) @@ -2466,12 +2499,13 @@ static void reg_check_chans_work(struct work_struct *work) struct cfg80211_registered_device *rdev; pr_debug("Verifying active interfaces after reg change\n"); - rtnl_lock(); - for_each_rdev(rdev) - reg_leave_invalid_chans(&rdev->wiphy); + rcu_read_lock(); - rtnl_unlock(); + list_for_each_entry_rcu(rdev, &cfg80211_rdev_list, list) + wiphy_work_queue(&rdev->wiphy, &rdev->reg_check_chans_wk); + + rcu_read_unlock(); } void reg_check_channels(void) diff --git a/net/wireless/reg.h b/net/wireless/reg.h index fc31c5f9a61a..c587079ead8f 100644 --- a/net/wireless/reg.h +++ b/net/wireless/reg.h @@ -178,6 +178,22 @@ int reg_reload_regdb(void); */ void reg_check_channels(void); +/** + * reg_leave_invalid_chans_wk - check if channels are no longer usable and leave + * @wiphy: the wiphy to check + * @work: the work struct + */ +void reg_leave_invalid_chans_wk(struct wiphy *wiphy, struct wiphy_work *work); + +/** + * reg_leave_invalid_nan_wk - check channels and tear down NAN when unusable + * @work: the work struct + * + * Stopping a NAN interface needs the RTNL, so it cannot be done from + * reg_leave_invalid_chans_wk() which runs with the wiphy mutex held. + */ +void reg_leave_invalid_nan_wk(struct work_struct *work); + extern const u8 shipped_regdb_certs[]; extern unsigned int shipped_regdb_certs_len; extern const u8 extra_regdb_certs[]; diff --git a/net/wireless/scan.c b/net/wireless/scan.c index 9e934b185e34..caa9c6495f20 100644 --- a/net/wireless/scan.c +++ b/net/wireless/scan.c @@ -1114,6 +1114,21 @@ int cfg80211_scan(struct cfg80211_registered_device *rdev) return 0; } +/* + * Release the scan request, but free it only if the driver is also done, + * e.g. mac80211 may cancel it asynchronously and still use it. + */ +static void cfg80211_put_scan_req(struct cfg80211_scan_request_int *req) +{ + if (!req) + return; + + if (req->driver_owns) + req->stale = true; + else + kfree(req); +} + void ___cfg80211_scan_done(struct cfg80211_registered_device *rdev, bool send_message) { @@ -1173,10 +1188,10 @@ void ___cfg80211_scan_done(struct cfg80211_registered_device *rdev, dev_put(wdev->netdev); - kfree(rdev->int_scan_req); + cfg80211_put_scan_req(rdev->int_scan_req); rdev->int_scan_req = NULL; - kfree(rdev->scan_req); + cfg80211_put_scan_req(rdev->scan_req); rdev->scan_req = NULL; if (!send_message) @@ -1199,6 +1214,18 @@ void cfg80211_scan_done(struct cfg80211_scan_request *request, struct cfg80211_scan_info old_info = intreq->info; trace_cfg80211_scan_done(intreq, info); + + intreq->driver_owns = false; + + if (intreq->stale) { + /* + * The scan is already completed as far as we're concerned, + * it was just kept around for the driver - done now, free it. + */ + kfree(intreq); + return; + } + WARN_ON(intreq != rdev->scan_req && intreq != rdev->int_scan_req); @@ -2050,6 +2077,13 @@ __cfg80211_bss_update(struct cfg80211_registered_device *rdev, if (!hidden) hidden = rb_find_bss(rdev, tmp, BSS_CMP_HIDE_NUL); + /* + * Only group with an entry with beacon data, otherwise + * beacon data can never be filled/updated. + */ + if (hidden && + !rcu_access_pointer(hidden->pub.beacon_ies)) + hidden = NULL; if (hidden) { new->pub.hidden_beacon_bss = &hidden->pub; list_add(&new->hidden_list, @@ -3468,11 +3502,6 @@ void cfg80211_update_assoc_bss_entry(struct wireless_dev *wdev, cbss->pub.channel = chan; list_for_each_entry(bss, &rdev->bss_list, list) { - if (!cfg80211_bss_type_match(bss->pub.capability, - bss->pub.channel->band, - wdev->conn_bss_type)) - continue; - if (bss == cbss) continue; diff --git a/net/wireless/util.c b/net/wireless/util.c index 3e584d0ca3e2..f2464d2ce0d5 100644 --- a/net/wireless/util.c +++ b/net/wireless/util.c @@ -1039,12 +1039,30 @@ unsigned int cfg80211_classify8021d(struct sk_buff *skb, } switch (skb->protocol) { - case htons(ETH_P_IP): - dscp = ipv4_get_dsfield(ip_hdr(skb)) & 0xfc; + case htons(ETH_P_IP): { + const struct iphdr *iph; + struct iphdr _iph; + + iph = skb_header_pointer(skb, sizeof(struct ethhdr), + sizeof(*iph), &_iph); + if (!iph) + return 0; + + dscp = ipv4_get_dsfield(iph) & 0xfc; break; - case htons(ETH_P_IPV6): - dscp = ipv6_get_dsfield(ipv6_hdr(skb)) & 0xfc; + } + case htons(ETH_P_IPV6): { + const struct ipv6hdr *ip6h; + struct ipv6hdr _ip6h; + + ip6h = skb_header_pointer(skb, sizeof(struct ethhdr), + sizeof(*ip6h), &_ip6h); + if (!ip6h) + return 0; + + dscp = ipv6_get_dsfield(ip6h) & 0xfc; break; + } case htons(ETH_P_MPLS_UC): case htons(ETH_P_MPLS_MC): { struct mpls_label mpls_tmp, *mpls; @@ -1217,8 +1235,7 @@ void cfg80211_process_wdev_events(struct wireless_dev *wdev) !ev->dc.locally_generated); break; case EVENT_IBSS_JOINED: - __cfg80211_ibss_joined(wdev->netdev, ev->ij.bssid, - ev->ij.channel); + __cfg80211_ibss_joined(wdev->netdev, ev->ij.bss); break; case EVENT_STOPPED: /* @@ -2477,16 +2494,15 @@ static void cfg80211_calculate_bi_data(struct wiphy *wiphy, u32 new_beacon_int, if (wdev->valid_links) continue; + wdev_bi = cfg80211_wdev_bi(wdev); + if (!wdev_bi) + continue; + /* skip wdevs not active on the given wiphy radio */ if (radio_idx >= 0 && !(rdev_get_radio_mask(rdev, wdev->netdev) & BIT(radio_idx))) continue; - wdev_bi = cfg80211_wdev_bi(wdev); - - if (!wdev_bi) - continue; - if (!*beacon_int_gcd) { *beacon_int_gcd = wdev_bi; continue; diff --git a/net/xfrm/espintcp.c b/net/xfrm/espintcp.c index 674aedc5af5a..3e72b9f067b9 100644 --- a/net/xfrm/espintcp.c +++ b/net/xfrm/espintcp.c @@ -30,7 +30,11 @@ static void handle_esp(struct sk_buff *skb, struct sock *sk) { struct tcp_skb_cb *tcp_cb = (struct tcp_skb_cb *)skb->cb; - skb_reset_transport_header(skb); + if (!skb_reset_transport_header_careful(skb)) { + XFRM_INC_STATS(sock_net(sk), LINUX_MIB_XFRMINERROR); + kfree_skb(skb); + return; + } /* restore IP CB, we need at least IP6CB->nhoff */ memmove(skb->cb, &tcp_cb->header, sizeof(tcp_cb->header)); diff --git a/net/xfrm/xfrm_input.c b/net/xfrm/xfrm_input.c index eecab337bd0a..5ed87d51392a 100644 --- a/net/xfrm/xfrm_input.c +++ b/net/xfrm/xfrm_input.c @@ -474,6 +474,7 @@ int xfrm_input(struct sk_buff *skb, int nexthdr, __be32 spi, int encap_type) struct xfrm_state *x = NULL; xfrm_address_t *daddr; u32 mark = skb->mark; + u8 xfrm_proto = nexthdr; unsigned int family = AF_UNSPEC; int decaps = 0; int async = 0; @@ -485,6 +486,7 @@ int xfrm_input(struct sk_buff *skb, int nexthdr, __be32 spi, int encap_type) if (encap_type < 0 || (xo && (xo->flags & XFRM_GRO || encap_type == 0 || encap_type == UDP_ENCAP_ESPINUDP))) { x = xfrm_input_state(skb); + xfrm_proto = x->type ? x->type->proto : nexthdr; if (unlikely(x->km.state != XFRM_STATE_VALID)) { if (x->km.state == XFRM_STATE_ACQ) @@ -592,11 +594,13 @@ int xfrm_input(struct sk_buff *skb, int nexthdr, __be32 spi, int encap_type) x = xfrm_input_state_lookup(net, mark, daddr, spi, nexthdr, family); if (x == NULL) { + xfrm_proto = nexthdr; secpath_reset(skb); XFRM_INC_STATS(net, LINUX_MIB_XFRMINNOSTATES); xfrm_audit_state_notfound(skb, family, spi, seq); goto drop; } + xfrm_proto = x->type ? x->type->proto : nexthdr; if (unlikely(x->dir && x->dir != XFRM_SA_DIR_IN)) { secpath_reset(skb); @@ -604,6 +608,7 @@ int xfrm_input(struct sk_buff *skb, int nexthdr, __be32 spi, int encap_type) xfrm_audit_state_notfound(skb, family, spi, seq); xfrm_state_put(x); x = NULL; + xfrm_proto = nexthdr; goto drop; } @@ -728,7 +733,7 @@ resume_decapped: } while (!err); rcu_read_lock(); - err = xfrm_rcv_cb(skb, family, x->type->proto, 0); + err = xfrm_rcv_cb(skb, family, xfrm_proto, 0); if (err) { rcu_read_unlock(); goto drop; @@ -753,7 +758,7 @@ resume_decapped: xfrm_gro = xo->flags & XFRM_GRO; err = -EAFNOSUPPORT; - afinfo = xfrm_state_afinfo_get_rcu(x->props.family); + afinfo = xfrm_state_afinfo_get_rcu(family); if (likely(afinfo)) err = afinfo->transport_finish(skb, xfrm_gro || async); if (xfrm_gro) { @@ -776,7 +781,7 @@ drop_unlock: drop: if (async) dev_put(dev); - xfrm_rcv_cb(skb, family, x && x->type ? x->type->proto : nexthdr, -1); + xfrm_rcv_cb(skb, family, xfrm_proto, -1); kfree_skb(skb); return 0; } @@ -800,12 +805,17 @@ static void xfrm_trans_reinject(struct work_struct *work) spin_unlock_bh(&trans->queue_lock); local_bh_disable(); + rcu_read_lock(); while ((skb = __skb_dequeue(&queue))) { struct net *net = XFRM_TRANS_SKB_CB(skb)->net; + struct net_device *dev = skb->dev; XFRM_TRANS_SKB_CB(skb)->finish(net, NULL, skb); + if (dev) + dev_put(dev); put_net(net); } + rcu_read_unlock(); local_bh_enable(); } @@ -821,12 +831,18 @@ int xfrm_trans_queue_net(struct net *net, struct sk_buff *skb, if (skb_queue_len(&trans->queue) >= READ_ONCE(net_hotdata.max_backlog)) return -ENOBUFS; + if (skb_dst(skb) && !skb_dst_force(skb)) + return -EHOSTUNREACH; + BUILD_BUG_ON(sizeof(struct xfrm_trans_cb) > sizeof(skb->cb)); hold_net = maybe_get_net(net); if (!hold_net) return -ENODEV; + if (skb->dev) + dev_hold(skb->dev); + XFRM_TRANS_SKB_CB(skb)->finish = finish; XFRM_TRANS_SKB_CB(skb)->net = hold_net; spin_lock_bh(&trans->queue_lock); diff --git a/net/xfrm/xfrm_iptfs.c b/net/xfrm/xfrm_iptfs.c index 597aedeac26e..6920940a35b4 100644 --- a/net/xfrm/xfrm_iptfs.c +++ b/net/xfrm/xfrm_iptfs.c @@ -416,6 +416,14 @@ static bool iptfs_skb_can_add_frags(const struct sk_buff *skb, if (skb_has_frag_list(skb) || skb->pp_recycle != walk->pp_recycle) return false; + /* Reject an @offset that is at or beyond the end of the walk's data + * before calling iptfs_skb_reset_frag_walk(), whose fragment-advance + * loop is otherwise unbounded and would index past walk->frags[]. + * This mirrors the guard already present in iptfs_skb_add_frags(). + */ + if (!walk->nr_frags || offset >= walk->total + walk->initial_offset) + return false; + /* Make offset relative to current frag after setting that */ offset = iptfs_skb_reset_frag_walk(walk, offset); @@ -820,8 +828,8 @@ static u32 iptfs_reassem_cont(struct xfrm_iptfs_data *xtfs, u64 seq, * allocate an in progress skb */ ipremain = __iptfs_iplen(xtfs->ra_runt); - if (ipremain < sizeof(xtfs->ra_runt)) { - /* length has to be at least runtsize large */ + if (ipremain < __iptfs_iphlen(xtfs->ra_runt)) { + /* length has to be at least the IP header size */ XFRM_INC_STATS(xs_net(xtfs->x), LINUX_MIB_XFRMINIPTFSERROR); goto abandon; diff --git a/net/xfrm/xfrm_policy.c b/net/xfrm/xfrm_policy.c index 932a313b9460..513c9f228334 100644 --- a/net/xfrm/xfrm_policy.c +++ b/net/xfrm/xfrm_policy.c @@ -2770,9 +2770,12 @@ static struct dst_entry *xfrm_bundle_create(struct xfrm_policy *policy, xdst0->path = dst; err = -ENODEV; - dev = dst->dev; - if (!dev) + rcu_read_lock(); + dev = dst_dev_rcu(dst); + if (!dev) { + rcu_read_unlock(); goto free_dst; + } xfrm_init_path(xdst0, dst, nfheader_len); xfrm_init_pmtu(bundle, nx); @@ -2780,8 +2783,10 @@ static struct dst_entry *xfrm_bundle_create(struct xfrm_policy *policy, for (xdst_prev = xdst0; xdst_prev != (struct xfrm_dst *)dst; xdst_prev = (struct xfrm_dst *) xfrm_dst_child(&xdst_prev->u.dst)) { err = xfrm_fill_dst(xdst_prev, dev, fl); - if (err) + if (err) { + rcu_read_unlock(); goto free_dst; + } xdst_prev->u.dst.header_len = header_len; xdst_prev->u.dst.trailer_len = trailer_len; @@ -2789,6 +2794,7 @@ static struct dst_entry *xfrm_bundle_create(struct xfrm_policy *policy, trailer_len -= xdst_prev->u.dst.xfrm->props.trailer_len; } + rcu_read_unlock(); return &xdst0->u.dst; put_states: @@ -3058,11 +3064,15 @@ static struct xfrm_dst *xfrm_create_dummy_bundle(struct net *net, xfrm_init_path((struct xfrm_dst *)dst1, dst, 0); err = -ENODEV; - dev = dst->dev; - if (!dev) + rcu_read_lock(); + dev = dst_dev_rcu(dst); + if (!dev) { + rcu_read_unlock(); goto free_dst; + } err = xfrm_fill_dst(xdst, dev, fl); + rcu_read_unlock(); if (err) goto free_dst; diff --git a/net/xfrm/xfrm_state.c b/net/xfrm/xfrm_state.c index 36a4f6793ede..e45aa1ed5b96 100644 --- a/net/xfrm/xfrm_state.c +++ b/net/xfrm/xfrm_state.c @@ -226,6 +226,7 @@ static struct xfrm_state_afinfo __rcu *xfrm_state_afinfo[NPROTO]; static DEFINE_SPINLOCK(xfrm_state_gc_lock); static DEFINE_SPINLOCK(xfrm_state_dev_gc_lock); +static DEFINE_MUTEX(xfrm_state_gc_mutex); int __xfrm_state_delete(struct xfrm_state *x); @@ -632,8 +633,10 @@ static void xfrm_state_gc_task(struct work_struct *work) synchronize_rcu(); + mutex_lock(&xfrm_state_gc_mutex); hlist_for_each_entry_safe(x, tmp, &gc_list, gclist) xfrm_state_gc_destroy(x); + mutex_unlock(&xfrm_state_gc_mutex); } static enum hrtimer_restart xfrm_timer_handler(struct hrtimer *me) @@ -823,9 +826,9 @@ int __xfrm_state_delete(struct xfrm_state *x) if (!hlist_unhashed(&x->byseq)) hlist_del_init_rcu(&x->byseq); if (!hlist_unhashed(&x->state_cache)) - hlist_del_rcu(&x->state_cache); + hlist_del_init_rcu(&x->state_cache); if (!hlist_unhashed(&x->state_cache_input)) - hlist_del_rcu(&x->state_cache_input); + hlist_del_init_rcu(&x->state_cache_input); if (!hlist_unhashed(&x->byspi)) hlist_del_init_rcu(&x->byspi); @@ -1000,6 +1003,7 @@ restart: out: spin_unlock_bh(&net->xfrm.xfrm_state_lock); + mutex_lock(&xfrm_state_gc_mutex); spin_lock_bh(&xfrm_state_dev_gc_lock); restart_gc: hlist_for_each_entry_safe(x, tmp, &xfrm_state_dev_gc_list, dev_gclist) { @@ -1014,6 +1018,7 @@ restart_gc: } spin_unlock_bh(&xfrm_state_dev_gc_lock); + mutex_unlock(&xfrm_state_gc_mutex); xfrm_flush_gc(); diff --git a/net/xfrm/xfrm_user.c b/net/xfrm/xfrm_user.c index 6266a92cf302..a2587c7e796b 100644 --- a/net/xfrm/xfrm_user.c +++ b/net/xfrm/xfrm_user.c @@ -1877,7 +1877,6 @@ static int xfrm_alloc_userspi(struct sk_buff *skb, struct nlmsghdr *nlh, struct net *net = sock_net(skb->sk); struct xfrm_state *x; struct xfrm_userspi_info *p; - struct xfrm_translator *xtr; struct sk_buff *resp_skb; xfrm_address_t *daddr; int family; @@ -1943,17 +1942,6 @@ static int xfrm_alloc_userspi(struct sk_buff *skb, struct nlmsghdr *nlh, goto out; } - xtr = xfrm_get_translator(); - if (xtr) { - err = xtr->alloc_compat(skb, nlmsg_hdr(skb)); - - xfrm_put_translator(xtr); - if (err) { - kfree_skb(resp_skb); - goto out; - } - } - err = nlmsg_unicast(xfrm_net_nlsk(net, skb), resp_skb, NETLINK_CB(skb).portid); out: @@ -3337,7 +3325,11 @@ static int xfrm_send_migrate_state(struct net *net, return err; } - return xfrm_nlmsg_multicast(net, skb, 0, XFRMNLGRP_MIGRATE); + rcu_read_lock(); + err = xfrm_nlmsg_multicast(net, skb, 0, XFRMNLGRP_MIGRATE); + rcu_read_unlock(); + + return err; } static int xfrm_do_migrate_state(struct sk_buff *skb, struct nlmsghdr *nlh, diff --git a/rust/kernel/net/phy.rs b/rust/kernel/net/phy.rs index 956cda573ddb..c4e7b1d6c6f4 100644 --- a/rust/kernel/net/phy.rs +++ b/rust/kernel/net/phy.rs @@ -123,39 +123,37 @@ impl Device { /// Gets the current link state. /// /// It returns true if the link is up. + #[inline] pub fn is_link_up(&self) -> bool { - const LINK_IS_UP: u64 = 1; - // TODO: the code to access to the bit field will be replaced with automatically - // generated code by bindgen when it becomes possible. - // SAFETY: The struct invariant ensures that we may access - // this field without additional synchronization. - let bit_field = unsafe { &(*self.0.get())._bitfield_1 }; - bit_field.get(14, 1) == LINK_IS_UP + let phydev = self.0.get().cast_const(); + // SAFETY: By the type invariant of `Device`, `phydev` points to a valid + // `struct phy_device`, and there is no concurrent write to this field. + let link = unsafe { bindings::phy_device::link_raw(phydev) }; + link == 1 } /// Gets the current auto-negotiation configuration. /// /// It returns true if auto-negotiation is enabled. + #[inline] pub fn is_autoneg_enabled(&self) -> bool { - // TODO: the code to access to the bit field will be replaced with automatically - // generated code by bindgen when it becomes possible. - // SAFETY: The struct invariant ensures that we may access - // this field without additional synchronization. - let bit_field = unsafe { &(*self.0.get())._bitfield_1 }; - bit_field.get(13, 1) == u64::from(bindings::AUTONEG_ENABLE) + let phydev = self.0.get().cast_const(); + // SAFETY: By the type invariant of `Device`, `phydev` points to a valid + // `struct phy_device`, and there is no concurrent write to this field. + let autoneg = unsafe { bindings::phy_device::autoneg_raw(phydev) }; + autoneg == bindings::AUTONEG_ENABLE } /// Gets the current auto-negotiation state. /// /// It returns true if auto-negotiation is completed. + #[inline] pub fn is_autoneg_completed(&self) -> bool { - const AUTONEG_COMPLETED: u64 = 1; - // TODO: the code to access to the bit field will be replaced with automatically - // generated code by bindgen when it becomes possible. - // SAFETY: The struct invariant ensures that we may access - // this field without additional synchronization. - let bit_field = unsafe { &(*self.0.get())._bitfield_1 }; - bit_field.get(15, 1) == AUTONEG_COMPLETED + let phydev = self.0.get().cast_const(); + // SAFETY: By the type invariant of `Device`, `phydev` points to a valid + // `struct phy_device`, and there is no concurrent write to this field. + let completed = unsafe { bindings::phy_device::autoneg_complete_raw(phydev) }; + completed == 1 } /// Sets the speed of the PHY. diff --git a/scripts/generate_rust_target.rs b/scripts/generate_rust_target.rs index 3bf296581a88..7687b0dd5474 100644 --- a/scripts/generate_rust_target.rs +++ b/scripts/generate_rust_target.rs @@ -224,6 +224,11 @@ fn main() { features += ",+harden-sls-ijmp"; features += ",+harden-sls-ret"; } + if cfg.has("X86_NATIVE_CPU") { + // Prevent the backend from generating APX instructions. The kernel is not yet prepared + // for general in-kernel EGPR use. + features += ",-apxf"; + } ts.push("features", features); ts.push("llvm-target", "x86_64-linux-gnu"); ts.push("supported-sanitizers", ["kcfi", "kernel-address"]); diff --git a/security/keys/encrypted-keys/encrypted.c b/security/keys/encrypted-keys/encrypted.c index 59cb77b237b3..e07092ea301a 100644 --- a/security/keys/encrypted-keys/encrypted.c +++ b/security/keys/encrypted-keys/encrypted.c @@ -19,6 +19,7 @@ #include <linux/parser.h> #include <linux/string.h> #include <linux/err.h> +#include <linux/overflow.h> #include <keys/user-type.h> #include <keys/trusted-type.h> #include <keys/encrypted-type.h> @@ -579,6 +580,7 @@ static struct encrypted_key_payload *encrypted_key_alloc(struct key *key, { struct encrypted_key_payload *epayload = NULL; unsigned short datablob_len; + unsigned short payload_totallen; unsigned short decrypted_datalen; unsigned short payload_datalen; unsigned int encrypted_datalen; @@ -632,16 +634,22 @@ static struct encrypted_key_payload *encrypted_key_alloc(struct key *key, encrypted_datalen = roundup(decrypted_datalen, blksize); - datablob_len = format_len + 1 + strlen(master_desc) + 1 - + strlen(datalen) + 1 + ivsize + 1 + encrypted_datalen; + if (check_add_overflow(format_len + 1 + strlen(master_desc) + 1 + + strlen(datalen) + 1 + ivsize + 1, + encrypted_datalen, &datablob_len)) + return ERR_PTR(-EINVAL); + + if (check_add_overflow(datablob_len, + payload_datalen + HASH_SIZE + 1, + &payload_totallen)) + return ERR_PTR(-EINVAL); - ret = key_payload_reserve(key, payload_datalen + datablob_len - + HASH_SIZE + 1); + ret = key_payload_reserve(key, payload_totallen); if (ret < 0) return ERR_PTR(ret); - epayload = kzalloc(sizeof(*epayload) + payload_datalen + - datablob_len + HASH_SIZE + 1, GFP_KERNEL); + epayload = kzalloc_flex(*epayload, payload_data, payload_totallen, + GFP_KERNEL); if (!epayload) return ERR_PTR(-ENOMEM); diff --git a/security/keys/gc.c b/security/keys/gc.c index 748e83818a76..eda445f815d4 100644 --- a/security/keys/gc.c +++ b/security/keys/gc.c @@ -318,9 +318,7 @@ maybe_resched: if (unlikely(gc_state & KEY_GC_REAPING_DEAD_3)) { kdebug("dead wake"); - smp_mb(); - clear_bit(KEY_GC_REAPING_KEYTYPE, &key_gc_flags); - wake_up_bit(&key_gc_flags, KEY_GC_REAPING_KEYTYPE); + clear_and_wake_up_bit(KEY_GC_REAPING_KEYTYPE, &key_gc_flags); } if (gc_state & KEY_GC_REAP_AGAIN) diff --git a/security/keys/request_key_auth.c b/security/keys/request_key_auth.c index 282e09d8fa46..ed6f55b9cdd9 100644 --- a/security/keys/request_key_auth.c +++ b/security/keys/request_key_auth.c @@ -9,6 +9,8 @@ #include <linux/sched.h> #include <linux/err.h> +#include <linux/pid.h> +#include <linux/proc_fs.h> #include <linux/seq_file.h> #include <linux/slab.h> #include <linux/uaccess.h> @@ -73,7 +75,10 @@ static void request_key_auth_describe(const struct key *key, seq_puts(m, "key:"); seq_puts(m, key->description); if (key_is_positive(key)) - seq_printf(m, " pid:%d ci:%zu", rka->pid, rka->callout_len); + seq_printf(m, " pid:%d ci:%zu", + pid_nr_ns(rka->pid, + proc_pid_ns(file_inode(m->file)->i_sb)), + rka->callout_len); } /* @@ -113,6 +118,7 @@ static void free_request_key_auth(struct request_key_auth *rka) if (rka->cred) put_cred(rka->cred); kfree(rka->callout_info); + put_pid(rka->pid); kfree(rka); } @@ -226,14 +232,14 @@ struct key *request_key_auth_new(struct key *target, const char *op, irka = cred->request_key_auth->payload.data[0]; rka->cred = get_cred(irka->cred); - rka->pid = irka->pid; + rka->pid = get_pid(irka->pid); up_read(&cred->request_key_auth->sem); } else { /* it isn't - use this process as the context */ rka->cred = get_cred(cred); - rka->pid = current->pid; + rka->pid = get_pid(task_pid(current)); } rka->target_key = key_get(target); diff --git a/security/keys/trusted-keys/trusted_tpm2.c b/security/keys/trusted-keys/trusted_tpm2.c index 67225dd562a9..01f18bb37047 100644 --- a/security/keys/trusted-keys/trusted_tpm2.c +++ b/security/keys/trusted-keys/trusted_tpm2.c @@ -99,7 +99,7 @@ struct tpm2_key_context { static int tpm2_key_decode(struct trusted_key_payload *payload, struct trusted_key_options *options, - u8 **buf) + u8 **buf, unsigned int *blob_len) { int ret; struct tpm2_key_context ctx; @@ -120,6 +120,7 @@ static int tpm2_key_decode(struct trusted_key_payload *payload, return -ENOMEM; *buf = blob; + *blob_len = ctx.priv_len + ctx.pub_len; options->keyhandle = ctx.parent; memcpy(blob, ctx.priv, ctx.priv_len); @@ -384,10 +385,11 @@ static int tpm2_load_cmd(struct tpm_chip *chip, int rc; u32 attrs; - rc = tpm2_key_decode(payload, options, &blob); + rc = tpm2_key_decode(payload, options, &blob, &blob_len); if (rc) { /* old form */ blob = payload->blob; + blob_len = payload->blob_len; payload->old_format = 1; } else { /* Bind for cleanup: */ @@ -399,17 +401,17 @@ static int tpm2_load_cmd(struct tpm_chip *chip, return -EINVAL; /* must be big enough for at least the two be16 size counts */ - if (payload->blob_len < 4) + if (blob_len < 4) return -EINVAL; private_len = get_unaligned_be16(blob); /* must be big enough for following public_len */ - if (private_len + 2 + 2 > (payload->blob_len)) + if (private_len + 2 + 2 > blob_len) return -E2BIG; public_len = get_unaligned_be16(blob + 2 + private_len); - if (private_len + 2 + public_len + 2 > payload->blob_len) + if (private_len + 2 + public_len + 2 > blob_len) return -E2BIG; pub = blob + 2 + private_len + 2; diff --git a/security/selinux/avc.c b/security/selinux/avc.c index a9401d6c2e5f..560a82c6d682 100644 --- a/security/selinux/avc.c +++ b/security/selinux/avc.c @@ -1149,8 +1149,11 @@ inline int avc_has_perm_noaudit(u32 ssid, u32 tsid, u32 denied; struct avc_node *node; - if (WARN_ON(!requested)) + if (WARN_ON(!requested)) { + /* Provide a deny-all, audit-all decision to the caller. */ + *avd = (struct av_decision){ .auditdeny = 0xffffffff }; return -EACCES; + } rcu_read_lock(); node = avc_lookup(ssid, tsid, tclass); diff --git a/security/selinux/hooks.c b/security/selinux/hooks.c index e5e17f100aae..3f4a6ddee322 100644 --- a/security/selinux/hooks.c +++ b/security/selinux/hooks.c @@ -1674,26 +1674,32 @@ static int cred_has_capability(const struct cred *cred, return rc; } -/* Check whether a task has a particular permission to an inode. - The 'adp' parameter is optional and allows other audit - data to be passed (e.g. the dentry). */ -static int inode_has_perm(const struct cred *cred, - struct inode *inode, - u32 perms, - struct common_audit_data *adp) +/* + * Check whether a SID has a particular permission to an inode. The 'adp' + * parameter is optional and allows other audit data to be passed (e.g. the + * dentry). + */ +static int inode_sid_has_perm(u32 sid, struct inode *inode, u32 perms, + struct common_audit_data *adp) { struct inode_security_struct *isec; - u32 sid; if (unlikely(IS_PRIVATE(inode))) return 0; - sid = cred_sid(cred); isec = selinux_inode(inode); return avc_has_perm(sid, isec->sid, isec->sclass, perms, adp); } +static int inode_has_perm(const struct cred *cred, + struct inode *inode, + u32 perms, + struct common_audit_data *adp) +{ + return inode_sid_has_perm(cred_sid(cred), inode, perms, adp); +} + /* Same as inode_has_perm, but pass explicit audit data containing the dentry to help the auditing code to more easily generate the pathname if needed. */ @@ -3843,17 +3849,74 @@ static int selinux_file_alloc_security(struct file *file) return 0; } +static inline u32 selinux_file_user_sid(const struct file *file) +{ + if (unlikely(file->f_mode & FMODE_BACKING)) + return selinux_backing_file(file)->uf_sid; + return selinux_file(file)->sid; +} + static int selinux_backing_file_alloc(struct file *backing_file, const struct file *user_file) { struct backing_file_security_struct *bfsec; + const struct backing_file_security_struct *ubfsec; + struct backing_file_security_layer *layer; + u32 i; bfsec = selinux_backing_file(backing_file); - bfsec->uf_sid = selinux_file(user_file)->sid; + bfsec->uf_sid = selinux_file_user_sid(user_file); + if (!(user_file->f_mode & FMODE_BACKING)) + return 0; + + ubfsec = selinux_backing_file(user_file); + /* a wrapped count would make kmalloc_array() return ZERO_SIZE_PTR */ + if (unlikely(ubfsec->layer_count == U32_MAX)) + return -EOVERFLOW; + + /* + * The final VMA only retains the lowest backing file, so record the + * whole chain here rather than in the mmap hook, where concurrent + * mappings would have to be serialized. Size it dynamically: erofs + * inode sharing adds a backing file without bumping s_stack_depth. + */ + bfsec->layers = kmalloc_array(ubfsec->layer_count + 1, + sizeof(*bfsec->layers), GFP_KERNEL); + if (!bfsec->layers) + return -ENOMEM; + + for (i = 0; i < ubfsec->layer_count; i++) { + layer = &bfsec->layers[i]; + *layer = ubfsec->layers[i]; + path_get(&layer->path); + } + + /* f_path, not file_user_path(): this layer, not the top-level file */ + layer = &bfsec->layers[i]; + layer->path = user_file->f_path; + layer->mounter_sid = cred_sid(user_file->f_cred); + layer->fd_sid = selinux_file(user_file)->sid; + path_get(&layer->path); + bfsec->layer_count = ubfsec->layer_count + 1; return 0; } +static void selinux_backing_file_free(struct file *backing_file) +{ + struct backing_file_security_struct *bfsec; + + /* security_backing_file_free() may be called twice after an error */ + if (!backing_file_security(backing_file)) + return; + + bfsec = selinux_backing_file(backing_file); + while (bfsec->layer_count) + path_put(&bfsec->layers[--bfsec->layer_count].path); + kfree(bfsec->layers); + bfsec->layers = NULL; +} + /* * Check whether a task has the ioctl permission and cmd * operation to an inode. @@ -3971,6 +4034,53 @@ static int selinux_file_ioctl_compat(struct file *file, unsigned int cmd, static int default_noexec __ro_after_init; +static u32 file_map_prot_to_av(unsigned long prot, bool shared) +{ + u32 av = FILE__READ; + + if (shared && (prot & PROT_WRITE)) + av |= FILE__WRITE; + if (prot & PROT_EXEC) + av |= FILE__EXECUTE; + + return av; +} + +static int backing_mounters_has_perm(const struct file *file, u32 av) +{ + const struct backing_file_security_struct *bfsec; + const struct backing_file_security_layer *layer; + struct common_audit_data ad; + struct inode *inode; + u32 i; + int rc; + + if (WARN_ON_ONCE(!(file->f_mode & FMODE_BACKING))) + return -EIO; + + bfsec = selinux_backing_file(file); + for (i = 0; i < bfsec->layer_count; i++) { + layer = &bfsec->layers[i]; + inode = d_inode(layer->path.dentry); + + ad.type = LSM_AUDIT_DATA_PATH; + ad.u.path = layer->path; + + if (layer->mounter_sid != layer->fd_sid) { + rc = avc_has_perm(layer->mounter_sid, layer->fd_sid, + SECCLASS_FD, FD__USE, &ad); + if (rc) + return rc; + } + + rc = inode_sid_has_perm(layer->mounter_sid, inode, av, &ad); + if (rc) + return rc; + } + + return 0; +} + static int __file_map_prot_check(const struct file *file, unsigned long prot, bool shared, bool mounter_check, bool bf_user_file) @@ -4004,14 +4114,10 @@ static int __file_map_prot_check(const struct file *file, unsigned long prot, if (file) { const struct cred *cred = mounter_check ? file->f_cred : current_cred(); - /* "read" always possible, "write" only if shared */ - u32 av = FILE__READ; - if (shared && prot_write) - av |= FILE__WRITE; - if (prot_exec) - av |= FILE__EXECUTE; - return __file_has_perm(cred, file, av, bf_user_file); + return __file_has_perm(cred, file, + file_map_prot_to_av(prot, shared), + bf_user_file); } return 0; @@ -4106,6 +4212,7 @@ static int selinux_file_mprotect(struct vm_area_struct *vma, int rc; const struct cred *cred = current_cred(); u32 sid = cred_sid(cred); + u32 av; const struct file *file = vma->vm_file; bool backing_file; bool shared = vma->vm_flags & VM_SHARED; @@ -4149,6 +4256,10 @@ static int selinux_file_mprotect(struct vm_area_struct *vma, if (rc) return rc; if (backing_file) { + rc = backing_mounters_has_perm(file, + FILE__EXECMOD); + if (rc) + return rc; rc = file_has_perm(file->f_cred, file, FILE__EXECMOD); if (rc) @@ -4161,6 +4272,10 @@ static int selinux_file_mprotect(struct vm_area_struct *vma, if (rc) return rc; if (backing_file) { + av = file_map_prot_to_av(prot, shared); + rc = backing_mounters_has_perm(file, av); + if (rc) + return rc; rc = file_map_prot_check(file, prot, shared, true); if (rc) return rc; @@ -7619,6 +7734,7 @@ static struct security_hook_list selinux_hooks[] __ro_after_init = { LSM_HOOK_INIT(file_permission, selinux_file_permission), LSM_HOOK_INIT(file_alloc_security, selinux_file_alloc_security), LSM_HOOK_INIT(backing_file_alloc, selinux_backing_file_alloc), + LSM_HOOK_INIT(backing_file_free, selinux_backing_file_free), LSM_HOOK_INIT(file_ioctl, selinux_file_ioctl), LSM_HOOK_INIT(file_ioctl_compat, selinux_file_ioctl_compat), LSM_HOOK_INIT(mmap_file, selinux_mmap_file), diff --git a/security/selinux/include/objsec.h b/security/selinux/include/objsec.h index 3c0a16ec978b..2f21568251ff 100644 --- a/security/selinux/include/objsec.h +++ b/security/selinux/include/objsec.h @@ -86,8 +86,16 @@ struct file_security_struct { u32 pseqno; /* Policy seqno at the time of file open */ }; +struct backing_file_security_layer { + struct path path; /* this layer's real path */ + u32 mounter_sid; /* SID of the mounter that opened it */ + u32 fd_sid; /* SID of its open file description */ +}; + struct backing_file_security_struct { - u32 uf_sid; /* associated user file fsec->sid */ + u32 uf_sid; /* top-level user file fsec->sid */ + u32 layer_count; /* number of intermediate backing files */ + struct backing_file_security_layer *layers; }; struct superblock_security_struct { diff --git a/sound/core/init.c b/sound/core/init.c index 9693e646b3bb..1bcb6a2e7550 100644 --- a/sound/core/init.c +++ b/sound/core/init.c @@ -310,7 +310,7 @@ static int snd_card_init(struct snd_card *card, struct device *parent, kfree(card); /* manually free here, as no destructor called */ return err; } - card->dev = parent; + card->dev = get_device(parent); card->number = idx; WARN_ON(IS_MODULE(CONFIG_SND) && !module); card->module = module; @@ -603,6 +603,7 @@ static int snd_card_do_free(struct snd_card *card) dev_warn(card->dev, "unable to free card info\n"); /* Not fatal error */ } + put_device(card->dev); if (card->release_completion) complete(card->release_completion); if (!managed) diff --git a/sound/core/pcm_timer.c b/sound/core/pcm_timer.c index ab0e5bd70f8f..18bedd66435d 100644 --- a/sound/core/pcm_timer.c +++ b/sound/core/pcm_timer.c @@ -111,12 +111,15 @@ void snd_pcm_timer_init(struct snd_pcm_substream *substream) snd_pcm_direction_name(substream->stream), tid.card, tid.device, tid.subdevice); timer->hw = snd_pcm_timer; + /* Set before registering: a concurrent reader can invoke our hw + * callbacks as soon as the timer is on the global list. + */ + timer->private_data = substream; + timer->private_free = snd_pcm_timer_free; if (snd_device_register(timer->card, timer) < 0) { snd_device_free(timer->card, timer); return; } - timer->private_data = substream; - timer->private_free = snd_pcm_timer_free; substream->timer = timer; } diff --git a/sound/hda/codecs/realtek/alc269.c b/sound/hda/codecs/realtek/alc269.c index 3abee617e86e..5127a111c59c 100644 --- a/sound/hda/codecs/realtek/alc269.c +++ b/sound/hda/codecs/realtek/alc269.c @@ -4303,6 +4303,7 @@ enum { ALC287_FIXUP_LEGION_16ACHG6, ALC287_FIXUP_CS35L41_I2C_2, ALC287_FIXUP_CS35L41_I2C_2_HP_GPIO_LED, + ALC287_FIXUP_CS35L41_I2C_2_HP_MUTE_LEDS, ALC287_FIXUP_CS35L41_I2C_4, ALC245_FIXUP_CS35L41_SPI_1, ALC245_FIXUP_CS35L41_SPI_2, @@ -6614,6 +6615,12 @@ static const struct hda_fixup alc269_fixups[] = { .chained = true, .chain_id = ALC285_FIXUP_HP_MUTE_LED, }, + [ALC287_FIXUP_CS35L41_I2C_2_HP_MUTE_LEDS] = { + .type = HDA_FIXUP_FUNC, + .v.func = cs35l41_fixup_i2c_two, + .chained = true, + .chain_id = ALC245_FIXUP_HP_X360_MUTE_LEDS, + }, [ALC287_FIXUP_CS35L41_I2C_4] = { .type = HDA_FIXUP_FUNC, .v.func = cs35l41_fixup_i2c_four, @@ -7358,6 +7365,7 @@ static const struct hda_quirk alc269_fixup_tbl[] = { SND_PCI_QUIRK(0x103c, 0x8158, "HP", ALC256_FIXUP_HP_HEADSET_MIC), SND_PCI_QUIRK(0x103c, 0x820d, "HP Pavilion 15", ALC295_FIXUP_HP_X360), SND_PCI_QUIRK(0x103c, 0x8256, "HP", ALC221_FIXUP_HP_FRONT_MIC), + SND_PCI_QUIRK(0x103c, 0x8259, "HP OMEN 15-ax202nf", ALC269_FIXUP_HP_MUTE_LED_MIC3), SND_PCI_QUIRK(0x103c, 0x827e, "HP x360", ALC295_FIXUP_HP_X360), SND_PCI_QUIRK(0x103c, 0x827f, "HP x360", ALC269_FIXUP_HP_MUTE_LED_MIC3), SND_PCI_QUIRK(0x103c, 0x82bf, "HP G3 mini", ALC221_FIXUP_HP_MIC_NO_PRESENCE), @@ -7568,6 +7576,7 @@ static const struct hda_quirk alc269_fixup_tbl[] = { SND_PCI_QUIRK(0x103c, 0x8b96, "HP", ALC236_FIXUP_HP_MUTE_LED_MICMUTE_VREF), SND_PCI_QUIRK(0x103c, 0x8b97, "HP", ALC236_FIXUP_HP_MUTE_LED_MICMUTE_VREF), SND_PCI_QUIRK(0x103c, 0x8ba9, "HP Omen 16-wd0xxx", ALC245_FIXUP_HP_MUTE_LED_V1_COEFBIT), + SND_PCI_QUIRK(0x103c, 0x8bb1, "HP Victus 15-fa1xxx (MB 8BB1)", ALC245_FIXUP_HP_MUTE_LED_COEFBIT), SND_PCI_QUIRK(0x103c, 0x8bb3, "HP Slim OMEN", ALC287_FIXUP_CS35L41_I2C_2), SND_PCI_QUIRK(0x103c, 0x8bb4, "HP Slim OMEN", ALC287_FIXUP_CS35L41_I2C_2), SND_PCI_QUIRK(0x103c, 0x8bb6, "HP Laptop 15-fd0039nt", ALC236_FIXUP_HP_15_FD0XXX), @@ -7660,7 +7669,7 @@ static const struct hda_quirk alc269_fixup_tbl[] = { SND_PCI_QUIRK(0x103c, 0x8d92, "HP ZBook Firefly 16 G12", ALC285_FIXUP_HP_GPIO_LED), SND_PCI_QUIRK(0x103c, 0x8dcd, "HP Victus 15-fa2xxx", ALC245_FIXUP_HP_MUTE_LED_COEFBIT), SND_PCI_QUIRK(0x103c, 0x8d9b, "HP 17 Turbine OmniBook 7 UMA", ALC287_FIXUP_CS35L41_I2C_2), - SND_PCI_QUIRK(0x103c, 0x8d9c, "HP 17 Turbine OmniBook 7 DIS", ALC287_FIXUP_CS35L41_I2C_2), + SND_PCI_QUIRK(0x103c, 0x8d9c, "HP 17 Turbine OmniBook 7 DIS", ALC287_FIXUP_CS35L41_I2C_2_HP_MUTE_LEDS), SND_PCI_QUIRK(0x103c, 0x8d9d, "HP 17 Turbine OmniBook X UMA", ALC287_FIXUP_CS35L41_I2C_2), SND_PCI_QUIRK(0x103c, 0x8d9e, "HP 17 Turbine OmniBook X DIS", ALC287_FIXUP_CS35L41_I2C_2), SND_PCI_QUIRK(0x103c, 0x8d9f, "HP 14 Cadet (x360)", ALC287_FIXUP_CS35L41_I2C_2), diff --git a/sound/hda/common/controller.c b/sound/hda/common/controller.c index afec5c5546ec..18dae022b324 100644 --- a/sound/hda/common/controller.c +++ b/sound/hda/common/controller.c @@ -586,11 +586,11 @@ static int azx_pcm_open(struct snd_pcm_substream *substream) snd_hda_codec_pcm_get(apcm->info); mutex_lock(&chip->open_mutex); azx_dev = azx_assign_device(chip, substream); - trace_azx_pcm_open(chip, azx_dev); if (azx_dev == NULL) { err = -EBUSY; goto unlock; } + trace_azx_pcm_open(chip, azx_dev); runtime->private_data = azx_dev; runtime->hw = azx_pcm_hw; diff --git a/sound/soc/amd/acp/acp-sdw-legacy-mach.c b/sound/soc/amd/acp/acp-sdw-legacy-mach.c index 6eac42bac855..1a05d4288a46 100644 --- a/sound/soc/amd/acp/acp-sdw-legacy-mach.c +++ b/sound/soc/amd/acp/acp-sdw-legacy-mach.c @@ -205,9 +205,19 @@ static int create_sdw_dailink(struct snd_soc_card *card, return -EINVAL; } + if (!soc_end->link_mask) { + dev_err(dev, "invalid zero link_mask\n"); + return -EINVAL; + } + if ((ffs(soc_end->link_mask) - 1) >= amd_ctx->max_sdw_links) { + dev_err(dev, "link_id %d exceeds max_sdw_links %d\n", + ffs(soc_end->link_mask) - 1, amd_ctx->max_sdw_links); + return -EINVAL; + } + switch (amd_ctx->acp_rev) { case ACP63_PCI_REV: - ret = get_acp63_cpu_pin_id(ffs(soc_end->link_mask - 1), + ret = get_acp63_cpu_pin_id(ffs(soc_end->link_mask) - 1, *be_id, &cpu_pin_id, dev); if (ret) return ret; @@ -215,7 +225,7 @@ static int create_sdw_dailink(struct snd_soc_card *card, case ACP70_PCI_REV: case ACP71_PCI_REV: case ACP72_PCI_REV: - ret = get_acp70_cpu_pin_id(ffs(soc_end->link_mask - 1), + ret = get_acp70_cpu_pin_id(ffs(soc_end->link_mask) - 1, *be_id, &cpu_pin_id, dev); if (ret) return ret; diff --git a/sound/soc/amd/acp/acp-sdw-sof-mach.c b/sound/soc/amd/acp/acp-sdw-sof-mach.c index a9cd1f335167..ec3e1f5f1052 100644 --- a/sound/soc/amd/acp/acp-sdw-sof-mach.c +++ b/sound/soc/amd/acp/acp-sdw-sof-mach.c @@ -121,9 +121,18 @@ static int create_sdw_dailink(struct snd_soc_card *card, return -EINVAL; } + if (!sof_end->link_mask) { + dev_err(dev, "invalid zero link_mask\n"); + return -EINVAL; + } + if ((ffs(sof_end->link_mask) - 1) >= amd_ctx->max_sdw_links) { + dev_err(dev, "link_id %d exceeds max_sdw_links %d\n", + ffs(sof_end->link_mask) - 1, amd_ctx->max_sdw_links); + return -EINVAL; + } switch (amd_ctx->acp_rev) { case ACP63_PCI_REV: - ret = get_acp63_cpu_pin_id(ffs(sof_end->link_mask - 1), + ret = get_acp63_cpu_pin_id(ffs(sof_end->link_mask) - 1, *be_id, &cpu_pin_id, dev); if (ret) return ret; @@ -131,7 +140,7 @@ static int create_sdw_dailink(struct snd_soc_card *card, case ACP70_PCI_REV: case ACP71_PCI_REV: case ACP72_PCI_REV: - ret = get_acp70_cpu_pin_id(ffs(sof_end->link_mask - 1), + ret = get_acp70_cpu_pin_id(ffs(sof_end->link_mask) - 1, *be_id, &cpu_pin_id, dev); if (ret) return ret; @@ -277,6 +286,7 @@ static int sof_card_dai_links_create(struct snd_soc_card *card) int num_devs = 0; int num_ends = 0; int num_aux = 0; + int num_confs; int num_links; int be_id = 0; int ret; @@ -287,6 +297,7 @@ static int sof_card_dai_links_create(struct snd_soc_card *card) return ret; } + num_confs = num_ends; /* One per DAI link, worst case is a DAI link for every endpoint */ struct asoc_sdw_dailink *sof_dais __free(kfree) = kzalloc_objs(*sof_dais, num_ends); @@ -303,7 +314,7 @@ static int sof_card_dai_links_create(struct snd_soc_card *card) if (!sof_aux) return -ENOMEM; - ret = asoc_sdw_parse_sdw_endpoints(dev, ctx, sof_aux, sof_dais, sof_ends, &num_devs); + ret = asoc_sdw_parse_sdw_endpoints(dev, ctx, sof_aux, sof_dais, sof_ends, &num_confs); if (ret < 0) return ret; @@ -315,7 +326,7 @@ static int sof_card_dai_links_create(struct snd_soc_card *card) dev_dbg(dev, "sdw %d, dmic %d", sdw_be_num, dmic_num); - codec_conf = devm_kcalloc(dev, num_devs, sizeof(*codec_conf), GFP_KERNEL); + codec_conf = devm_kcalloc(dev, num_confs, sizeof(*codec_conf), GFP_KERNEL); if (!codec_conf) return -ENOMEM; @@ -326,7 +337,7 @@ static int sof_card_dai_links_create(struct snd_soc_card *card) return -ENOMEM; card->codec_conf = codec_conf; - card->num_configs = num_devs; + card->num_configs = num_confs; card->dai_link = dai_links; card->num_links = num_links; card->aux_dev = sof_aux; @@ -379,7 +390,7 @@ static int mc_probe(struct platform_device *pdev) ctx->private = amd_ctx; card = &ctx->card; card->dev = &pdev->dev; - card->name = "amd-soundwire"; + card->name = "amd-sdw"; card->owner = THIS_MODULE; card->late_probe = asoc_sdw_card_late_probe; diff --git a/sound/soc/codecs/Kconfig b/sound/soc/codecs/Kconfig index f9a47e262a77..d88593c2bac8 100644 --- a/sound/soc/codecs/Kconfig +++ b/sound/soc/codecs/Kconfig @@ -524,13 +524,13 @@ config SND_SOC_ADAU1977 tristate config SND_SOC_ADAU1977_SPI - tristate + tristate "Analog Devices ADAU1977/ADAU1978/ADAU1979 CODEC - SPI" depends on SPI_MASTER select SND_SOC_ADAU1977 select REGMAP_SPI config SND_SOC_ADAU1977_I2C - tristate + tristate "Analog Devices ADAU1977/ADAU1978/ADAU1979 CODEC - I2C" depends on I2C select SND_SOC_ADAU1977 select REGMAP_I2C diff --git a/sound/soc/codecs/adau1977-i2c.c b/sound/soc/codecs/adau1977-i2c.c index d1c6c4ddf506..5a11cafdff36 100644 --- a/sound/soc/codecs/adau1977-i2c.c +++ b/sound/soc/codecs/adau1977-i2c.c @@ -34,9 +34,18 @@ static const struct i2c_device_id adau1977_i2c_ids[] = { }; MODULE_DEVICE_TABLE(i2c, adau1977_i2c_ids); +static const struct of_device_id adau1977_i2c_of_match[] = { + { .compatible = "adi,adau1977" }, + { .compatible = "adi,adau1978" }, + { .compatible = "adi,adau1979" }, + { }, +}; +MODULE_DEVICE_TABLE(of, adau1977_i2c_of_match); + static struct i2c_driver adau1977_i2c_driver = { .driver = { .name = "adau1977", + .of_match_table = adau1977_i2c_of_match, }, .probe = adau1977_i2c_probe, .id_table = adau1977_i2c_ids, diff --git a/sound/soc/codecs/adau1977-spi.c b/sound/soc/codecs/adau1977-spi.c index 878cde9d1014..c98da5ba9e9e 100644 --- a/sound/soc/codecs/adau1977-spi.c +++ b/sound/soc/codecs/adau1977-spi.c @@ -53,18 +53,18 @@ static const struct spi_device_id adau1977_spi_ids[] = { }; MODULE_DEVICE_TABLE(spi, adau1977_spi_ids); -static const struct of_device_id adau1977_spi_of_match[] __maybe_unused = { - { .compatible = "adi,adau1977" }, - { .compatible = "adi,adau1978" }, - { .compatible = "adi,adau1979" }, - { }, +static const struct of_device_id adau1977_spi_of_match[] = { + { .compatible = "adi,adau1977" }, + { .compatible = "adi,adau1978" }, + { .compatible = "adi,adau1979" }, + { }, }; MODULE_DEVICE_TABLE(of, adau1977_spi_of_match); static struct spi_driver adau1977_spi_driver = { .driver = { .name = "adau1977", - .of_match_table = of_match_ptr(adau1977_spi_of_match), + .of_match_table = adau1977_spi_of_match, }, .probe = adau1977_spi_probe, .id_table = adau1977_spi_ids, diff --git a/sound/soc/codecs/cs-amp-lib.c b/sound/soc/codecs/cs-amp-lib.c index 41a9a5b005c6..9bc19d2e1639 100644 --- a/sound/soc/codecs/cs-amp-lib.c +++ b/sound/soc/codecs/cs-amp-lib.c @@ -317,6 +317,8 @@ static void *cs_amp_alloc_get_efi_variable(efi_char16_t *name, unsigned long size = 0; status = cs_amp_get_efi_variable(name, guid, NULL, &size, NULL); + if (status == EFI_SUCCESS) + return ERR_PTR(-ENOENT); if (status != EFI_BUFFER_TOO_SMALL) return ERR_PTR(cs_amp_convert_efi_status(status)); diff --git a/sound/soc/codecs/hdmi-codec.c b/sound/soc/codecs/hdmi-codec.c index bc2c22436ba6..7aa50c5bd3df 100644 --- a/sound/soc/codecs/hdmi-codec.c +++ b/sound/soc/codecs/hdmi-codec.c @@ -426,10 +426,14 @@ static int hdmi_codec_iec958_default_put(struct snd_kcontrol *kcontrol, struct snd_soc_component *component = snd_kcontrol_chip(kcontrol); struct hdmi_codec_priv *hcp = snd_soc_component_get_drvdata(component); + if (!memcmp(hcp->iec_status, ucontrol->value.iec958.status, + sizeof(hcp->iec_status))) + return 0; + memcpy(hcp->iec_status, ucontrol->value.iec958.status, sizeof(hcp->iec_status)); - return 0; + return 1; } static int hdmi_codec_iec958_mask_get(struct snd_kcontrol *kcontrol, diff --git a/sound/soc/codecs/rt712-sdca-dmic.c b/sound/soc/codecs/rt712-sdca-dmic.c index 8860d81134e7..a9f3aa4e143a 100644 --- a/sound/soc/codecs/rt712-sdca-dmic.c +++ b/sound/soc/codecs/rt712-sdca-dmic.c @@ -13,6 +13,7 @@ #include <sound/core.h> #include <sound/pcm.h> #include <sound/pcm_params.h> +#include <sound/sdw.h> #include <sound/tlv.h> #include "rt712-sdca.h" #include "rt712-sdca-dmic.h" @@ -632,10 +633,10 @@ static int rt712_sdca_dmic_hw_params(struct snd_pcm_substream *substream, { struct snd_soc_component *component = dai->component; struct rt712_sdca_dmic_priv *rt712 = snd_soc_component_get_drvdata(component); - struct sdw_stream_config stream_config; + struct sdw_stream_config stream_config = {0}; struct sdw_port_config port_config; struct sdw_stream_runtime *sdw_stream; - int retval, num_channels; + int retval; unsigned int sampling_rate; dev_dbg(dai->dev, "%s %s", __func__, dai->name); @@ -647,13 +648,8 @@ static int rt712_sdca_dmic_hw_params(struct snd_pcm_substream *substream, if (!rt712->slave) return -EINVAL; - stream_config.frame_rate = params_rate(params); - stream_config.ch_count = params_channels(params); - stream_config.bps = snd_pcm_format_width(params_format(params)); - stream_config.direction = SDW_DATA_DIR_TX; - - num_channels = params_channels(params); - port_config.ch_mask = GENMASK(num_channels - 1, 0); + /* SoundWire specific configuration */ + snd_sdw_params_to_config(substream, params, &stream_config, &port_config); port_config.num = 2; retval = sdw_stream_add_slave(rt712->slave, &stream_config, diff --git a/sound/soc/codecs/rt712-sdca-sdw.c b/sound/soc/codecs/rt712-sdca-sdw.c index c50e74e20a88..edba0367d9ce 100644 --- a/sound/soc/codecs/rt712-sdca-sdw.c +++ b/sound/soc/codecs/rt712-sdca-sdw.c @@ -18,12 +18,16 @@ static bool rt712_sdca_readable_register(struct device *dev, unsigned int reg) { switch (reg) { + case 0x004d: case 0x201a ... 0x201f: case 0x2029 ... 0x202a: case 0x202d ... 0x2034: case 0x2230 ... 0x2232: case 0x2f01 ... 0x2f0a: case 0x2f35 ... 0x2f36: + case 0x2f3a: + case 0x2f3d: + case 0x2f41: case 0x2f50: case 0x2f54: case 0x2f58 ... 0x2f5d: @@ -48,6 +52,7 @@ static bool rt712_sdca_readable_register(struct device *dev, unsigned int reg) static bool rt712_sdca_volatile_register(struct device *dev, unsigned int reg) { switch (reg) { + case 0x004d: case 0x201b: case 0x201c: case 0x201d: @@ -56,6 +61,9 @@ static bool rt712_sdca_volatile_register(struct device *dev, unsigned int reg) case 0x2230: case 0x2f01: case 0x2f35: + case 0x2f3a: + case 0x2f3d: + case 0x2f41: case 0x320c: case SDW_SDCA_CTL(FUNC_NUM_JACK_CODEC, RT712_SDCA_ENT_GE49, RT712_SDCA_CTL_DETECTED_MODE, 0): case SDW_SDCA_CTL(FUNC_NUM_HID, RT712_SDCA_ENT_HID01, RT712_SDCA_CTL_HIDTX_CURRENT_OWNER, 0) ... diff --git a/sound/soc/codecs/rt712-sdca.c b/sound/soc/codecs/rt712-sdca.c index eda87eb9ab66..38052cb19790 100644 --- a/sound/soc/codecs/rt712-sdca.c +++ b/sound/soc/codecs/rt712-sdca.c @@ -73,14 +73,57 @@ static int rt712_sdca_index_update_bits(struct rt712_sdca_priv *rt712, return rt712_sdca_index_write(rt712, nid, reg, tmp); } +static void rt712_sdca_clk_patch(struct rt712_sdca_priv *rt712) +{ + rt712_sdca_index_write(rt712, RT712_VENDOR_REG, 0x65, 0x0000); + regmap_write(rt712->regmap, RT712_SDW_ROOT_CLK, 0x03); + usleep_range(1000, 1100); + regmap_write(rt712->regmap, RT712_SDW_ROOT_CLK, 0x02); + usleep_range(1000, 1100); + regmap_update_bits(rt712->regmap, RT712_PLL2_CONF2, 0x0080, 0x0000); + regmap_update_bits(rt712->regmap, RT712_PLL2_CONF2, 0x001f, 0x0017); + regmap_update_bits(rt712->regmap, RT712_PLL2_CONF3, 0x0010, 0x0000); + regmap_update_bits(rt712->regmap, RT712_PLL2_CONF1, 0x0081, 0x0001); + regmap_write(rt712->regmap, RT712_SDW_ROOT_CLK, 0x03); + usleep_range(1000, 1100); + regmap_update_bits(rt712->regmap, RT712_PLL2_CONF1, 0x0081, 0x0081); + regmap_update_bits(rt712->regmap, RT712_PLL2_CONF2, 0x0080, 0x0080); + regmap_update_bits(rt712->regmap, RT712_PLL2_CONF2, 0x001f, 0x0000); + regmap_update_bits(rt712->regmap, RT712_PLL2_CONF3, 0x0010, 0x0010); + usleep_range(1000, 1100); + rt712_sdca_index_write(rt712, RT712_VENDOR_REG, 0x65, 0x0081); +} + +static void rt712_sdca_clk_patch2(struct rt712_sdca_priv *rt712) +{ + rt712_sdca_index_update_bits(rt712, RT712_VENDOR_REG, 0x49, 0x0800, + 0x0000); + rt712_sdca_index_update_bits(rt712, RT712_VENDOR_REG, 0x49, 0xf000, + 0x0000); + rt712_sdca_index_write(rt712, RT712_VENDOR_REG, 0x65, 0x0000); + rt712_sdca_index_update_bits(rt712, RT712_VENDOR_ANALOG_CTL, 0x0c, 0xc000, + 0xc000); + rt712_sdca_index_update_bits(rt712, RT712_VENDOR_ANALOG_CTL, 0x00, 0xc000, + 0xc000); + rt712_sdca_index_write(rt712, RT712_VENDOR_REG, 0x65, 0x0081); + regmap_write(rt712->regmap, RT712_SDW_ROOT_CLK, 0x02); + usleep_range(1000, 1100); + regmap_write(rt712->regmap, RT712_SDW_ROOT_CLK, 0x03); + usleep_range(1000, 1100); + rt712_sdca_index_write(rt712, RT712_VENDOR_REG, 0x65, 0x0000); +} + static int rt712_sdca_calibration(struct rt712_sdca_priv *rt712) { unsigned int val, loop_rc = 0, loop_dc = 0; struct device *dev; struct regmap *regmap = rt712->regmap; + unsigned int clk_base; int chk_cnt = 100; int ret = 0; + regmap_read(rt712->regmap, RT712_SDW_ROOT_CLK, &clk_base); + mutex_lock(&rt712->calibrate_mutex); dev = regmap_get_device(regmap); @@ -109,8 +152,35 @@ static int rt712_sdca_calibration(struct rt712_sdca_priv *rt712) if (ret < 0) goto _cali_fail_; } - if (loop_dc == chk_cnt) - dev_err(dev, "%s, calibration time-out!\n", __func__); + + if (loop_dc == chk_cnt) { + if (clk_base == RT712_CLK_FREQ_24_576MHZ) { + rt712_sdca_clk_patch(rt712); + rt712_sdca_clk_patch2(rt712); + } + rt712_sdca_index_write(rt712, RT712_VENDOR_REG, RT712_FSM_CTL, 0x4100); + rt712_sdca_index_write(rt712, RT712_VENDOR_CALI, + RT712_DAC_DC_CALI_CTL1, 0x7883); + rt712_sdca_index_write(rt712, RT712_VENDOR_CALI, + RT712_DAC_DC_CALI_CTL1, 0xf893); + rt712_sdca_index_read(rt712, RT712_VENDOR_CALI, + RT712_DAC_DC_CALI_CTL1, &val); + + for (loop_dc = 0; loop_dc < chk_cnt && + (val & RT712_DAC_DC_CALI_TRIGGER); loop_dc++) { + usleep_range(10000, 11000); + ret = rt712_sdca_index_read(rt712, RT712_VENDOR_CALI, + RT712_DAC_DC_CALI_CTL1, &val); + + if (ret < 0) + goto _cali_fail_; + } + + if (loop_dc == chk_cnt) + dev_err(dev, "%s, calibration time-out!\n", __func__); + else + dev_dbg(dev, "%s, calibration success!\n", __func__); + } if (loop_dc == chk_cnt || loop_rc == chk_cnt) ret = -ETIMEDOUT; @@ -1759,9 +1829,13 @@ static void rt712_sdca_va_io_init(struct rt712_sdca_priv *rt712) static void rt712_sdca_vb_io_init(struct rt712_sdca_priv *rt712) { - int ret = 0; unsigned int jack_func_status, mic_func_status, amp_func_status; struct device *dev = &rt712->slave->dev; + unsigned int clk_base; + int ret = 0; + + regmap_read(rt712->regmap, RT712_SDW_ROOT_CLK, &clk_base); + dev_dbg(dev, "%s clk_base=%x", __func__, clk_base); regmap_read(rt712->regmap, SDW_SDCA_CTL(FUNC_NUM_JACK_CODEC, RT712_SDCA_ENT0, RT712_SDCA_CTL_FUNC_STATUS, 0), &jack_func_status); @@ -1773,6 +1847,12 @@ static void rt712_sdca_vb_io_init(struct rt712_sdca_priv *rt712) __func__, jack_func_status, mic_func_status, amp_func_status); rt712_sdca_index_write(rt712, RT712_VENDOR_REG, RT712_JD_CTL3, 0x7778); + + if (clk_base == RT712_CLK_FREQ_24_576MHZ) { + rt712_sdca_clk_patch(rt712); + rt712_sdca_clk_patch2(rt712); + } + /* DMIC */ if ((mic_func_status & FUNCTION_NEEDS_INITIALIZATION) || (!rt712->first_hw_init)) { rt712_sdca_index_write(rt712, RT712_VENDOR_HDA_CTL, RT712_DMIC2_FU_IT_FLOAT_CTL, 0x1526); diff --git a/sound/soc/codecs/rt712-sdca.h b/sound/soc/codecs/rt712-sdca.h index 46740281a5c1..6229fe341bb5 100644 --- a/sound/soc/codecs/rt712-sdca.h +++ b/sound/soc/codecs/rt712-sdca.h @@ -162,6 +162,16 @@ struct rt712_dmic_kctrl_priv { #define RT712_EAPD_HIGH 0x2 #define RT712_EAPD_LOW 0x0 +/* SDW clock root frequency */ +#define RT712_SDW_ROOT_CLK 0x004d +#define RT712_SDW_SCALE_CLK0 0x0062 +#define RT712_SDW_SCALE_CLK1 0x0072 + +/* PLL2 config */ +#define RT712_PLL2_CONF1 0x2f3a +#define RT712_PLL2_CONF2 0x2f3d +#define RT712_PLL2_CONF3 0x2f41 + /* RC Calibration register */ #define RT712_RC_CAL 0x3201 @@ -254,6 +264,14 @@ enum rt712_sdca_version { RT712_VB, }; +enum { + RT712_CLK_FREQ_19_2_MHZ = 1, + RT712_CLK_FREQ_24MHZ = 2, + RT712_CLK_FREQ_24_576MHZ = 3, + RT712_CLK_FREQ_22_5792MHZ = 4, +}; + + int rt712_sdca_io_init(struct device *dev, struct sdw_slave *slave); int rt712_sdca_init(struct device *dev, struct regmap *regmap, struct regmap *mbq_regmap, struct sdw_slave *slave); diff --git a/sound/soc/codecs/rt721-sdca-sdw.c b/sound/soc/codecs/rt721-sdca-sdw.c index eae7d662efae..910583162d3e 100644 --- a/sound/soc/codecs/rt721-sdca-sdw.c +++ b/sound/soc/codecs/rt721-sdca-sdw.c @@ -70,6 +70,7 @@ static bool rt721_sdca_mbq_readable_register(struct device *dev, unsigned int re case 0x0310100: case 0x2000000 ... 0x2000003: case 0x2000013: + case 0x2000026: case 0x200002c: case 0x200003c: case 0x2000046: @@ -142,6 +143,7 @@ static bool rt721_sdca_mbq_volatile_register(struct device *dev, unsigned int re case 0x200000d: case 0x2000019: case 0x2000020: + case 0x2000026: case 0x200002c: case 0x2000030: case 0x2000046: @@ -155,6 +157,7 @@ static bool rt721_sdca_mbq_volatile_register(struct device *dev, unsigned int re case 0x5810039: case 0x5b10018: case 0x5b10019: + case 0x6100006: return true; default: return false; diff --git a/sound/soc/codecs/rt721-sdca.c b/sound/soc/codecs/rt721-sdca.c index a9479d0e4941..738644018fb9 100644 --- a/sound/soc/codecs/rt721-sdca.c +++ b/sound/soc/codecs/rt721-sdca.c @@ -1497,6 +1497,15 @@ int rt721_sdca_init(struct device *dev, struct regmap *regmap, &soc_sdca_dev_rt721, rt721_sdca_dai, ARRAY_SIZE(rt721_sdca_dai)); } +static void rt721_sdca_reset(struct rt721_sdca_priv *rt721) +{ + rt_sdca_index_update_bits(rt721->mbq_regmap, RT721_VENDOR_REG, + RT721_VD_HIDDEN_CTRL, RT721_HIDDEN_REG_SW_RESET, + RT721_HIDDEN_REG_SW_RESET); + rt_sdca_index_update_bits(rt721->mbq_regmap, RT721_HDA_SDCA_FLOAT, + RT721_HDA_LEGACY_RESET_CTL, 0x1, 0x1); +} + int rt721_sdca_io_init(struct device *dev, struct sdw_slave *slave) { struct rt721_sdca_priv *rt721 = dev_get_drvdata(dev); @@ -1530,9 +1539,17 @@ int rt721_sdca_io_init(struct device *dev, struct sdw_slave *slave) } pm_runtime_get_noresume(&slave->dev); + + if (!rt721->first_hw_init) + rt721_sdca_reset(rt721); + rt721_sdca_dmic_preset(rt721); rt721_sdca_amp_preset(rt721); rt721_sdca_jack_preset(rt721); + + if (rt721->hs_jack && (!rt721->first_hw_init)) + rt721_sdca_jack_init(rt721); + if (rt721->first_hw_init) { regcache_cache_bypass(rt721->regmap, false); regcache_mark_dirty(rt721->regmap); diff --git a/sound/soc/codecs/wm_adsp.c b/sound/soc/codecs/wm_adsp.c index 90c24c4b318e..b8f3035af3ba 100644 --- a/sound/soc/codecs/wm_adsp.c +++ b/sound/soc/codecs/wm_adsp.c @@ -775,9 +775,12 @@ static int wm_adsp_request_firmware_file(struct wm_adsp *dsp, s++; } + adsp_dbg(dsp, "Try '%s'\n", fw->filename); ret = wm_adsp_firmware_request(&fw->firmware, fw->filename, cs_dsp->dev); if (ret < 0) { - adsp_dbg(dsp, "Failed to request '%s': %d\n", fw->filename, ret); + if (ret != -ENOENT) + adsp_dbg(dsp, "Failed to request '%s': %d\n", fw->filename, ret); + kfree(fw->filename); fw->filename = NULL; if (ret != -ENOENT) diff --git a/sound/soc/intel/boards/sof_es8336.c b/sound/soc/intel/boards/sof_es8336.c index 9b016136c639..f1e62c2e79fe 100644 --- a/sound/soc/intel/boards/sof_es8336.c +++ b/sound/soc/intel/boards/sof_es8336.c @@ -360,6 +360,15 @@ static const struct dmi_system_id sof_es8336_quirk_table[] = { .driver_data = (void *)(SOF_ES8336_HEADPHONE_GPIO | SOC_ES8336_HEADSET_MIC1) }, + { + .callback = sof_es8336_quirk_cb, + .matches = { + DMI_MATCH(DMI_SYS_VENDOR, "HUAWEI"), + DMI_MATCH(DMI_PRODUCT_NAME, "NDZ-WXX9"), + }, + .driver_data = (void *)(SOF_ES8336_HEADPHONE_GPIO | + SOC_ES8336_HEADSET_MIC1) + }, {} }; diff --git a/sound/soc/sdw_utils/soc_sdw_cs_amp.c b/sound/soc/sdw_utils/soc_sdw_cs_amp.c index 325ab7230481..6e21ef8f87e2 100644 --- a/sound/soc/sdw_utils/soc_sdw_cs_amp.c +++ b/sound/soc/sdw_utils/soc_sdw_cs_amp.c @@ -14,7 +14,6 @@ #include <sound/soc-dai.h> #include <sound/soc_sdw_utils.h> -#define CS_AMP_CHANNELS_PER_AMP 4 #define CS35L56_SPK_VOLUME_0DB 400 /* 0dB Max */ int asoc_sdw_cs35l56_volume_limit(struct snd_soc_card *card, const char *name_prefix) @@ -64,51 +63,6 @@ int asoc_sdw_cs_spk_rtd_init(struct snd_soc_pcm_runtime *rtd, struct snd_soc_dai } EXPORT_SYMBOL_NS(asoc_sdw_cs_spk_rtd_init, "SND_SOC_SDW_UTILS"); -int asoc_sdw_cs_spk_feedback_rtd_init(struct snd_soc_pcm_runtime *rtd, struct snd_soc_dai *dai) -{ - const struct snd_soc_dai_link *dai_link = rtd->dai_link; - const struct snd_soc_dai_link_ch_map *ch_map; - const struct snd_soc_dai_link_component *codec_dlc; - struct snd_soc_dai *codec_dai; - u8 ch_slot[8] = {}; - unsigned int amps_per_bus, ch_per_amp, mask; - int i, ret; - - WARN_ON(dai_link->num_cpus > ARRAY_SIZE(ch_slot)); - - /* - * CS35L56 has 4 TX channels. When the capture is aggregated the - * same bus slots will be allocated to all the amps on a bus. Only - * one amp on that bus can be transmitting in each slot so divide - * the available 4 slots between all the amps on a bus. - */ - amps_per_bus = dai_link->num_codecs / dai_link->num_cpus; - if ((amps_per_bus == 0) || (amps_per_bus > CS_AMP_CHANNELS_PER_AMP)) { - dev_err(rtd->card->dev, "Illegal num_codecs:%u / num_cpus:%u\n", - dai_link->num_codecs, dai_link->num_cpus); - return -EINVAL; - } - - ch_per_amp = CS_AMP_CHANNELS_PER_AMP / amps_per_bus; - - for_each_rtd_ch_maps(rtd, i, ch_map) { - codec_dlc = snd_soc_link_to_codec(rtd->dai_link, i); - codec_dai = snd_soc_find_dai(codec_dlc); - mask = GENMASK(ch_per_amp - 1, 0) << ch_slot[ch_map->cpu]; - - ret = snd_soc_dai_set_tdm_slot(codec_dai, 0, mask, 4, 32); - if (ret < 0) { - dev_err(rtd->card->dev, "Failed to set TDM slot:%d\n", ret); - return ret; - } - - ch_slot[ch_map->cpu] += ch_per_amp; - } - - return 0; -} -EXPORT_SYMBOL_NS(asoc_sdw_cs_spk_feedback_rtd_init, "SND_SOC_SDW_UTILS"); - int asoc_sdw_cs_amp_init(struct snd_soc_card *card, struct snd_soc_dai_link *dai_links, struct asoc_sdw_codec_info *info, diff --git a/sound/soc/sdw_utils/soc_sdw_utils.c b/sound/soc/sdw_utils/soc_sdw_utils.c index a66dcc02fb59..d2eeef4931c6 100644 --- a/sound/soc/sdw_utils/soc_sdw_utils.c +++ b/sound/soc/sdw_utils/soc_sdw_utils.c @@ -811,7 +811,6 @@ struct asoc_sdw_codec_info codec_info_list[] = { .dai_name = "cs35l56-sdw1c", .dai_type = SOC_SDW_DAI_TYPE_AMP, .dailink = {SOC_SDW_UNUSED_DAI_ID, SOC_SDW_AMP_IN_DAI_ID}, - .rtd_init = asoc_sdw_cs_spk_feedback_rtd_init, }, }, .dai_num = 2, @@ -840,7 +839,6 @@ struct asoc_sdw_codec_info codec_info_list[] = { .dai_name = "cs35l56-sdw1c", .dai_type = SOC_SDW_DAI_TYPE_AMP, .dailink = {SOC_SDW_UNUSED_DAI_ID, SOC_SDW_AMP_IN_DAI_ID}, - .rtd_init = asoc_sdw_cs_spk_feedback_rtd_init, }, }, .dai_num = 2, @@ -869,7 +867,6 @@ struct asoc_sdw_codec_info codec_info_list[] = { .dai_name = "cs35l56-sdw1c", .dai_type = SOC_SDW_DAI_TYPE_AMP, .dailink = {SOC_SDW_UNUSED_DAI_ID, SOC_SDW_AMP_IN_DAI_ID}, - .rtd_init = asoc_sdw_cs_spk_feedback_rtd_init, }, }, .dai_num = 2, @@ -898,7 +895,6 @@ struct asoc_sdw_codec_info codec_info_list[] = { .dai_name = "cs35l56-sdw1c", .dai_type = SOC_SDW_DAI_TYPE_AMP, .dailink = {SOC_SDW_UNUSED_DAI_ID, SOC_SDW_AMP_IN_DAI_ID}, - .rtd_init = asoc_sdw_cs_spk_feedback_rtd_init, }, }, .dai_num = 2, @@ -1565,7 +1561,7 @@ int asoc_sdw_hw_params(struct snd_pcm_substream *substream, struct snd_soc_pcm_runtime *rtd = snd_soc_substream_to_rtd(substream); struct snd_soc_dai_link_ch_map *ch_maps; int ch = params_channels(params); - unsigned int ch_mask; + unsigned int cpu_ch_mask, codec_ch_mask; int num_codecs; int step; int i; @@ -1575,8 +1571,9 @@ int asoc_sdw_hw_params(struct snd_pcm_substream *substream, /* Identical data will be sent to all codecs in playback */ if (substream->stream == SNDRV_PCM_STREAM_PLAYBACK) { - ch_mask = GENMASK(ch - 1, 0); + cpu_ch_mask = GENMASK(ch - 1, 0); step = 0; + codec_ch_mask = 0; } else { num_codecs = rtd->dai_link->num_codecs; @@ -1586,17 +1583,24 @@ int asoc_sdw_hw_params(struct snd_pcm_substream *substream, return -EINVAL; } - ch_mask = GENMASK(ch / num_codecs - 1, 0); - step = hweight_long(ch_mask); + cpu_ch_mask = GENMASK(ch / num_codecs - 1, 0); + step = hweight_long(cpu_ch_mask); + codec_ch_mask = cpu_ch_mask; } /* * The captured data will be combined from each cpu DAI if the dai * link has more than one codec DAIs. Set codec channel mask and * ASoC will set the corresponding channel numbers for each cpu dai. + * + * sdw_stream_add_slave() assigns different payload offsets to each + * codec in a capture stream, so that the same channels on each + * codec map to different channels on the CPU. */ - for_each_link_ch_maps(rtd->dai_link, i, ch_maps) - ch_maps->ch_mask = ch_mask << (i * step); + for_each_link_ch_maps(rtd->dai_link, i, ch_maps) { + ch_maps->cpu_ch_mask = cpu_ch_mask << (i * step); + ch_maps->codec_ch_mask = codec_ch_mask; + } return 0; } diff --git a/sound/soc/soc-pcm.c b/sound/soc/soc-pcm.c index 0e49290a8c90..3137c091bdb8 100644 --- a/sound/soc/soc-pcm.c +++ b/sound/soc/soc-pcm.c @@ -1206,7 +1206,9 @@ static int __soc_pcm_hw_params(struct snd_pcm_substream *substream, goto out; for_each_rtd_codec_dais(rtd, i, codec_dai) { - unsigned int tdm_mask = snd_soc_dai_tdm_mask_get(codec_dai, substream->stream); + unsigned int ch_mask = snd_soc_dai_tdm_mask_get(codec_dai, substream->stream); + struct snd_soc_dai_link_ch_map *ch_maps; + int j; /* * Skip CODECs which don't support the current stream type, @@ -1228,9 +1230,15 @@ static int __soc_pcm_hw_params(struct snd_pcm_substream *substream, /* copy params for each codec */ tmp_params = *params; - /* fixup params based on TDM slot masks */ - if (tdm_mask) - soc_pcm_codec_params_fixup(&tmp_params, tdm_mask); + /* fixup params based on TDM or ch_map masks */ + if (!ch_mask) { + for_each_rtd_ch_maps(rtd, j, ch_maps) + if (ch_maps->codec == i) + ch_mask |= ch_maps->codec_ch_mask; + } + + if (ch_mask) + soc_pcm_codec_params_fixup(&tmp_params, ch_mask); ret = snd_soc_dai_hw_params(codec_dai, substream, &tmp_params); @@ -1264,7 +1272,7 @@ static int __soc_pcm_hw_params(struct snd_pcm_substream *substream, */ for_each_rtd_ch_maps(rtd, j, ch_maps) if (ch_maps->cpu == i) - ch_mask |= ch_maps->ch_mask; + ch_mask |= ch_maps->cpu_ch_mask; /* fixup cpu channel number */ if (ch_mask) diff --git a/sound/soc/ux500/ux500_msp_i2s.h b/sound/soc/ux500/ux500_msp_i2s.h index 2bf2699bdc49..c66ef455e138 100644 --- a/sound/soc/ux500/ux500_msp_i2s.h +++ b/sound/soc/ux500/ux500_msp_i2s.h @@ -147,8 +147,8 @@ enum msp_direction { #define RCKPOL_MASK BIT(0) #define TCKPOL_MASK BIT(0) #define SPICKM_MASK (BIT(1) | BIT(0)) -#define MSP_RX_CLKPOL_BIT(n) ((n & RCKPOL_MASK) << RCKPOL_SHIFT) -#define MSP_TX_CLKPOL_BIT(n) ((n & TCKPOL_MASK) << TCKPOL_SHIFT) +#define MSP_RX_CLKPOL_BIT(n) (((n) & RCKPOL_MASK) << RCKPOL_SHIFT) +#define MSP_TX_CLKPOL_BIT(n) (((n) & TCKPOL_MASK) << TCKPOL_SHIFT) #define P1ELEN_SHIFT 0 #define P1FLEN_SHIFT 3 diff --git a/sound/usb/6fire/pcm.c b/sound/usb/6fire/pcm.c index 21789db6657d..0285d79ace0f 100644 --- a/sound/usb/6fire/pcm.c +++ b/sound/usb/6fire/pcm.c @@ -335,11 +335,19 @@ static void usb6fire_pcm_in_urb_handler(struct urb *usb_urb) /* setup out urb structure */ for (i = 0; i < PCM_N_PACKETS_PER_URB; i++) { + unsigned int frames = 0; + isoc_out = &out_urb->instance->iso_frame_desc[i]; isoc_in = &in_urb->instance->iso_frame_desc[i]; + if (isoc_in->actual_length > 4) + frames = (isoc_in->actual_length - 4) + / (rt->in_n_analog << 2); + frames = min_t(unsigned int, frames, + (rt->out_packet_size - 4) + / (rt->out_n_analog << 2)); + isoc_out->offset = total_length; - isoc_out->length = (isoc_in->actual_length - 4) / (rt->in_n_analog << 2) - * (rt->out_n_analog << 2) + 4; + isoc_out->length = frames * (rt->out_n_analog << 2) + 4; isoc_out->status = 0; total_length += isoc_out->length; } diff --git a/sound/usb/bcd2000/bcd2000.c b/sound/usb/bcd2000/bcd2000.c index c5c542d17ccc..2bd49bf82748 100644 --- a/sound/usb/bcd2000/bcd2000.c +++ b/sound/usb/bcd2000/bcd2000.c @@ -43,6 +43,7 @@ struct bcd2000 { struct usb_interface *intf; int card_index; + spinlock_t midi_lock; int midi_out_active; struct snd_rawmidi *rmidi; struct snd_rawmidi_substream *midi_receive_substream; @@ -90,6 +91,8 @@ static void bcd2000_midi_input_trigger(struct snd_rawmidi_substream *substream, int up) { struct bcd2000 *bcd2k = substream->rmidi->private_data; + + guard(spinlock_irqsave)(&bcd2k->midi_lock); bcd2k->midi_receive_substream = up ? substream : NULL; } @@ -195,6 +198,8 @@ static void bcd2000_midi_output_trigger(struct snd_rawmidi_substream *substream, { struct bcd2000 *bcd2k = substream->rmidi->private_data; + guard(spinlock_irqsave)(&bcd2k->midi_lock); + if (up) { bcd2k->midi_out_substream = substream; /* check if there is data userspace wants to send */ @@ -219,6 +224,7 @@ static void bcd2000_output_complete(struct urb *urb) return; /* check if there is more data userspace wants to send */ + guard(spinlock_irqsave)(&bcd2k->midi_lock); bcd2000_midi_send(bcd2k); } @@ -234,6 +240,8 @@ static void bcd2000_input_complete(struct urb *urb) if (!bcd2k || urb->status == -ESHUTDOWN) return; + guard(spinlock_irqsave)(&bcd2k->midi_lock); + if (urb->actual_length > 0) bcd2000_midi_handle_input(bcd2k, urb->transfer_buffer, urb->actual_length); @@ -348,16 +356,26 @@ static int bcd2000_init_midi(struct bcd2000 *bcd2k) return 0; } +static void bcd2000_midi_free(struct bcd2000 *bcd2k, + struct urb **urb_p) +{ + struct urb *urb = *urb_p; + + if (!urb) + return; + + usb_poison_urb(urb); + scoped_guard(spinlock_irq, &bcd2k->midi_lock) + *urb_p = NULL; + + usb_free_urb(urb); +} + static void bcd2000_free_usb_related_resources(struct bcd2000 *bcd2k, struct usb_interface *interface) { - usb_poison_urb(bcd2k->midi_out_urb); - usb_poison_urb(bcd2k->midi_in_urb); - - usb_free_urb(bcd2k->midi_out_urb); - usb_free_urb(bcd2k->midi_in_urb); - bcd2k->midi_out_urb = NULL; - bcd2k->midi_in_urb = NULL; + bcd2000_midi_free(bcd2k, &bcd2k->midi_out_urb); + bcd2000_midi_free(bcd2k, &bcd2k->midi_in_urb); if (bcd2k->intf) { usb_set_intfdata(bcd2k->intf, NULL); @@ -393,6 +411,7 @@ static int bcd2000_probe(struct usb_interface *interface, bcd2k->card = card; bcd2k->card_index = card_index; bcd2k->intf = interface; + spin_lock_init(&bcd2k->midi_lock); snd_card_set_dev(card, &interface->dev); diff --git a/sound/usb/card.h b/sound/usb/card.h index e34d92d576a2..8299ac241c60 100644 --- a/sound/usb/card.h +++ b/sound/usb/card.h @@ -116,6 +116,7 @@ struct snd_usb_endpoint { unsigned int phase; /* phase accumulator */ unsigned int maxpacksize; /* max packet size in bytes */ unsigned int maxframesize; /* max packet size in frames */ + unsigned int max_urb_packs; /* packets allocated per data URB */ unsigned int max_urb_frames; /* max URB size in frames */ unsigned int curpacksize; /* current packet size in bytes (for capture) */ unsigned int curframesize; /* current packet size in frames (for capture) */ diff --git a/sound/usb/endpoint.c b/sound/usb/endpoint.c index 0835943d7b0e..879d0451536e 100644 --- a/sound/usb/endpoint.c +++ b/sound/usb/endpoint.c @@ -449,7 +449,9 @@ static void push_back_to_ready_list(struct snd_usb_endpoint *ep, struct snd_urb_ctx *ctx) { guard(spinlock_irqsave)(&ep->lock); - list_add_tail(&ctx->ready_list, &ep->ready_playback_urbs); + /* ctx may still be linked: a stale completion racing a stop/restart. */ + if (list_empty(&ctx->ready_list)) + list_add_tail(&ctx->ready_list, &ep->ready_playback_urbs); } /* @@ -492,9 +494,10 @@ int snd_usb_queue_pending_output_urbs(struct snd_usb_endpoint *ep, /* copy over the length information */ if (implicit_fb) { - ctx->packets = packet->packets; + ctx->packets = min_t(int, packet->packets, + ep->max_urb_packs); memcpy(ctx->packet_size, packet->packet_size, - packet->packets * sizeof(packet->packet_size[0])); + ctx->packets * sizeof(packet->packet_size[0])); } /* call the data handler to fill in playback data */ @@ -1036,6 +1039,7 @@ void snd_usb_endpoint_sync_pending_stop(struct snd_usb_endpoint *ep) */ static int stop_urbs(struct snd_usb_endpoint *ep, bool force, bool keep_pending) { + struct snd_urb_ctx *ctx, *n; unsigned int i; if (!force && atomic_read(&ep->running)) @@ -1045,7 +1049,9 @@ static int stop_urbs(struct snd_usb_endpoint *ep, bool force, bool keep_pending) return 0; scoped_guard(spinlock_irqsave, &ep->lock) { - INIT_LIST_HEAD(&ep->ready_playback_urbs); + /* Unlink each ctx; INIT_LIST_HEAD() alone would leave them looking linked. */ + list_for_each_entry_safe(ctx, n, &ep->ready_playback_urbs, ready_list) + list_del_init(&ctx->ready_list); ep->next_packet_head = 0; ep->next_packet_queued = 0; } @@ -1242,15 +1248,16 @@ static int data_ep_set_params(struct snd_usb_endpoint *ep) ep->nurbs = min(max_urbs, urbs_per_period * ep->cur_buffer_periods); } + if (fmt->fmt_type == UAC_FORMAT_TYPE_II) + urb_packs++; /* for transfer delimiter */ + ep->max_urb_packs = urb_packs; + /* allocate and initialize data urbs */ for (i = 0; i < ep->nurbs; i++) { struct snd_urb_ctx *u = &ep->urb[i]; u->index = i; u->ep = ep; u->packets = urb_packs; - - if (fmt->fmt_type == UAC_FORMAT_TYPE_II) - u->packets++; /* for transfer delimiter */ u->buffer_size = maxsize * u->packets; u->urb = usb_alloc_urb(u->packets, GFP_KERNEL); if (!u->urb) diff --git a/sound/usb/implicit.c b/sound/usb/implicit.c index 77f06da93151..bd4d569a8190 100644 --- a/sound/usb/implicit.c +++ b/sound/usb/implicit.c @@ -76,6 +76,7 @@ static const struct snd_usb_implicit_fb_match playback_implicit_fb_quirks[] = { /* Implicit feedback quirk table for capture: only FIXED type */ static const struct snd_usb_implicit_fb_match capture_implicit_fb_quirks[] = { + IMPLICIT_FB_FIXED_DEV(0x1397, 0x0004, 0x01, 1), /* Behringer FCA1616 */ {} /* terminator */ }; diff --git a/sound/usb/mixer_maps.c b/sound/usb/mixer_maps.c index 69093c666282..41454cb403d0 100644 --- a/sound/usb/mixer_maps.c +++ b/sound/usb/mixer_maps.c @@ -532,6 +532,15 @@ static const struct usbmix_name_map audient_id24_map[] = { }; /* + * The GC553Pro returns no data for GET_CUR on its advertised mute control. + * SET_CUR succeeds but does not mute capture, so skip the control entirely. + */ +static const struct usbmix_name_map avermedia_gc553pro_map[] = { + { 3, NULL, UAC_FU_MUTE }, + {} +}; + +/* * Control map entries */ @@ -578,6 +587,10 @@ static const struct usbmix_ctl_map usbmix_ctl_maps[] = { .selector_map = c400_selectors, }, { + .id = USB_ID(0x07ca, 0x1553), + .map = avermedia_gc553pro_map, + }, + { .id = USB_ID(0x08bb, 0x2702), .map = linex_map, }, diff --git a/sound/virtio/virtio_card.c b/sound/virtio/virtio_card.c index 647190f4d5af..6f35276416fe 100644 --- a/sound/virtio/virtio_card.c +++ b/sound/virtio/virtio_card.c @@ -354,8 +354,8 @@ static void virtsnd_remove(struct virtio_device *vdev) if (snd->card) snd_card_free(snd->card); - vdev->config->del_vqs(vdev); virtio_reset_device(vdev); + vdev->config->del_vqs(vdev); for (i = 0; snd->substreams && i < snd->nsubstreams; ++i) { struct virtio_pcm_substream *vss = &snd->substreams[i]; @@ -383,8 +383,8 @@ static int virtsnd_freeze(struct virtio_device *vdev) virtsnd_disable_event_vq(snd); virtsnd_ctl_msg_cancel_all(snd); - vdev->config->del_vqs(vdev); virtio_reset_device(vdev); + vdev->config->del_vqs(vdev); for (i = 0; i < snd->nsubstreams; ++i) cancel_work_sync(&snd->substreams[i].elapsed_period); diff --git a/tools/objtool/Makefile b/tools/objtool/Makefile index a4484fd22a96..4cc2e756af84 100644 --- a/tools/objtool/Makefile +++ b/tools/objtool/Makefile @@ -89,9 +89,11 @@ LIBOPCODES_LIBS := $(shell \ "-lopcodes -lbfd" \ "-lopcodes -lbfd -liberty" \ "-lopcodes -lbfd -liberty -lz"; do \ - echo 'extern void disassemble_init_for_target(void *);' \ - 'int main(void) { disassemble_init_for_target(0); return 0; }' | \ - $(HOSTCC) -xc - -o /dev/null $$libs 2>/dev/null && \ + printf '%s\n' \ + '$(pound)include <bfd.h>' \ + '$(pound)include <dis-asm.h>' \ + 'int main(void) { disassemble_init_for_target(0); return 0; }' | \ + $(HOSTCC) $(HOSTCFLAGS) -DPACKAGE='"objtool"' -xc - -o /dev/null $$libs 2>/dev/null && \ echo "$$libs" && break; \ done) diff --git a/tools/sched_ext/include/scx/common.bpf.h b/tools/sched_ext/include/scx/common.bpf.h index 76f5e025e107..2ddb01a059fd 100644 --- a/tools/sched_ext/include/scx/common.bpf.h +++ b/tools/sched_ext/include/scx/common.bpf.h @@ -113,6 +113,7 @@ s32 scx_bpf_this_cid(void) __ksym __weak; struct task_struct *scx_bpf_cid_curr(s32 cid) __ksym __weak; u32 scx_bpf_nr_cids(void) __ksym __weak; u32 scx_bpf_nr_online_cids(void) __ksym __weak; +const void __arena *scx_bpf_online_cmask(void) __ksym __weak; u32 scx_bpf_cidperf_cap(s32 cid) __ksym __weak; u32 scx_bpf_cidperf_cur(s32 cid) __ksym __weak; s32 scx_bpf_cidperf_set(s32 cid, u32 perf) __ksym __weak; diff --git a/tools/sched_ext/scx_qmap.bpf.c b/tools/sched_ext/scx_qmap.bpf.c index 9f6e61d7ca07..67b7c01cae55 100644 --- a/tools/sched_ext/scx_qmap.bpf.c +++ b/tools/sched_ext/scx_qmap.bpf.c @@ -24,6 +24,9 @@ * time-share that stays self-local. * self - The excl cpus the node kept for itself, plus all of held_shared. * owner - Who holds a cid - a child slot, CID_SELF, or CID_NONE. + * avail - Cpus whose caps are in effect, per ops.sub_ecaps_updated(). + * usable - self AND avail. Placement decisions use this: self is the + * delegation split and can run ahead of what the cpus honor. * * The scheduler splits its held-excl cpus among self and the children in * proportion to each node's cpu.weight, handing each the floor of its share as @@ -208,8 +211,8 @@ static int qmap_spin_lock(struct bpf_res_spin_lock *lock) } /* - * Try prev_cid, then scan cpus_allowed AND idle_cids AND self_cids round-robin - * from prev_cid + 1. Atomic claim retries on race; bounded by + * Try prev_cid, then scan cpus_allowed AND idle_cids AND usable_cids + * round-robin from prev_cid + 1. Atomic claim retries on race; bounded by * IDLE_PICK_RETRIES to keep the verifier's insn budget in check. */ #define IDLE_PICK_RETRIES 16 @@ -221,7 +224,7 @@ static s32 pick_direct_dispatch_cid(struct task_struct *p, s32 prev_cid, s32 cid; u32 i; - if (cmask_test(prev_cid, &qa.self_cids.mask) && + if (cmask_test(prev_cid, &qa.usable_cids.mask) && cmask_test_and_clear(prev_cid, &qa.idle_cids.mask)) return prev_cid; @@ -229,7 +232,7 @@ static s32 pick_direct_dispatch_cid(struct task_struct *p, s32 prev_cid, bpf_for(i, 0, IDLE_PICK_RETRIES) { cid = cmask_next_and2_set_wrap(&taskc->cpus_allowed, &qa.idle_cids.mask, - &qa.self_cids.mask, cid + 1); + &qa.usable_cids.mask, cid + 1); barrier_var(cid); if (cid >= nr_cids) return -1; @@ -358,8 +361,8 @@ s32 BPF_STRUCT_OPS(qmap_select_cid, struct task_struct *p, } /* - * A received time-shared cid is held ENQ_IMMED-only, so inserts must set - * SCX_ENQ_IMMED. + * A received time-shared cid is held ENQ_IMMED-only, so inserts meant to run + * there must set SCX_ENQ_IMMED. */ static u64 needs_immed(s32 cid) { @@ -444,9 +447,11 @@ void BPF_STRUCT_OPS(qmap_enqueue, struct task_struct *p, u64 enq_flags) * didn't grant them or we delegated them to children - would starve in * SHARED/FIFO since we only pull from those on self cids. * - * Force it onto its first allowed cid's local DSQ. If we hold that cid - * it runs. Otherwise the insert carries SCX_ENQ_RESCUE and the kernel - * diverts the task to its rescue path. + * Force it onto its first allowed cid's local DSQ with SCX_ENQ_RESCUE. + * If we hold ENQ on that cid it runs. Otherwise the kernel diverts the + * task to its rescue path. IMMED would turn the insert into a legal + * placement on a time-shared cid and the kernel would bounce it back + * here instead of rescuing it. */ if (!cmask_intersects(&taskc->cpus_allowed, &qa.self_cids.mask)) { s32 c = cmask_next_set_wrap(&taskc->cpus_allowed, 0); @@ -455,7 +460,7 @@ void BPF_STRUCT_OPS(qmap_enqueue, struct task_struct *p, u64 enq_flags) taskc->force_local = false; __sync_fetch_and_add(&qa.nr_rescue_dsp, 1); scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL_ON | c, slice_ns, - enq_flags | needs_immed(c) | SCX_ENQ_RESCUE); + enq_flags | SCX_ENQ_RESCUE); return; } } @@ -540,7 +545,7 @@ void BPF_STRUCT_OPS(qmap_enqueue, struct task_struct *p, u64 enq_flags) scx_bpf_dsq_insert(p, SHARED_DSQ, 0, enq_flags); cid = cmask_next_and2_set_wrap(&taskc->cpus_allowed, &qa.idle_cids.mask, - &qa.self_cids.mask, 0); + &qa.usable_cids.mask, 0); if (cid < scx_bpf_nr_cids()) scx_bpf_kick_cid(cid, SCX_KICK_IDLE); return; @@ -618,7 +623,7 @@ static bool scan_shared_dsq(bool from_timer) if (c >= 0 && c < scx_bpf_nr_cids()) { __sync_fetch_and_add(&qa.nr_rescue_dsp, 1); scx_bpf_dsq_move(BPF_FOR_EACH_ITER, p, SCX_DSQ_LOCAL_ON | c, - needs_immed(c) | SCX_ENQ_RESCUE); + SCX_ENQ_RESCUE); } continue; } @@ -644,22 +649,27 @@ static bool scan_shared_dsq(bool from_timer) if (!(taskc = lookup_task_ctx(p))) return false; - /* only run highpri tasks on cids this node holds, not delegated ones */ + /* only run highpri tasks on cids this node can use right now */ if (cmask_test(this_cid, &taskc->cpus_allowed) && - cmask_test(this_cid, &qa.self_cids.mask)) + cmask_test(this_cid, &qa.usable_cids.mask)) cid = this_cid; else cid = cmask_next_and_set_wrap(&taskc->cpus_allowed, - &qa.self_cids.mask, + &qa.usable_cids.mask, this_cid + 1); if (cid >= nr_cids) { - /* stranded after the cull - rescue it from here */ - s32 c = cmask_next_set_wrap(&taskc->cpus_allowed, 0); + s32 c; + + /* self cids lack caps in effect yet, leave it queued */ + if (cmask_intersects(&taskc->cpus_allowed, &qa.self_cids.mask)) + continue; + /* stranded after the cull - rescue it from here */ + c = cmask_next_set_wrap(&taskc->cpus_allowed, 0); if (c >= 0 && c < nr_cids) { __sync_fetch_and_add(&qa.nr_rescue_dsp, 1); scx_bpf_dsq_move(BPF_FOR_EACH_ITER, p, SCX_DSQ_LOCAL_ON | c, - needs_immed(c) | SCX_ENQ_RESCUE); + SCX_ENQ_RESCUE); } continue; } @@ -808,10 +818,10 @@ void BPF_STRUCT_OPS(qmap_dispatch, s32 cid, struct task_struct *prev) batch--; cpuc->dsp_cnt--; if (!batch || !scx_bpf_dispatch_nr_slots()) { - if (scan_shared_dsq(false)) + if (scan_shared_dsq(false) || + scx_bpf_dsq_move_to_local(SHARED_DSQ, needs_immed(cid))) return; - scx_bpf_dsq_move_to_local(SHARED_DSQ, needs_immed(cid)); - return; + goto prev; } if (!cpuc->dsp_cnt) break; @@ -822,10 +832,14 @@ void BPF_STRUCT_OPS(qmap_dispatch, s32 cid, struct task_struct *prev) if (scan_shared_dsq(false)) return; - +prev: /* * No other tasks. @prev will keep running. Update its core_sched_seq as * if the task were enqueued and dispatched immediately. + * + * No @prev to keep running means the CPU goes idle. If its claim was + * never used, that is not a transition and ops.update_idle() stays + * silent. Restore the claim here. */ if (prev) { taskc = lookup_task_ctx(prev); @@ -834,6 +848,8 @@ void BPF_STRUCT_OPS(qmap_dispatch, s32 cid, struct task_struct *prev) taskc->core_sched_seq = qa.core_sched_tail_seqs[weight_to_idx(prev->scx.weight)]++; + } else { + cmask_set(cid, &qa.idle_cids.mask); } } @@ -1113,7 +1129,7 @@ void BPF_STRUCT_OPS(qmap_update_idle, s32 cid, bool idle) /* * The kernel delivers update_idle() for every cid this node holds * SCX_CAP_BASE on. Track every cid's idle state regardless of - * delegation: the direct-dispatch pick masks idle_cids with self_cids + * delegation: the direct-dispatch pick masks idle_cids with usable_cids * at selection, so a cid already idle when it returns to self needs no * reseed here. */ @@ -1285,11 +1301,16 @@ struct { __type(value, struct round_robin_timer); } round_robin_timer SEC(".maps"); +enum part_pending_flags { + PART_REFRESH = BIT_U64(0), + PART_REDISTRIBUTE = BIT_U64(1), +}; + /* * Partition update synchronization. qa.part can be written from concurrent * contexts. This single-runner guard admits one writer at a time without * holding a lock across the grant/revoke kfuncs. part_pending coalesces - * repartition requests that arrive while it is held. + * refresh and repartition requests that arrive while it is held. * * They live in .bss, not the arena: rr_advance() runs from a bpf_timer * callback, where the verifier rejects atomic ops on arena memory. @@ -1538,6 +1559,19 @@ static __noinline void account_alloc(void) } /* + * usable_cids = self_cids & avail_cids. The inputs have separate writers, + * apply_partition() and qmap_sub_ecaps_updated(), so the result is rebuilt in + * full under the partition guard, in scratch first so that readers never see + * self_cids alone. + */ +static void refresh_usable(void) +{ + cmask_copy(&qa.usable_scratch.mask, &qa.self_cids.mask); + cmask_and(&qa.usable_scratch.mask, &qa.avail_cids.mask); + cmask_copy(&qa.usable_cids.mask, &qa.usable_scratch.mask); +} + +/* * apply_partition - execute the plan compute_partition() built * * Turn the owner map into the per-child, shared and self cmasks and issue the @@ -1559,6 +1593,7 @@ __noinline void apply_partition(void) /* no excl cpu: run own tasks on the held shares, evict children */ if (!qa.part.nr_excl) { cmask_copy(&qa.self_cids.mask, &qa.held_shared.mask); + refresh_usable(); bpf_for(i, 0, MAX_SUB_SCHEDS) if (qa.sub_sched_ctxs[i].cgroup_id) scx_bpf_sub_kill(qa.sub_sched_ctxs[i].cgroup_id, @@ -1596,6 +1631,7 @@ __noinline void apply_partition(void) else if (o == CID_SELF) cmask_set(cid, &qa.self_cids.mask); } + refresh_usable(); /* * Apply each child's exclusive cids as a delta against its previous @@ -1643,33 +1679,46 @@ __noinline void apply_partition(void) } } -/* - * Recompute the split off the node's held caps and apply it. The contexts this - * runs from (the sub-sched and cgroup callbacks, the rr timer) are not - * serialized by the kernel, so a single runner does the work. A caller that - * finds the guard held leaves part_pending set; the holder drains it before - * releasing, with the rr timer as a backstop. +/** + * execute_partition - Run pending partition updates + * + * The rr timer is the backstop if the loop reaches its iteration limit. */ -static void redistribute(void) +static void execute_partition(void) { + u64 pending; s32 i; - __sync_fetch_and_or(&part_pending, 1); + bpf_for(i, 0, 1024) { + if (!part_try_start()) + break; - if (!part_try_start()) - return; + pending = __sync_fetch_and_and(&part_pending, 0); + if (pending & PART_REDISTRIBUTE) { + /* charge elapsed time before repartitioning */ + account_alloc(); + compute_partition(); + apply_partition(); + } else if (pending & PART_REFRESH) { + refresh_usable(); + } - bpf_for(i, 0, 1024) { - __sync_fetch_and_and(&part_pending, 0); - /* charge elapsed time to the current partition before rebuilding it */ - account_alloc(); - compute_partition(); - apply_partition(); + /* + * Requests are published before trying the guard. Releasing it + * before checking pending work ensures a racing request is + * either observed here or handled by a caller that acquires the + * guard. + */ + part_end(); if (!__sync_fetch_and_or(&part_pending, 0)) break; } +} - part_end(); +static void redistribute(void) +{ + __sync_fetch_and_or(&part_pending, PART_REDISTRIBUTE); + execute_partition(); } /* @@ -1683,6 +1732,7 @@ int flush_alloc(void *ctx) if (part_try_start()) { account_alloc(); part_end(); + execute_partition(); } return 0; } @@ -1740,9 +1790,7 @@ static void rr_advance(void) part_end(); - /* a resplit queued while we held the guard supersedes this rotation */ - if (__sync_fetch_and_or(&part_pending, 0)) - redistribute(); + execute_partition(); } /* advance the time-shared cid pool every round_robin_ns */ @@ -1837,8 +1885,11 @@ s32 BPF_STRUCT_OPS_SLEEPABLE(qmap_init) cmask_init(&qa.rr_cids.mask, 0, nr_cids); cmask_init(&qa.prev_rr_cids.mask, 0, nr_cids); cmask_init(&qa.self_cids.mask, 0, nr_cids); + cmask_init(&qa.avail_cids.mask, 0, nr_cids); + cmask_init(&qa.usable_cids.mask, 0, nr_cids); cmask_init(&qa.to_revoke_cids.mask, 0, nr_cids); cmask_init(&qa.to_grant_cids.mask, 0, nr_cids); + cmask_init(&qa.usable_scratch.mask, 0, nr_cids); cmask_init(&qa.held_excl.mask, 0, nr_cids); cmask_init(&qa.held_shared.mask, 0, nr_cids); @@ -1852,14 +1903,16 @@ s32 BPF_STRUCT_OPS_SLEEPABLE(qmap_init) } /* - * The root starts holding every cid. qmap_sub_ecaps_updated() maintains - * per-cid shared state as effective caps settle, and redistribute() - * rebuilds owner and self from held caps. A non-root node starts with - * nothing. + * The root starts holding every cid and gets no ecaps notifications, so + * its avail set is fixed here. qmap_sub_ecaps_updated() maintains the + * per-cid state as effective caps settle, and redistribute() rebuilds + * owner and self from held caps. A non-root node starts with nothing. */ bpf_for(i, 0, nr_cids) { if (!sub_cgroup_id) { cmask_set(i, &qa.self_cids.mask); + cmask_set(i, &qa.avail_cids.mask); + cmask_set(i, &qa.usable_cids.mask); qa.part.cid_owner[i] = CID_SELF; } else { qa.part.cid_owner[i] = CID_NONE; @@ -2000,12 +2053,19 @@ void BPF_STRUCT_OPS(qmap_sub_ecaps_updated, s32 cid, u64 before, u64 after) { /* * Effective caps updated. Track which cids hold shared caps so a self - * task placed there enqueues IMMED. + * task placed there enqueues IMMED, and which cids have ENQ_IMMED in + * effect at all (avail, see the header comment). */ - if (after & SCX_CAP_ENQ_IMMED) + if (after & SCX_CAP_ENQ_IMMED) { qa.cid_shared[cid] = (after & SCX_CAP_ENQ) ? 0 : 1; - else + cmask_set(cid, &qa.avail_cids.mask); + } else { qa.cid_shared[cid] = 0; + cmask_clear(cid, &qa.avail_cids.mask); + } + + __sync_fetch_and_or(&part_pending, PART_REFRESH); + execute_partition(); } SCX_OPS_CID_DEFINE(qmap_ops, diff --git a/tools/sched_ext/scx_qmap.h b/tools/sched_ext/scx_qmap.h index c78d61806b39..e95fffcf7b23 100644 --- a/tools/sched_ext/scx_qmap.h +++ b/tools/sched_ext/scx_qmap.h @@ -165,12 +165,15 @@ struct qmap_arena { /* bpf-internal cmasks (embedded, see struct qmap_cmask) */ struct qmap_cmask self_cids; /* cids this node runs its own tasks on */ + struct qmap_cmask avail_cids; /* cids with caps in effect on the cpu */ + struct qmap_cmask usable_cids; /* self_cids & avail_cids, placeable right now */ struct qmap_cmask idle_cids; /* idle state of all cids regardless of delegation */ struct qmap_cmask rr_cids; /* the shared pool, as a mask for grant/revoke */ /* scratch cmasks */ struct qmap_cmask to_revoke_cids; /* delta cids to revoke */ struct qmap_cmask to_grant_cids; /* delta cids to grant */ + struct qmap_cmask usable_scratch; /* refresh_usable() build area */ struct qmap_cmask prev_rr_cids; /* previous shared pool, to clear stale grants */ struct qmap_cmask held_excl; /* cids held excl (ENQ): delegatable */ struct qmap_cmask held_shared; /* cids held shared (ENQ_IMMED only): self-local */ diff --git a/tools/testing/selftests/arm64/mte/check_gcr_el1_cswitch.c b/tools/testing/selftests/arm64/mte/check_gcr_el1_cswitch.c index d23f154d3288..5d9dc8bcfbf5 100644 --- a/tools/testing/selftests/arm64/mte/check_gcr_el1_cswitch.c +++ b/tools/testing/selftests/arm64/mte/check_gcr_el1_cswitch.c @@ -69,7 +69,7 @@ fail: int execute_test(pid_t pid) { pthread_t thread_id[MAX_THREADS]; - int thread_data[MAX_THREADS]; + intptr_t thread_data[MAX_THREADS]; for (int i = 0; i < MAX_THREADS; i++) pthread_create(&thread_id[i], NULL, diff --git a/tools/testing/selftests/drivers/net/config b/tools/testing/selftests/drivers/net/config index b6989c7d3d9d..4838adf27fa1 100644 --- a/tools/testing/selftests/drivers/net/config +++ b/tools/testing/selftests/drivers/net/config @@ -21,5 +21,6 @@ CONFIG_NET_SCH_INGRESS=y CONFIG_NET_SCH_PRIO=m CONFIG_PPP=y CONFIG_PPPOE=y +CONFIG_TLS=y CONFIG_VLAN_8021Q=m CONFIG_XDP_SOCKETS=y diff --git a/tools/testing/selftests/drivers/net/psp.py b/tools/testing/selftests/drivers/net/psp.py index 315648a770d0..a5b1e14f120f 100755 --- a/tools/testing/selftests/drivers/net/psp.py +++ b/tools/testing/selftests/drivers/net/psp.py @@ -23,6 +23,8 @@ from lib.py import NetNSEnter from lib.py import bkg, rand_port, wait_port_listen from lib.py import ip +TCP_ULP = 31 + def _get_outq(s): one = b'\0' * 4 @@ -333,6 +335,50 @@ def assoc_version_mismatch(cfg): ksft_eq(the_exception.nl_msg.error, -errno.EINVAL) +def _require_tls_ulp(): + with socket.create_server(("localhost", 0)) as srv, \ + socket.create_connection(srv.getsockname()) as s: + try: + s.setsockopt(socket.SOL_TCP, TCP_ULP, b"tls") + except OSError as exc: + raise KsftSkipEx("kTLS not available") from exc + + +def assoc_psp_ulp_exclusive(cfg): + """ Test that a TCP ULP cannot be attached to a PSP socket """ + _init_psp_dev(cfg) + _require_tls_ulp() + + with _make_clr_conn(cfg) as s: + try: + cfg.pspnl.rx_assoc({"version": 0, + "dev-id": cfg.psp_dev_id, + "sock-fd": s.fileno()}) + with ksft_raises(OSError) as cm: + s.setsockopt(socket.SOL_TCP, TCP_ULP, b"tls") + ksft_eq(cm.exception.errno, errno.EINVAL) + finally: + _close_conn(cfg, s) + + +def assoc_ulp_psp_exclusive(cfg): + """ Test that a PSP assoc cannot be added to a socket with a TCP ULP """ + _init_psp_dev(cfg) + _require_tls_ulp() + + with _make_clr_conn(cfg) as s: + try: + s.setsockopt(socket.SOL_TCP, TCP_ULP, b"tls") + with ksft_raises(NlError) as cm: + cfg.pspnl.rx_assoc({"version": 0, + "dev-id": cfg.psp_dev_id, + "sock-fd": s.fileno()}) + ksft_eq(cm.exception.nl_msg.error, -errno.EINVAL) + ksft_eq(cm.exception.nl_msg.extack['bad-attr'], ".sock-fd") + finally: + _close_conn(cfg, s) + + def assoc_twice(cfg): """ Test reusing Tx assoc for two sockets """ _init_psp_dev(cfg) diff --git a/tools/testing/selftests/net/af_unix/scm_rights.c b/tools/testing/selftests/net/af_unix/scm_rights.c index d82a79c21c17..c165f250220a 100644 --- a/tools/testing/selftests/net/af_unix/scm_rights.c +++ b/tools/testing/selftests/net/af_unix/scm_rights.c @@ -378,4 +378,21 @@ TEST_F(scm_rights, backtrack_from_scc) close_sockets(10); } +TEST_F(scm_rights, mixed_lowpoint) +{ + create_sockets(6); + + send_fd(0, 1); + send_fd(1, 2); + send_fd(2, 1); + send_fd(1, 0); + + send_fd(3, 4); + send_fd(4, 5); + send_fd(5, 4); + send_fd(4, 3); + + close_sockets(6); +} + TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/net/packetdrill/tcp_rfc5961_reject-old-ack.pkt b/tools/testing/selftests/net/packetdrill/tcp_rfc5961_reject-old-ack.pkt new file mode 100644 index 000000000000..32dd9de1d366 --- /dev/null +++ b/tools/testing/selftests/net/packetdrill/tcp_rfc5961_reject-old-ack.pkt @@ -0,0 +1,29 @@ +// SPDX-License-Identifier: GPL-2.0 + +`./defaults.sh +sysctl -q net.ipv4.tcp_invalid_ratelimit=0 +` + +// Test rejection of data segments carrying excessively old ACKs + +0 socket(..., SOCK_STREAM, IPPROTO_TCP) = 3 ++0 setsockopt(3, SOL_SOCKET, SO_REUSEADDR, [1], 4) = 0 ++0 bind(3, ..., ...) = 0 ++0 listen(3, 1024) = 0 + +// ---------------- Handshake ------------------- // ++0 < S 0:0(0) win 65535 ++0 > S. 0:0(0) ack 1 <...> ++0 < . 1:1(0) ack 1 win 65535 ++0 accept(3, ..., ...) = 4 + +// Populate receive memory so the following segment can use +// header prediction. ++0 < P. 1:501(500) ack 1 win 65535 ++0 > . 1:1(0) ack 501 + +// Send an in-sequence data segment carrying an excessively old ACK. ++0 < P. 501:1501(1000) ack 2794967397 win 65535 + +// Challenge ACK; RCV.NXT must remain 501. ++0 > . 1:1(0) ack 501 diff --git a/tools/testing/selftests/tc-testing/tc-tests/actions/batch-delete.json b/tools/testing/selftests/tc-testing/tc-tests/actions/batch-delete.json new file mode 100644 index 000000000000..ef7ca4a6775b --- /dev/null +++ b/tools/testing/selftests/tc-testing/tc-tests/actions/batch-delete.json @@ -0,0 +1,115 @@ +[ + { + "id": "d710", + "name": "Release tail references after first action deletion fails", + "category": [ + "actions", + "gact" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + [ + "$TC actions flush action gact", + 0, + 1, + 255 + ], + "$TC qdisc add dev $DEV1 ingress", + "$TC actions add action pass index 1", + "$TC actions add action pass index 2", + "$TC actions add action pass index 3", + "$TC filter add dev $DEV1 protocol ip ingress u32 match u32 0 0 action gact index 1" + ], + "cmdUnderTest": "$TC actions del action gact index 1 action gact index 2 action gact index 3", + "expExitCode": "255", + "verifyCmd": "$TC actions ls action gact", + "matchPattern": "total acts 3\\b.*index 1 ref 2 bind 1\\b.*index 2 ref 1 bind 0\\b.*index 3 ref 1 bind 0\\b", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DEV1 ingress", + [ + "$TC actions flush action gact", + 0, + 1, + 255 + ] + ] + }, + { + "id": "d711", + "name": "Release tail references after middle action deletion fails", + "category": [ + "actions", + "gact" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + [ + "$TC actions flush action gact", + 0, + 1, + 255 + ], + "$TC qdisc add dev $DEV1 ingress", + "$TC actions add action pass index 1", + "$TC actions add action pass index 2", + "$TC actions add action pass index 3", + "$TC filter add dev $DEV1 protocol ip ingress u32 match u32 0 0 action gact index 2" + ], + "cmdUnderTest": "$TC actions del action gact index 1 action gact index 2 action gact index 3", + "expExitCode": "255", + "verifyCmd": "$TC actions ls action gact", + "matchPattern": "total acts 2\\b.*index 2 ref 2 bind 1\\b.*index 3 ref 1 bind 0\\b", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DEV1 ingress", + [ + "$TC actions flush action gact", + 0, + 1, + 255 + ] + ] + }, + { + "id": "d713", + "name": "Delete a tail action once after a failed batch", + "category": [ + "actions", + "gact" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + [ + "$TC actions flush action gact", + 0, + 1, + 255 + ], + "$TC qdisc add dev $DEV1 ingress", + "$TC actions add action pass index 1", + "$TC actions add action pass index 2", + "$TC filter add dev $DEV1 protocol ip ingress u32 match u32 0 0 action gact index 1" + ], + "cmdUnderTest": "$TC actions del action gact index 1 action gact index 2", + "expExitCode": "255", + "verifyCmd": "sh -c '$TC actions del action gact index 2 && $TC actions ls action gact'", + "matchPattern": "total acts 1\\b.*index 1 ref 2 bind 1\\b", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DEV1 ingress", + [ + "$TC actions flush action gact", + 0, + 1, + 255 + ] + ] + } +] diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/codel.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/codel.json index 6d515d0e5ed6..a894e6f0e267 100644 --- a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/codel.json +++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/codel.json @@ -213,5 +213,77 @@ "matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1p target 5ms interval 100ms", "matchCount": "1", "teardown": ["$TC qdisc del dev $DEV1 handle 1: root"] + }, + { + "id": "6e44", + "name": "Create CODEL with 1us interval, accepted (sub-tick, uAPI locked)", + "category": [ + "qdisc", + "codel" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root codel interval 1us", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1000p target 5ms interval 0us", + "matchCount": "1", + "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"] + }, + { + "id": "a8c3", + "name": "Create CODEL with 3us interval, accepted (two ticks)", + "category": [ + "qdisc", + "codel" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root codel interval 3us", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1000p target 5ms interval 2us", + "matchCount": "1", + "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"] + }, + { + "id": "a695", + "name": "Create CODEL with 1024us interval boundary accepted", + "category": [ + "qdisc", + "codel" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root codel interval 1024us", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1000p target 5ms interval 1.02ms", + "matchCount": "1", + "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"] + }, + { + "id": "9793", + "name": "Create CODEL with 1us target, accepted (target not in control law)", + "category": [ + "qdisc", + "codel" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root codel target 1us", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc codel 1: root refcnt [0-9]+ limit 1000p target 0us interval 100ms", + "matchCount": "1", + "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"] } ] diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/fq_codel.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/fq_codel.json index 4ce62b857fd7..de6a1b8d954a 100644 --- a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/fq_codel.json +++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/fq_codel.json @@ -316,5 +316,77 @@ "matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 1p flows 1024 quantum.*target 5ms interval 100ms memory_limit 32Mb ecn drop_batch 64", "matchCount": "1", "teardown": ["$TC qdisc del dev $DEV1 handle 1: root"] + }, + { + "id": "1b4d", + "name": "Create FQ_CODEL with 1us interval, accepted (sub-tick, uAPI locked)", + "category": [ + "qdisc", + "fq_codel" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root fq_codel interval 1us", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 10240p flows 1024 quantum [0-9]+ target 5ms interval 0us memory_limit 32Mb ecn drop_batch 64", + "matchCount": "1", + "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"] + }, + { + "id": "3540", + "name": "Create FQ_CODEL with 3us interval, accepted (two ticks)", + "category": [ + "qdisc", + "fq_codel" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root fq_codel interval 3us", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 10240p flows 1024 quantum [0-9]+ target 5ms interval 2us memory_limit 32Mb ecn drop_batch 64", + "matchCount": "1", + "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"] + }, + { + "id": "49c5", + "name": "Create FQ_CODEL with 1024us interval boundary accepted", + "category": [ + "qdisc", + "fq_codel" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root fq_codel interval 1024us", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 10240p flows 1024 quantum [0-9]+ target 5ms interval 1.02ms memory_limit 32Mb ecn drop_batch 64", + "matchCount": "1", + "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"] + }, + { + "id": "3e0f", + "name": "Create FQ_CODEL with 1us target, accepted (target not in control law)", + "category": [ + "qdisc", + "fq_codel" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root fq_codel target 1us", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc fq_codel 1: root refcnt [0-9]+ limit 10240p flows 1024 quantum [0-9]+ target 0us interval 100ms memory_limit 32Mb ecn drop_batch 64", + "matchCount": "1", + "teardown": ["$TC qdisc del dev $DUMMY handle 1: root"] } ] diff --git a/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hhf_flows_limit.json b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hhf_flows_limit.json new file mode 100644 index 000000000000..44538b9266b6 --- /dev/null +++ b/tools/testing/selftests/tc-testing/tc-tests/qdiscs/hhf_flows_limit.json @@ -0,0 +1,128 @@ +[ + { + "id": "e3cc", + "name": "HHF hh_limit rejects value above 2*HH_FLOWS_CNT cap (4294967295)", + "category": [ + "qdisc", + "hhf" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "$TC qdisc add dev $DUMMY handle 1: root hhf" + ], + "cmdUnderTest": "$TC qdisc change dev $DUMMY handle 1: root hhf hh_limit 4294967295", + "expExitCode": "2", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 2048", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DUMMY handle 1: root" + ] + }, + { + "id": "f681", + "name": "HHF hh_limit rejects 65536 (above 2*HH_FLOWS_CNT cap)", + "category": [ + "qdisc", + "hhf" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "$TC qdisc add dev $DUMMY handle 1: root hhf" + ], + "cmdUnderTest": "$TC qdisc change dev $DUMMY handle 1: root hhf hh_limit 65536", + "expExitCode": "2", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 2048", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DUMMY handle 1: root" + ] + }, + { + "id": "223d", + "name": "HHF hh_limit accepts boundary value 2048 (2*HH_FLOWS_CNT)", + "category": [ + "qdisc", + "hhf" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "$TC qdisc add dev $DUMMY handle 1: root hhf hh_limit 100" + ], + "cmdUnderTest": "$TC qdisc change dev $DUMMY handle 1: root hhf hh_limit 2048", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 2048", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DUMMY handle 1: root" + ] + }, + { + "id": "147f", + "name": "HHF hh_limit rejects first value above cap (2049)", + "category": [ + "qdisc", + "hhf" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [ + "$TC qdisc add dev $DUMMY handle 1: root hhf" + ], + "cmdUnderTest": "$TC qdisc change dev $DUMMY handle 1: root hhf hh_limit 2049", + "expExitCode": "2", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 2048", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DUMMY handle 1: root" + ] + }, + { + "id": "4d4f", + "name": "HHF add-time hh_limit 500 is preserved (init does not clobber user value)", + "category": [ + "qdisc", + "hhf" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root hhf hh_limit 500", + "expExitCode": "0", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc hhf 1: root refcnt [0-9]+.*hh_limit 500", + "matchCount": "1", + "teardown": [ + "$TC qdisc del dev $DUMMY handle 1: root" + ] + }, + { + "id": "ca99", + "name": "HHF add-time hh_limit 4294967295 is rejected (no qdisc installed)", + "category": [ + "qdisc", + "hhf" + ], + "plugins": { + "requires": "nsPlugin" + }, + "setup": [], + "cmdUnderTest": "$TC qdisc add dev $DUMMY handle 1: root hhf hh_limit 4294967295", + "expExitCode": "2", + "verifyCmd": "$TC qdisc show dev $DUMMY", + "matchPattern": "qdisc hhf 1: root", + "matchCount": "0", + "teardown": [] + } +] diff --git a/tools/testing/selftests/x86/Makefile b/tools/testing/selftests/x86/Makefile index 434065215d12..d478b13cc8d5 100644 --- a/tools/testing/selftests/x86/Makefile +++ b/tools/testing/selftests/x86/Makefile @@ -13,7 +13,7 @@ CAN_BUILD_WITH_NOPIE := $(shell ./check_cc.sh "$(CC)" trivial_program.c -no-pie) TARGETS_C_BOTHBITS := single_step_syscall sysret_ss_attrs syscall_nt test_mremap_vdso \ check_initial_reg_state sigreturn iopl ioperm \ test_vsyscall mov_ss_trap sigtrap_loop \ - syscall_arg_fault fsgsbase_restore sigaltstack + syscall_arg_fault fsgsbase_restore sigaltstack int_signal TARGETS_C_BOTHBITS += nx_stack TARGETS_C_32BIT_ONLY := entry_from_vm86 test_syscall_vdso unwind_vdso \ test_FCMOV test_FCOMI test_FISTTP \ diff --git a/tools/testing/selftests/x86/int_signal.c b/tools/testing/selftests/x86/int_signal.c new file mode 100644 index 000000000000..22676dac72b5 --- /dev/null +++ b/tools/testing/selftests/x86/int_signal.c @@ -0,0 +1,311 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* Check the signal context for INT instructions with IDT and FRED entry. */ +#define _GNU_SOURCE + +#include <cpuid.h> +#include <errno.h> +#include <stdbool.h> +#include <stddef.h> +#include <stdint.h> +#include <sys/ptrace.h> +#include <sys/user.h> +#include <sys/wait.h> +#include <unistd.h> +#include <ucontext.h> + +#include "helpers.h" + +#ifdef __x86_64__ +#define REG_IP REG_RIP +#define USER_IP rip +#define STACK_PTR "%rsp" +#else +#define REG_IP REG_EIP +#define USER_IP eip +#define STACK_PTR "%esp" +#endif + +/* + * Each instruction has normal and single-step entry points. Resume at the + * NOP after handling its signal, then expect a trace trap after that NOP + * when TF is set. Explicit labels avoid assuming the kernel's saved IP. + */ +#define PROBE(name, insn) \ + extern void name(void); \ + extern void name##_tf(void); \ + extern const char name##_end[], name##_step[]; \ + asm(".pushsection .text\n" \ + ".globl " #name "_tf\n" \ + ".type " #name "_tf, @function\n" \ + #name "_tf:\n" \ + "pushf\n" \ + "orl $0x100, (" STACK_PTR ")\n" \ + "popf\n" \ + ".globl " #name "\n" \ + ".type " #name ", @function\n" \ + #name ":\n" insn "\n" \ + ".globl " #name "_end\n" \ + #name "_end:\nnop\n" \ + ".globl " #name "_step\n" \ + #name "_step:\nret\n" \ + ".size " #name ", .-" #name "\n" \ + ".size " #name "_tf, .-" #name "_tf\n" \ + ".popsection\n") + +PROBE(int1, ".byte 0xcd, 0x01"); +PROBE(int29, ".byte 0xcd, 0x29"); +PROBE(int2c, ".byte 0xcd, 0x2c"); +PROBE(int2d, ".byte 0xcd, 0x2d"); +PROBE(prefixed_int2d, ".byte 0x66, 0xcd, 0x2d"); +PROBE(long_int2d, ".fill 13, 1, 0x2e\n.byte 0xcd, 0x2d"); +PROBE(int81, ".byte 0xcd, 0x81"); +PROBE(intff, ".byte 0xcd, 0xff"); +PROBE(short_int3, ".byte 0xcc"); +PROBE(long_int3, ".byte 0xcd, 0x03"); +PROBE(int4, ".byte 0xcd, 0x04"); +PROBE(ud2, ".byte 0x0f, 0x0b"); +PROBE(hlt, ".byte 0xf4"); + +struct test { + const char *name; + void (*run)(void); + void (*run_tf)(void); + const char *end, *step; + int signo, trap, error, ip_offset, flags, code; +}; + +#define TEST(name, sig, trap, error, offset, flags, code) \ + { #name, name, name##_tf, name##_end, name##_step, \ + sig, trap, error, offset, flags, code } + +#define GP(name, error) \ + TEST(name, SIGSEGV, 13, error, 0, X86_EFLAGS_RF, SI_KERNEL) + +static const struct test tests[] = { + GP(int1, 0x00a), + GP(int29, 0x14a), + GP(int2c, 0x162), + GP(int2d, 0x16a), + GP(prefixed_int2d, 0x16a), + GP(long_int2d, 0x16a), + GP(int81, 0x40a), + GP(intff, 0x7fa), + GP(hlt, 0), + TEST(short_int3, SIGTRAP, 3, 0, 1, 0, SI_KERNEL), + TEST(long_int3, SIGTRAP, 3, 0, 2, 0, SI_KERNEL), + TEST(int4, SIGSEGV, 4, 0, 2, 0, SI_KERNEL), + TEST(ud2, SIGILL, 6, 0, 0, X86_EFLAGS_RF, ILL_ILLOPN), +}; + +static const struct test *active; +static volatile sig_atomic_t seen, signo, trap, error, ip_offset, flags; +static volatile sig_atomic_t code, addr_ok, single_step, stepped, step_ok; + +static void handler(int sig, siginfo_t *info, void *context) +{ + ucontext_t *uc = context; + uintptr_t ip = uc->uc_mcontext.gregs[REG_IP]; + uintptr_t start = (uintptr_t)active->run; + uintptr_t end = (uintptr_t)active->end; + + if (seen && single_step && sig == SIGTRAP) { + if (stepped++) { + ksft_print_msg("%s: second trace trap at %#lx\n", + active->name, (unsigned long)ip); + _exit(KSFT_FAIL); + } + step_ok = ip == (uintptr_t)active->step && + uc->uc_mcontext.gregs[REG_TRAPNO] == 1 && + info->si_code == TRAP_TRACE; + uc->uc_mcontext.gregs[REG_EFL] &= ~X86_EFLAGS_TF; + return; + } + + if (seen || ip < start || ip > end) { + ksft_print_msg("%s: unexpected signal %d at %#lx\n", + active->name, sig, (unsigned long)ip); + _exit(KSFT_FAIL); + } + + signo = sig; + trap = uc->uc_mcontext.gregs[REG_TRAPNO]; + error = uc->uc_mcontext.gregs[REG_ERR]; + ip_offset = ip - start; + flags = uc->uc_mcontext.gregs[REG_EFL] & (X86_EFLAGS_RF | X86_EFLAGS_TF); + code = info->si_code; + /* force_sig() reports no address, force_sig_fault() reports the IP. */ + addr_ok = info->si_addr == (code == SI_KERNEL ? NULL : (void *)ip); + seen = 1; + uc->uc_mcontext.gregs[REG_IP] = end; +} + +static void wait_for_child(pid_t child, int *status) +{ + pid_t ret; + + do { + ret = waitpid(child, status, 0); + } while (ret < 0 && errno == EINTR); + if (ret != child) + ksft_exit_fail_perror("waitpid"); +} + +/* Resume the tracee and check where the next stop lands. */ +static bool resume_to(pid_t child, int *status, int request, int sig, + const void *ip, const char *what) +{ + struct user_regs_struct regs; + + if (ptrace(request, child, 0, 0)) + return false; + wait_for_child(child, status); + if (!WIFSTOPPED(*status)) { + ksft_print_msg("%s: tracee did not stop\n", what); + return false; + } + if (WSTOPSIG(*status) != sig) { + ksft_print_msg("%s: stopped with signal %d, expected %d\n", + what, WSTOPSIG(*status), sig); + return false; + } + if (ptrace(PTRACE_GETREGS, child, 0, ®s)) + return false; + if ((unsigned long)regs.USER_IP != (unsigned long)ip) { + ksft_print_msg("%s: stopped at %#lx, expected %#lx\n", what, + (unsigned long)regs.USER_IP, (unsigned long)ip); + return false; + } + return true; +} + +static bool set_ip(pid_t child, const void *ip, bool tf) +{ + struct user_regs_struct regs; + + if (ptrace(PTRACE_GETREGS, child, 0, ®s)) + return false; + regs.USER_IP = (unsigned long)ip; + if (tf) + regs.eflags |= X86_EFLAGS_TF; + return !ptrace(PTRACE_SETREGS, child, 0, ®s); +} + +/* + * Exercise the tracer paths that resume through the fault frame rather than + * sigreturn. A stale FRED software event flag on that frame traps before the + * NOP executes instead of after it. + */ +static void test_ptrace(void) +{ + bool into = false, step = false, cont = false; + pid_t child; + int status; + + child = fork(); + if (child < 0) + ksft_exit_fail_perror("fork"); + if (!child) { + if (ptrace(PTRACE_TRACEME, 0, 0, 0)) + _exit(KSFT_FAIL); + /* Start from a breakpoint frame, not the syscall frame of raise(). */ + asm volatile("int3"); + _exit(KSFT_FAIL); + } + + wait_for_child(child, &status); + if (!WIFSTOPPED(status) || WSTOPSIG(status) != SIGTRAP) + goto out; + if (ptrace(PTRACE_SETOPTIONS, child, 0, PTRACE_O_EXITKILL)) + goto out; + + /* Single-step into the INT. The fault must report the INT's address. */ + if (!set_ip(child, int2d, false)) + goto out; + into = resume_to(child, &status, PTRACE_SINGLESTEP, SIGSEGV, int2d, + "single-step into INT"); + if (!into) + goto out; + + /* Suppress SIGSEGV and single-step the NOP. */ + if (!set_ip(child, int2d_end, false)) + goto out; + step = resume_to(child, &status, PTRACE_SINGLESTEP, SIGTRAP, int2d_step, + "single-step after INT"); + if (!step) + goto out; + + /* Fault again, then suppress SIGSEGV and continue with TF set. */ + if (!set_ip(child, int2d, false)) + goto out; + if (!resume_to(child, &status, PTRACE_CONT, SIGSEGV, int2d, + "continue to INT")) + goto out; + if (!set_ip(child, int2d_end, true)) + goto out; + cont = resume_to(child, &status, PTRACE_CONT, SIGTRAP, int2d_step, + "continue with TF after INT"); +out: + if (WIFSTOPPED(status)) { + kill(child, SIGKILL); + wait_for_child(child, &status); + } + ksft_test_result(into, "ptrace single-step into INT faults at the INT\n"); + ksft_test_result(step, "ptrace single-step after suppressing SIGSEGV\n"); + ksft_test_result(cont, "ptrace continue with TF after suppressing SIGSEGV\n"); +} + +static bool cpu_has_fred(void) +{ + unsigned int eax, ebx, ecx, edx; + + if (__get_cpuid_max(0, NULL) < 7) + return false; + __cpuid_count(7, 1, eax, ebx, ecx, edx); + return eax & (1 << 17); +} + +int main(void) +{ + unsigned int i, tf; + int expected_flags, ok; + + ksft_print_header(); + ksft_set_plan(2 * ARRAY_SIZE(tests) + 3); + ksft_print_msg("CPU %s FRED\n", cpu_has_fred() ? "supports" : "lacks"); + sethandler(SIGSEGV, handler, 0); + sethandler(SIGTRAP, handler, 0); + sethandler(SIGILL, handler, 0); + + for (tf = 0; tf < 2; tf++) { + for (i = 0; i < ARRAY_SIZE(tests); i++) { + active = &tests[i]; + single_step = tf; + seen = signo = trap = error = ip_offset = flags = 0; + code = addr_ok = stepped = step_ok = 0; + expected_flags = active->flags | (tf ? X86_EFLAGS_TF : 0); + if (tf) + active->run_tf(); + else + active->run(); + + ok = seen && signo == active->signo && trap == active->trap && + error == active->error && ip_offset == active->ip_offset && + flags == expected_flags && code == active->code && addr_ok && + (!tf || (stepped && step_ok)); + ksft_test_result(ok, "%s%s\n", active->name, tf ? " with TF" : ""); + if (!ok) { + ksft_print_msg("got signal=%d trap=%d error=%#x ip=%d\n", + signo, trap, error, ip_offset); + ksft_print_msg("got flags=%#x code=%d addr_ok=%d step_ok=%d\n", + flags, code, addr_ok, step_ok); + ksft_print_msg("expected signal=%d trap=%d error=%#x ip=%d\n", + active->signo, active->trap, active->error, + active->ip_offset); + ksft_print_msg("expected flags=%#x code=%d\n", + expected_flags, active->code); + } + } + } + test_ptrace(); + ksft_finished(); +} |
