From 2bfbb09b505c26803bd723ba227714cc041e8958 Mon Sep 17 00:00:00 2001 From: Joel Granados Date: Mon, 10 Aug 2026 11:51:27 +0200 Subject: sysctl: Split data conversion and file position handling Apply the conversions to the data in a new helper function (apply_conv_on_vec) while file position handling and argument validation stay in the original function. Rename function to proc_vec (from do_proc_vec). This is a prep commit to isolate the logic that needs to change to prevent partial sysctl vector writes. No functional change intended Reviewed-by: Bradley Morgan Signed-off-by: Joel Granados --- kernel/sysctl.c | 209 +++++++++++++++++++++++++++++++++----------------------- 1 file changed, 125 insertions(+), 84 deletions(-) (limited to 'kernel') diff --git a/kernel/sysctl.c b/kernel/sysctl.c index f7b75985d542..47a92cbbcb69 100644 --- a/kernel/sysctl.c +++ b/kernel/sysctl.c @@ -595,7 +595,7 @@ static int do_proc_int_conv_minmax(bool *negp, unsigned long *u_ptr, int *k_ptr, static const char proc_wspace_sep[] = { ' ', '\t', '\n' }; /* - * Element type processed by do_proc_vec(). The tag selects the element size + * Element type processed by proc_vec(). The tag selects the element size * and signedness, and it selects which member of union proc_vec_conv is live. */ enum proc_vec_type { @@ -605,7 +605,7 @@ enum proc_vec_type { }; /* - * Converter passed to do_proc_vec(). Only the member matching the + * Converter passed to proc_vec(). Only the member matching the * enum proc_vec_type tag is ever read, so every dispatch stays fully typed and * no void * converter pointer is needed. */ @@ -637,33 +637,106 @@ static int proc_vec_conv(enum proc_vec_type type, union proc_vec_conv conv, return -EINVAL; } -/* - * Read/write a vector of @type elements. The element size and signedness are - * derived from @type, so a single runtime function replaces the per-type - * variants. table->data is walked as raw bytes (@i) advanced by @size; the - * converter performs the actual typed load/store. +/** + * apply_conv_on_vec - Apply converter function on data vector + * + * @conv: The converter to be applied + * @table: The sysctl table + * @data_type: Type used in converter appliation (INT, UINT or ULONG) + * @data_size: Number of data elements. + * @conv_dir: %TRUE if this is a write to the sysctl file + * @buf_nbyte: Number of bytes for buf + * @buf: The user buffer + * @buf_left_final: Number of outstanding (non converted) bytes. + * + * Element signedness is derived from @data_type. table->data is walked + * as raw bytes (@data) advanced by @data_size; the converter performs + * the actual typed load/store. Sets buf_left_final to the number of + * bytes that where left outstanding after conversion. Can be > 0. + * + * Returns: %0 on success. Non-zero on error. */ -static int do_proc_vec(const struct ctl_table *table, int dir, - void *buffer, size_t *lenp, loff_t *ppos, - enum proc_vec_type type, union proc_vec_conv conv) +static int apply_conv_on_vec(const union proc_vec_conv conv, + const struct ctl_table *table, + const enum proc_vec_type data_type, + const size_t data_size, const int conv_dir, + const size_t buf_nbyte, void *buf, + size_t *buf_left_final) +{ + int vec_left, first = 1, err = 0; + size_t buf_left; + char *data, *p; + bool is_unsigned = data_type == PROC_VEC_UINT || data_type == PROC_VEC_ULONG; + + data = table->data; + vec_left = table->maxlen / data_size; + buf_left = buf_nbyte; + + if (SYSCTL_USER_TO_KERN(conv_dir)) { + if (buf_left > PAGE_SIZE - 1) + buf_left = PAGE_SIZE - 1; + p = buf; + } + + for (; buf_left && vec_left--; data += data_size, first = 0) { + unsigned long lval; + bool neg = false; + + if (SYSCTL_USER_TO_KERN(conv_dir)) { + proc_skip_spaces(&p, &buf_left); + + if (!buf_left) + break; + err = proc_get_long(&p, &buf_left, &lval, &neg, + proc_wspace_sep, + sizeof(proc_wspace_sep), NULL); + if (!err && neg && is_unsigned) + err = -EINVAL; + if (err) + break; + if (proc_vec_conv(data_type, conv, &neg, &lval, data, conv_dir, table)) { + err = -EINVAL; + break; + } + } else { + if (proc_vec_conv(data_type, conv, &neg, &lval, data, conv_dir, table)) { + err = -EINVAL; + break; + } + if (!first) + proc_put_char(&buf, &buf_left, '\t'); + proc_put_long(&buf, &buf_left, lval, neg); + } + } + + if (SYSCTL_KERN_TO_USER(conv_dir) && !first && buf_left && !err) + proc_put_char(&buf, &buf_left, '\n'); + if (SYSCTL_USER_TO_KERN(conv_dir) && !err && buf_left) + proc_skip_spaces(&p, &buf_left); + if (SYSCTL_USER_TO_KERN(conv_dir) && first) + return err ? : -EINVAL; + *buf_left_final = buf_left; + + return err; +} + +/* Read/write a vector of @type elements. */ +static int proc_vec(const struct ctl_table *table, int dir, void *buffer, + size_t *lenp, loff_t *ppos, enum proc_vec_type type, + union proc_vec_conv conv) { - int vleft, first = 1, err = 0; - size_t left, size; - bool is_unsigned; - char *i, *p; + int err = 0; + size_t data_size, left_nbyte = SIZE_MAX; switch (type) { case PROC_VEC_INT: - size = sizeof(int); - is_unsigned = false; + data_size = sizeof(int); break; case PROC_VEC_UINT: - size = sizeof(uint); - is_unsigned = true; + data_size = sizeof(uint); break; case PROC_VEC_ULONG: - size = sizeof(ulong); - is_unsigned = true; + data_size = sizeof(ulong); break; default: return -EINVAL; @@ -675,63 +748,31 @@ static int do_proc_vec(const struct ctl_table *table, int dir, return 0; } - i = table->data; - vleft = table->maxlen / size; - left = *lenp; - /* uint arrays are not supported, *Do not* add support for them. */ - if (type == PROC_VEC_UINT && vleft != 1) + if (type == PROC_VEC_UINT && (table->maxlen / data_size) != 1) return -EINVAL; if (SYSCTL_USER_TO_KERN(dir)) { if (proc_first_pos_non_zero_ignore(ppos, table)) goto out; - - if (left > PAGE_SIZE - 1) - left = PAGE_SIZE - 1; - p = buffer; } - for (; left && vleft--; i += size, first = 0) { - unsigned long lval; - bool neg = false; + err = apply_conv_on_vec(conv, table, type, data_size, dir, *lenp, buffer, + &left_nbyte); - if (SYSCTL_USER_TO_KERN(dir)) { - proc_skip_spaces(&p, &left); + /* + * An unchanged left_nbyte signals a write with no parsed element; which + * is an error. Using SIZE_MAX to detect this error is possible because: + * 1. lenp is bounded by KMALLOC_MAX_SIZE in proc_sys_call_handler + * 2. lenp could never be SIZE_MAX as it is a "ridiculous" (exabyte) allocation. + */ + if (left_nbyte == SIZE_MAX) + return err; - if (!left) - break; - err = proc_get_long(&p, &left, &lval, &neg, - proc_wspace_sep, - sizeof(proc_wspace_sep), NULL); - if (!err && neg && is_unsigned) - err = -EINVAL; - if (err) - break; - if (proc_vec_conv(type, conv, &neg, &lval, i, dir, table)) { - err = -EINVAL; - break; - } - } else { - if (proc_vec_conv(type, conv, &neg, &lval, i, dir, table)) { - err = -EINVAL; - break; - } - if (!first) - proc_put_char(&buffer, &left, '\t'); - proc_put_long(&buffer, &left, lval, neg); - } - } - - if (SYSCTL_KERN_TO_USER(dir) && !first && left && !err) - proc_put_char(&buffer, &left, '\n'); - if (SYSCTL_USER_TO_KERN(dir) && !err && left) - proc_skip_spaces(&p, &left); - if (SYSCTL_USER_TO_KERN(dir) && first) - return err ? : -EINVAL; - *lenp -= left; + *lenp -= left_nbyte; out: *ppos += *lenp; + return err; } @@ -760,8 +801,8 @@ int proc_douintvec_conv(const struct ctl_table *table, int dir, void *buffer, if (!conv) conv = do_proc_uint_conv; - return do_proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_UINT, - (union proc_vec_conv){ .uint_conv = conv }); + return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_UINT, + (union proc_vec_conv){ .uint_conv = conv }); } /** @@ -820,8 +861,8 @@ int proc_dobool(const struct ctl_table *table, int dir, void *buffer, int proc_dointvec(const struct ctl_table *table, int dir, void *buffer, size_t *lenp, loff_t *ppos) { - return do_proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_INT, - (union proc_vec_conv){ .int_conv = do_proc_int_conv }); + return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_INT, + (union proc_vec_conv){ .int_conv = do_proc_int_conv }); } /** @@ -840,8 +881,8 @@ int proc_dointvec(const struct ctl_table *table, int dir, void *buffer, int proc_douintvec(const struct ctl_table *table, int dir, void *buffer, size_t *lenp, loff_t *ppos) { - return do_proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_UINT, - (union proc_vec_conv){ .uint_conv = do_proc_uint_conv }); + return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_UINT, + (union proc_vec_conv){ .uint_conv = do_proc_uint_conv }); } /** @@ -864,8 +905,8 @@ int proc_douintvec(const struct ctl_table *table, int dir, void *buffer, int proc_dointvec_minmax(const struct ctl_table *table, int dir, void *buffer, size_t *lenp, loff_t *ppos) { - return do_proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_INT, - (union proc_vec_conv){ .int_conv = do_proc_int_conv_minmax }); + return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_INT, + (union proc_vec_conv){ .int_conv = do_proc_int_conv_minmax }); } /** @@ -891,8 +932,8 @@ int proc_dointvec_minmax(const struct ctl_table *table, int dir, int proc_douintvec_minmax(const struct ctl_table *table, int dir, void *buffer, size_t *lenp, loff_t *ppos) { - return do_proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_UINT, - (union proc_vec_conv){ .uint_conv = do_proc_uint_conv_minmax }); + return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_UINT, + (union proc_vec_conv){ .uint_conv = do_proc_uint_conv_minmax }); } /** @@ -935,8 +976,8 @@ int proc_dou8vec_minmax(const struct ctl_table *table, int dir, tmp.extra2 = (unsigned int *) &max; val = READ_ONCE(*data); - res = do_proc_vec(&tmp, dir, buffer, lenp, ppos, PROC_VEC_UINT, - (union proc_vec_conv){ .uint_conv = do_proc_uint_conv_minmax }); + res = proc_vec(&tmp, dir, buffer, lenp, ppos, PROC_VEC_UINT, + (union proc_vec_conv){ .uint_conv = do_proc_uint_conv_minmax }); if (res) return res; if (SYSCTL_USER_TO_KERN(dir)) @@ -1066,8 +1107,8 @@ int proc_doulongvec_conv(const struct ctl_table *table, int dir, int (*conv)(bool *negp, ulong *u_ptr, ulong *k_ptr, int dir, const struct ctl_table *table)) { - return do_proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_ULONG, - (union proc_vec_conv){ .ulong_conv = conv }); + return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_ULONG, + (union proc_vec_conv){ .ulong_conv = conv }); } /** @@ -1087,10 +1128,10 @@ int proc_doulongvec_conv(const struct ctl_table *table, int dir, * Returns: %0 on success. */ int proc_doulongvec_minmax(const struct ctl_table *table, int dir, - void *buffer, size_t *lenp, loff_t *ppos) + void *buffer, size_t *lenp, loff_t *ppos) { - return do_proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_ULONG, - (union proc_vec_conv){ .ulong_conv = do_proc_ulong_conv }); + return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_ULONG, + (union proc_vec_conv){ .ulong_conv = do_proc_ulong_conv }); } /** @@ -1114,8 +1155,8 @@ int proc_dointvec_conv(const struct ctl_table *table, int dir, void *buffer, { if (!conv) conv = do_proc_int_conv; - return do_proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_INT, - (union proc_vec_conv){ .int_conv = conv }); + return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_INT, + (union proc_vec_conv){ .int_conv = conv }); } /** -- cgit v1.2.3 From d08ab285b0968c96ee33bc66c6a3c5a348a45d62 Mon Sep 17 00:00:00 2001 From: Joel Granados Date: Tue, 11 Aug 2026 12:37:30 +0200 Subject: sysctl: Reject uint arrays before calling the general proc_vec Move the UINT vector size check to proc_douintvec_conv; the function that routes UINT types only. Route all the UINT calls (including proc_dou8vec_minmax) through proc_douintvec_conv. UINT proc handlers that incorrectly define maxlen will now return -EINVAL instead of 0 in the cases where data is missing, lenp is 0 or ppos is 0. Note that maxlen == 0 is not considered as miss-defined. Reviewed-by: Bradley Morgan Signed-off-by: Joel Granados --- kernel/sysctl.c | 17 +++++++---------- 1 file changed, 7 insertions(+), 10 deletions(-) (limited to 'kernel') diff --git a/kernel/sysctl.c b/kernel/sysctl.c index 47a92cbbcb69..7e9024899be6 100644 --- a/kernel/sysctl.c +++ b/kernel/sysctl.c @@ -748,10 +748,6 @@ static int proc_vec(const struct ctl_table *table, int dir, void *buffer, return 0; } - /* uint arrays are not supported, *Do not* add support for them. */ - if (type == PROC_VEC_UINT && (table->maxlen / data_size) != 1) - return -EINVAL; - if (SYSCTL_USER_TO_KERN(dir)) { if (proc_first_pos_non_zero_ignore(ppos, table)) goto out; @@ -797,6 +793,9 @@ int proc_douintvec_conv(const struct ctl_table *table, int dir, void *buffer, int (*conv)(bool *negp, ulong *u_ptr, uint *k_ptr, int dir, const struct ctl_table *table)) { + /* uint arrays are not supported, *Do not* add support for them. */ + if (table->maxlen && (table->maxlen / sizeof(uint)) != 1) + return -EINVAL; if (!conv) conv = do_proc_uint_conv; @@ -881,8 +880,7 @@ int proc_dointvec(const struct ctl_table *table, int dir, void *buffer, int proc_douintvec(const struct ctl_table *table, int dir, void *buffer, size_t *lenp, loff_t *ppos) { - return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_UINT, - (union proc_vec_conv){ .uint_conv = do_proc_uint_conv }); + return proc_douintvec_conv(table, dir, buffer, lenp, ppos, do_proc_uint_conv); } /** @@ -932,8 +930,8 @@ int proc_dointvec_minmax(const struct ctl_table *table, int dir, int proc_douintvec_minmax(const struct ctl_table *table, int dir, void *buffer, size_t *lenp, loff_t *ppos) { - return proc_vec(table, dir, buffer, lenp, ppos, PROC_VEC_UINT, - (union proc_vec_conv){ .uint_conv = do_proc_uint_conv_minmax }); + return proc_douintvec_conv(table, dir, buffer, lenp, ppos, + do_proc_uint_conv_minmax); } /** @@ -976,8 +974,7 @@ int proc_dou8vec_minmax(const struct ctl_table *table, int dir, tmp.extra2 = (unsigned int *) &max; val = READ_ONCE(*data); - res = proc_vec(&tmp, dir, buffer, lenp, ppos, PROC_VEC_UINT, - (union proc_vec_conv){ .uint_conv = do_proc_uint_conv_minmax }); + res = proc_douintvec_minmax(&tmp, dir, buffer, lenp, ppos); if (res) return res; if (SYSCTL_USER_TO_KERN(dir)) -- cgit v1.2.3 From 242bac52294e437218f5815b16d3de984bb55593 Mon Sep 17 00:00:00 2001 From: Joel Granados Date: Thu, 13 Aug 2026 08:29:18 +0200 Subject: sysctl: Disallow partial updates for erroneous sysctl vectors When updating the kernel sysctl vectors there is a chance that not all vector elements are updated due to erroneous input. Use a staging variable that holds a copy of the vector and commits to the actual table->data only when all input is successfully updated. This does **not** make the write atomic as a reader can still see a partially updated vector. The staging is only for vectors; cases where table->data points to a variable should not be staged as they will not be updated on input error. PROC_VEC_UINT is not included because UINT arrays are not allowed. Replace first with nr_conv, incremented where first was cleared. first is exactly nr_conv == 0, and the counter doubles as the number of elements to publish. Example of behavior that is being prevented: # echo "4 4 1 7" > /proc/sys/kernel/printk # echo "1 x" > /proc/sys/kernel/printk -bash: echo: write error: Invalid argument # cat /proc/sys/kernel/printk 1 4 1 7 <- incorrect It should be unchanged ("4 4 1 7") on error. Link: https://lore.kernel.org/all/tencent_A860C873956A52E26AD8D309A308A241BA08@qq.com/ Reviewed-by: Bradley Morgan Signed-off-by: Joel Granados --- kernel/sysctl.c | 65 +++++++++++++++++++++++++++++++++++++++++++++------------ 1 file changed, 52 insertions(+), 13 deletions(-) (limited to 'kernel') diff --git a/kernel/sysctl.c b/kernel/sysctl.c index 7e9024899be6..25577ffc2801 100644 --- a/kernel/sysctl.c +++ b/kernel/sysctl.c @@ -637,6 +637,26 @@ static int proc_vec_conv(enum proc_vec_type type, union proc_vec_conv conv, return -EINVAL; } +static int commit_conv_vec(const enum proc_vec_type data_type, void *dst, + const void *src, size_t nr) +{ + size_t i; + + switch (data_type) { + case PROC_VEC_INT: + for (i = 0; i < nr; i++) + WRITE_ONCE(((int *)dst)[i], ((const int *)src)[i]); + return 0; + + case PROC_VEC_ULONG: + for (i = 0; i < nr; i++) + WRITE_ONCE(((ulong *)dst)[i], ((const ulong *)src)[i]); + return 0; + default: + return -EINVAL; + } +} + /** * apply_conv_on_vec - Apply converter function on data vector * @@ -663,22 +683,31 @@ static int apply_conv_on_vec(const union proc_vec_conv conv, const size_t buf_nbyte, void *buf, size_t *buf_left_final) { - int vec_left, first = 1, err = 0; - size_t buf_left; - char *data, *p; + int vec_left, err = 0; + size_t buf_left, nr_conv = 0; + char *data, *data_stage = NULL, *p; bool is_unsigned = data_type == PROC_VEC_UINT || data_type == PROC_VEC_ULONG; - data = table->data; - vec_left = table->maxlen / data_size; buf_left = buf_nbyte; + data = table->data; if (SYSCTL_USER_TO_KERN(conv_dir)) { if (buf_left > PAGE_SIZE - 1) buf_left = PAGE_SIZE - 1; p = buf; + + if (table->maxlen > data_size) { + data_stage = kmemdup(table->data, table->maxlen, GFP_KERNEL); + if (!data_stage) { + err = -ENOMEM; + goto out; + } + data = data_stage; + } } - for (; buf_left && vec_left--; data += data_size, first = 0) { + vec_left = table->maxlen / data_size; + for (; buf_left && vec_left--; data += data_size, nr_conv++) { unsigned long lval; bool neg = false; @@ -703,20 +732,30 @@ static int apply_conv_on_vec(const union proc_vec_conv conv, err = -EINVAL; break; } - if (!first) + if (nr_conv) proc_put_char(&buf, &buf_left, '\t'); proc_put_long(&buf, &buf_left, lval, neg); } } - if (SYSCTL_KERN_TO_USER(conv_dir) && !first && buf_left && !err) - proc_put_char(&buf, &buf_left, '\n'); - if (SYSCTL_USER_TO_KERN(conv_dir) && !err && buf_left) - proc_skip_spaces(&p, &buf_left); - if (SYSCTL_USER_TO_KERN(conv_dir) && first) - return err ? : -EINVAL; + if (SYSCTL_USER_TO_KERN(conv_dir)) { + if (!err && buf_left) + proc_skip_spaces(&p, &buf_left); + if (!nr_conv) { + err = err ? : -EINVAL; + goto out; + } + if (!err && data_stage) + err = commit_conv_vec(data_type, table->data, data_stage, nr_conv); + } else { + if (nr_conv && buf_left && !err) + proc_put_char(&buf, &buf_left, '\n'); + } + *buf_left_final = buf_left; +out: + kfree(data_stage); return err; } -- cgit v1.2.3 From acd19f2c56f20725fde8bc60067d165634dda932 Mon Sep 17 00:00:00 2001 From: Oleg Nesterov Date: Sun, 30 Aug 2026 17:33:33 +0200 Subject: sysctl: collapse redundant CONFIG_SYSCTL nesting in kernel/sysctl.c After the removal of CONFIG_PROC_SYSCTL, several #ifdef CONFIG_SYSCTL blocks ended up nested inside the outer (now the same) CONFIG_SYSCTL guards. Collapse them and remove a now-stale "/proc/sys support" comment which referred to CONFIG_PROC_SYSCTL. Signed-off-by: Oleg Nesterov Reviewed-by: Bradley Morgan Signed-off-by: Joel Granados --- kernel/sysctl.c | 16 ++-------------- 1 file changed, 2 insertions(+), 14 deletions(-) (limited to 'kernel') diff --git a/kernel/sysctl.c b/kernel/sysctl.c index 25577ffc2801..4efe084455c2 100644 --- a/kernel/sysctl.c +++ b/kernel/sysctl.c @@ -29,14 +29,12 @@ EXPORT_SYMBOL(sysctl_vals); const unsigned long sysctl_long_vals[] = { 0, 1, LONG_MAX }; EXPORT_SYMBOL_GPL(sysctl_long_vals); -#if defined(CONFIG_SYSCTL) +#ifdef CONFIG_SYSCTL /* Constants used for minimum and maximum */ static const int ngroups_max = NGROUPS_MAX; static const int cap_last_cap = CAP_LAST_CAP; -#ifdef CONFIG_SYSCTL - /** * enum sysctl_writes_mode - supported sysctl write modes * @@ -64,14 +62,6 @@ enum sysctl_writes_mode { }; static enum sysctl_writes_mode sysctl_writes_strict = SYSCTL_WRITES_STRICT; -#endif /* CONFIG_SYSCTL */ -#endif /* CONFIG_SYSCTL */ - -/* - * /proc/sys support - */ - -#ifdef CONFIG_SYSCTL static int _proc_do_string(char *data, int maxlen, int dir, char *buffer, size_t *lenp, loff_t *ppos) @@ -1443,7 +1433,7 @@ int proc_do_large_bitmap(const struct ctl_table *table, int dir, #endif /* CONFIG_SYSCTL */ -#if defined(CONFIG_SYSCTL) +#ifdef CONFIG_SYSCTL int proc_do_static_key(const struct ctl_table *table, int dir, void *buffer, size_t *lenp, loff_t *ppos) { @@ -1475,7 +1465,6 @@ int proc_do_static_key(const struct ctl_table *table, int dir, } static const struct ctl_table sysctl_subsys_table[] = { -#ifdef CONFIG_SYSCTL { .procname = "sysctl_writes_strict", .data = &sysctl_writes_strict, @@ -1485,7 +1474,6 @@ static const struct ctl_table sysctl_subsys_table[] = { .extra1 = SYSCTL_NEG_ONE, .extra2 = SYSCTL_ONE, }, -#endif { .procname = "ngroups_max", .data = (void *)&ngroups_max, -- cgit v1.2.3 From 443bea2f71c534ff482c4383cd0d7004d128c9bd Mon Sep 17 00:00:00 2001 From: Oleg Nesterov Date: Sun, 30 Aug 2026 17:33:38 +0200 Subject: sysctl: consolidate CONFIG_SYSCTL into a single block in kernel/sysctl.c After the previous cleanup there are two CONFIG_SYSCTL blocks. Move the second block up into the first one and eliminate the redundant Also add a proc_do_static_key() stub to the #else block. Not strictly necessary today, but consistent with the other stubs declared in include/linux/sysctl.h. Signed-off-by: Oleg Nesterov Signed-off-by: Joel Granados --- kernel/sysctl.c | 160 +++++++++++++++++++++++++++++--------------------------- 1 file changed, 82 insertions(+), 78 deletions(-) (limited to 'kernel') diff --git a/kernel/sysctl.c b/kernel/sysctl.c index 4efe084455c2..f001908d534e 100644 --- a/kernel/sysctl.c +++ b/kernel/sysctl.c @@ -1332,6 +1332,87 @@ int proc_do_large_bitmap(const struct ctl_table *table, int dir, return err; } +int proc_do_static_key(const struct ctl_table *table, int dir, + void *buffer, size_t *lenp, loff_t *ppos) +{ + struct static_key *key = (struct static_key *)table->data; + static DEFINE_MUTEX(static_key_mutex); + int val, ret; + struct ctl_table tmp = { + .data = &val, + .maxlen = sizeof(val), + .mode = table->mode, + .extra1 = SYSCTL_ZERO, + .extra2 = SYSCTL_ONE, + }; + + if (SYSCTL_USER_TO_KERN(dir) && !capable(CAP_SYS_ADMIN)) + return -EPERM; + + mutex_lock(&static_key_mutex); + val = static_key_enabled(key); + ret = proc_dointvec_minmax(&tmp, dir, buffer, lenp, ppos); + if (SYSCTL_USER_TO_KERN(dir) && !ret) { + if (val) + static_key_enable(key); + else + static_key_disable(key); + } + mutex_unlock(&static_key_mutex); + return ret; +} + +static const struct ctl_table sysctl_subsys_table[] = { + { + .procname = "sysctl_writes_strict", + .data = &sysctl_writes_strict, + .maxlen = sizeof(int), + .mode = 0644, + .proc_handler = proc_dointvec_minmax, + .extra1 = SYSCTL_NEG_ONE, + .extra2 = SYSCTL_ONE, + }, + { + .procname = "ngroups_max", + .data = (void *)&ngroups_max, + .maxlen = sizeof (int), + .mode = 0444, + .proc_handler = proc_dointvec, + }, + { + .procname = "cap_last_cap", + .data = (void *)&cap_last_cap, + .maxlen = sizeof(int), + .mode = 0444, + .proc_handler = proc_dointvec, + }, +#ifdef CONFIG_SYSCTL_ARCH_UNALIGN_ALLOW + { + .procname = "unaligned-trap", + .data = &unaligned_enabled, + .maxlen = sizeof(int), + .mode = 0644, + .proc_handler = proc_dointvec, + }, +#endif +#ifdef CONFIG_SYSCTL_ARCH_UNALIGN_NO_WARN + { + .procname = "ignore-unaligned-usertrap", + .data = &no_unaligned_warning, + .maxlen = sizeof (int), + .mode = 0644, + .proc_handler = proc_dointvec, + }, +#endif +}; + +int __init sysctl_init_bases(void) +{ + register_sysctl_init("kernel", sysctl_subsys_table); + + return 0; +} + #else /* CONFIG_SYSCTL */ int proc_dostring(const struct ctl_table *table, int dir, @@ -1431,89 +1512,12 @@ int proc_do_large_bitmap(const struct ctl_table *table, int dir, return -ENOSYS; } -#endif /* CONFIG_SYSCTL */ - -#ifdef CONFIG_SYSCTL int proc_do_static_key(const struct ctl_table *table, int dir, void *buffer, size_t *lenp, loff_t *ppos) { - struct static_key *key = (struct static_key *)table->data; - static DEFINE_MUTEX(static_key_mutex); - int val, ret; - struct ctl_table tmp = { - .data = &val, - .maxlen = sizeof(val), - .mode = table->mode, - .extra1 = SYSCTL_ZERO, - .extra2 = SYSCTL_ONE, - }; - - if (SYSCTL_USER_TO_KERN(dir) && !capable(CAP_SYS_ADMIN)) - return -EPERM; - - mutex_lock(&static_key_mutex); - val = static_key_enabled(key); - ret = proc_dointvec_minmax(&tmp, dir, buffer, lenp, ppos); - if (SYSCTL_USER_TO_KERN(dir) && !ret) { - if (val) - static_key_enable(key); - else - static_key_disable(key); - } - mutex_unlock(&static_key_mutex); - return ret; + return -ENOSYS; } -static const struct ctl_table sysctl_subsys_table[] = { - { - .procname = "sysctl_writes_strict", - .data = &sysctl_writes_strict, - .maxlen = sizeof(int), - .mode = 0644, - .proc_handler = proc_dointvec_minmax, - .extra1 = SYSCTL_NEG_ONE, - .extra2 = SYSCTL_ONE, - }, - { - .procname = "ngroups_max", - .data = (void *)&ngroups_max, - .maxlen = sizeof (int), - .mode = 0444, - .proc_handler = proc_dointvec, - }, - { - .procname = "cap_last_cap", - .data = (void *)&cap_last_cap, - .maxlen = sizeof(int), - .mode = 0444, - .proc_handler = proc_dointvec, - }, -#ifdef CONFIG_SYSCTL_ARCH_UNALIGN_ALLOW - { - .procname = "unaligned-trap", - .data = &unaligned_enabled, - .maxlen = sizeof(int), - .mode = 0644, - .proc_handler = proc_dointvec, - }, -#endif -#ifdef CONFIG_SYSCTL_ARCH_UNALIGN_NO_WARN - { - .procname = "ignore-unaligned-usertrap", - .data = &no_unaligned_warning, - .maxlen = sizeof (int), - .mode = 0644, - .proc_handler = proc_dointvec, - }, -#endif -}; - -int __init sysctl_init_bases(void) -{ - register_sysctl_init("kernel", sysctl_subsys_table); - - return 0; -} #endif /* CONFIG_SYSCTL */ /* * No sense putting this after each symbol definition, twice, -- cgit v1.2.3 From b60237dc2146b45b0a6f72d2645220bd9fb4cb81 Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Mon, 28 Sep 2026 10:02:04 +0800 Subject: sysctl: Negate before converting in the int read path proc_int_k2u_conv_kop() reports the sign through *negp and the magnitude through *u_ptr, but for a negative value it hands the sign-extended int to the converter and negates the result: *u_ptr = k_ptr_op ? -k_ptr_op((ulong)val) : -(ulong)val; With div_hz() and CONFIG_HZ=1000 a stored -1000 becomes (ulong)-1000 / 1000 == 18446744073709550, and negating that wraps: # echo -1 > /proc/sys/net/ipv4/tcp_fin_timeout # cat /proc/sys/net/ipv4/tcp_fin_timeout -18428297329635842066 The magnitude is HZ dependent but the wrap is not: the negation happens after the division at every CONFIG_HZ. Take the magnitude first and convert that, which is what the open-coded version did before commit 2dc164a48e6f ("sysctl: Create converter functions with two new macros") folded it into a macro. The k_ptr_op == NULL branch was already correct. All three int converters that pass a k_ptr_op are affected: proc_dointvec_jiffies(), proc_dointvec_userhz_jiffies() and proc_dointvec_ms_jiffies(). Fixes: 2dc164a48e6f ("sysctl: Create converter functions with two new macros") Cc: stable@vger.kernel.org Reviewed-by: Bradley Morgan Signed-off-by: Zhan Xusheng Signed-off-by: Joel Granados --- kernel/sysctl.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) (limited to 'kernel') diff --git a/kernel/sysctl.c b/kernel/sysctl.c index f001908d534e..f5145b32f6be 100644 --- a/kernel/sysctl.c +++ b/kernel/sysctl.c @@ -473,7 +473,7 @@ int proc_int_k2u_conv_kop(ulong *u_ptr, const int *k_ptr, bool *negp, if (val < 0) { *negp = true; - *u_ptr = k_ptr_op ? -k_ptr_op((ulong)val) : -(ulong)val; + *u_ptr = k_ptr_op ? k_ptr_op(-(ulong)val) : -(ulong)val; } else { *negp = false; *u_ptr = k_ptr_op ? k_ptr_op((ulong)val) : (ulong) val; -- cgit v1.2.3 From 4991c8b72b564cd16cb48124f880617fd49f1013 Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Mon, 28 Sep 2026 10:02:05 +0800 Subject: time/jiffies: Saturate in mult_hz() instead of wrapping mult_hz() converts a user-supplied seconds value to jiffies for proc_dointvec_jiffies(). proc_int_u2k_conv_uop() rejects a result above INT_MAX, but it inspects the product, so a product that wraps arrives as a small value and is stored. The input has to exceed ULONG_MAX / HZ for the product to wrap, so the value below is specific to CONFIG_HZ=1000: # echo 18446744073709552 > /proc/sys/net/ipv4/tcp_keepalive_time # cat /proc/sys/net/ipv4/tcp_keepalive_time 0 18446744073709551, one less, is correctly rejected. Dozens of sysctls use proc_dointvec_jiffies(), among them tcp_keepalive_time, tcp_fin_timeout and the conntrack timeouts. Bound the input in the shape clock_t_to_jiffies() already uses and leave the INT_MAX policy to the caller. The bound was open-coded as "*lvalp > INT_MAX / HZ" until commit 2dc164a48e6f ("sysctl: Create converter functions with two new macros"). Fixes: 2dc164a48e6f ("sysctl: Create converter functions with two new macros") Cc: stable@vger.kernel.org Signed-off-by: Zhan Xusheng Signed-off-by: Joel Granados --- kernel/time/jiffies.c | 2 ++ 1 file changed, 2 insertions(+) (limited to 'kernel') diff --git a/kernel/time/jiffies.c b/kernel/time/jiffies.c index 213ae1d6a014..f42a6503fdd3 100644 --- a/kernel/time/jiffies.c +++ b/kernel/time/jiffies.c @@ -101,6 +101,8 @@ void __init register_refined_jiffies(long cycles_per_second) #ifdef CONFIG_SYSCTL static ulong mult_hz(const ulong val) { + if (val >= ULONG_MAX / HZ) + return ULONG_MAX; return val * HZ; } -- cgit v1.2.3