summaryrefslogtreecommitdiff
path: root/include
diff options
context:
space:
mode:
authorMark Brown <broonie@kernel.org>2026-09-30 11:58:30 +0100
committerMark Brown <broonie@kernel.org>2026-09-30 11:58:30 +0100
commit5eb3fdca6ca7667a73d52adfd047d94e8b8b105e (patch)
treec935b3fa5b47c9d70e0e66c7cc7830e74988b13f /include
parentf60deca1c3741ab8fcf318a7bb82a3d026ea7c49 (diff)
parent84086827932b58e7645d93d970bbc566c4ee408b (diff)
downloadlinux-next-5eb3fdca6ca7667a73d52adfd047d94e8b8b105e.tar.gz
linux-next-5eb3fdca6ca7667a73d52adfd047d94e8b8b105e.zip
Merge branch 'vfs.all' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git
# Conflicts: # fs/smb/server/smb2pdu.c # fs/smb/server/vfs.c # fs/smb/server/vfs.h
Diffstat (limited to 'include')
-rw-r--r--include/linux/binfmts.h3
-rw-r--r--include/linux/bio-integrity.h3
-rw-r--r--include/linux/bio.h11
-rw-r--r--include/linux/blkdev.h8
-rw-r--r--include/linux/buffer_head.h86
-rw-r--r--include/linux/capability.h8
-rw-r--r--include/linux/cleanup.h7
-rw-r--r--include/linux/coredump.h37
-rw-r--r--include/linux/dax.h12
-rw-r--r--include/linux/dcache.h38
-rw-r--r--include/linux/fdtable.h15
-rw-r--r--include/linux/file.h130
-rw-r--r--include/linux/fileattr.h2
-rw-r--r--include/linux/fs.h127
-rw-r--r--include/linux/fs_context.h4
-rw-r--r--include/linux/fscache-cache.h2
-rw-r--r--include/linux/fscache.h53
-rw-r--r--include/linux/iomap.h38
-rw-r--r--include/linux/lsm_hook_defs.h25
-rw-r--r--include/linux/mnt_idmapping.h24
-rw-r--r--include/linux/mount.h4
-rw-r--r--include/linux/namei.h19
-rw-r--r--include/linux/netfs.h112
-rw-r--r--include/linux/nfs_fs.h6
-rw-r--r--include/linux/posix_acl.h24
-rw-r--r--include/linux/quotaops.h6
-rw-r--r--include/linux/sched.h2
-rw-r--r--include/linux/sched/signal.h33
-rw-r--r--include/linux/security.h61
-rw-r--r--include/linux/splice.h4
-rw-r--r--include/linux/uidgid.h12
-rw-r--r--include/linux/user_namespace.h11
-rw-r--r--include/linux/wait_bit.h26
-rw-r--r--include/linux/xattr.h20
-rw-r--r--include/trace/events/cachefiles.h81
-rw-r--r--include/trace/events/fscache.h10
-rw-r--r--include/trace/events/netfs.h145
-rw-r--r--include/uapi/linux/close_range.h31
-rw-r--r--include/uapi/linux/coredump.h149
-rw-r--r--include/uapi/linux/fs.h2
40 files changed, 907 insertions, 484 deletions
diff --git a/include/linux/binfmts.h b/include/linux/binfmts.h
index f686a37f7a0a..2e87faf9a8c2 100644
--- a/include/linux/binfmts.h
+++ b/include/linux/binfmts.h
@@ -128,7 +128,8 @@ struct linux_binfmt {
struct module *module;
int (*load_binary)(struct linux_binprm *);
#ifdef CONFIG_COREDUMP
- int (*core_dump)(struct coredump_params *cprm);
+ /* Returns true if the whole coredump was written. */
+ bool (*core_dump)(struct coredump_params *cprm);
unsigned long min_coredump; /* minimal dump size */
#endif
} __randomize_layout;
diff --git a/include/linux/bio-integrity.h b/include/linux/bio-integrity.h
index 0ea2a8bf7efb..a954c97be0b3 100644
--- a/include/linux/bio-integrity.h
+++ b/include/linux/bio-integrity.h
@@ -151,7 +151,6 @@ void bio_integrity_setup_default(struct bio *bio);
unsigned int fs_bio_integrity_alloc(struct bio *bio);
void fs_bio_integrity_free(struct bio *bio);
void fs_bio_integrity_generate(struct bio *bio);
-int fs_bio_integrity_verify(struct bio *bio, sector_t sector,
- unsigned int size);
+int fs_bio_integrity_verify(struct bio *bio, struct bvec_iter *data_iter);
#endif /* _LINUX_BIO_INTEGRITY_H */
diff --git a/include/linux/bio.h b/include/linux/bio.h
index bb3235497e67..17944e44b584 100644
--- a/include/linux/bio.h
+++ b/include/linux/bio.h
@@ -479,6 +479,7 @@ static inline void bio_init_inline(struct bio *bio, struct block_device *bdev,
extern void bio_uninit(struct bio *);
void bio_reset(struct bio *bio, struct block_device *bdev, blk_opf_t opf);
void bio_reuse(struct bio *bio, blk_opf_t opf);
+void bio_prepare_reissue(struct bio *bio, struct block_device *bdev);
void bio_chain(struct bio *, struct bio *);
void bio_await(struct bio *bio, void *priv,
void (*submit)(struct bio *bio, void *priv));
@@ -516,16 +517,18 @@ int bdev_rw_virt(struct block_device *bdev, sector_t sector, void *data,
size_t len, enum req_op op);
int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter,
- unsigned mem_align_mask, unsigned len_align_mask);
+ unsigned maxlen, unsigned mem_align_mask,
+ unsigned len_align_mask);
bool bio_iov_iter_set(struct bio *bio, const struct iov_iter *iter);
void __bio_release_pages(struct bio *bio, bool mark_dirty);
extern void bio_set_pages_dirty(struct bio *bio);
extern void bio_check_pages_dirty(struct bio *bio);
-int bio_iov_iter_bounce(struct bio *bio, struct iov_iter *iter, size_t maxlen,
- size_t minsize);
-void bio_iov_iter_unbounce(struct bio *bio, bool is_error, bool mark_dirty);
+int bio_alloc_bounce_folios(struct bio *bio, size_t total_len, size_t minsize);
+void bio_free_folios(struct bio *bio);
+int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter,
+ size_t maxlen, size_t minsize);
extern void bio_copy_data(struct bio *dst, struct bio *src);
extern void bio_free_pages(struct bio *bio);
diff --git a/include/linux/blkdev.h b/include/linux/blkdev.h
index 4f7905c3412b..098a65f3e48b 100644
--- a/include/linux/blkdev.h
+++ b/include/linux/blkdev.h
@@ -1816,9 +1816,11 @@ static inline int bio_split_rw_at(struct bio *bio,
*/
static inline unsigned int max_integrity_io_size(struct queue_limits *lim)
{
- return min_t(unsigned int, lim->max_segment_size,
- (BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size) <<
- lim->integrity.interval_exp);
+ u64 max_intervals;
+
+ max_intervals = BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size;
+ return min_t(u64, lim->max_segment_size,
+ max_intervals << lim->integrity.interval_exp);
}
#define DEFINE_IO_COMP_BATCH(name) struct io_comp_batch name = { }
diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h
index fd2c7115c054..e7a701ce029d 100644
--- a/include/linux/buffer_head.h
+++ b/include/linux/buffer_head.h
@@ -59,10 +59,7 @@ struct address_space;
struct buffer_head {
unsigned long b_state; /* buffer state bitmap (see above) */
struct buffer_head *b_this_page;/* circular list of page's buffers */
- union {
- struct page *b_page; /* the page this bh is mapped to */
- struct folio *b_folio; /* the folio this bh is mapped to */
- };
+ struct folio *b_folio; /* the folio this bh is mapped to */
sector_t b_blocknr; /* start block number */
size_t b_size; /* size of mapping */
@@ -172,7 +169,36 @@ static __always_inline int buffer_uptodate(const struct buffer_head *bh)
static inline unsigned long bh_offset(const struct buffer_head *bh)
{
- return (unsigned long)(bh)->b_data & (page_size(bh->b_page) - 1);
+ return (unsigned long)(bh)->b_data & (folio_size(bh->b_folio) - 1);
+}
+
+/**
+ * kmap_local_bh - Map the data of a buffer.
+ * @bh: The buffer.
+ *
+ * Buffers usually live in the page cache, but a few are built over memory
+ * which is not. Those carry no folio and b_data is already a kernel address
+ * which is always mapped, so there is nothing to do for them. Pair with
+ * kunmap_local_bh().
+ *
+ * Return: A pointer to the buffer's data.
+ */
+static inline void *kmap_local_bh(const struct buffer_head *bh)
+{
+ if (!bh->b_folio)
+ return bh->b_data;
+ return kmap_local_folio(bh->b_folio, bh_offset(bh));
+}
+
+/**
+ * kunmap_local_bh - Unmap the data of a buffer.
+ * @bh: The buffer.
+ * @addr: The address returned by kmap_local_bh().
+ */
+static inline void kunmap_local_bh(const struct buffer_head *bh, void *addr)
+{
+ if (bh->b_folio)
+ kunmap_local(addr);
}
/* If we *know* page->private refers to buffer_heads */
@@ -338,20 +364,58 @@ static inline void bforget(struct buffer_head *bh)
__bforget(bh);
}
-static inline struct buffer_head *
-sb_bread(struct super_block *sb, sector_t block)
+/**
+ * sb_bread - Read a block.
+ * @sb: The superblock to read from.
+ * @block: Block number in units of block size.
+ *
+ * Read a specified block, and return the buffer head that refers
+ * to it. The memory is allocated from the movable area so that it can
+ * be migrated. The returned buffer head has its refcount increased.
+ * The caller should call brelse() when it has finished with the buffer.
+ *
+ * Context: May sleep waiting for I/O.
+ * Return: NULL if the block was unreadable.
+ */
+static inline
+struct buffer_head *sb_bread(struct super_block *sb, sector_t block)
{
return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, __GFP_MOVABLE);
}
-static inline struct buffer_head *
-sb_bread_unmovable(struct super_block *sb, sector_t block)
+/**
+ * sb_bread_unmovable - Read a block.
+ * @sb: The superblock to read from.
+ * @block: Block number in units of block size.
+ *
+ * Read a specified block, and return the buffer head that refers to it.
+ * The memory is allocated from the unmovable area so that pointers into
+ * it remain valid after compaction runs. The returned buffer head has
+ * its refcount increased. The caller should call brelse() when it has
+ * finished with the buffer.
+ *
+ * Context: May sleep waiting for I/O.
+ * Return: NULL if the block was unreadable.
+ */
+static inline
+struct buffer_head *sb_bread_unmovable(struct super_block *sb, sector_t block)
{
return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, 0);
}
-static inline void
-sb_breadahead(struct super_block *sb, sector_t block)
+/**
+ * sb_breadahead - Start readahead.
+ * @sb: Superblock identifying the block device.
+ * @block: The block to read.
+ *
+ * Read this block. The I/O will be flagged as being readahead rather
+ * than immediate read, but (unlike the page cache), surrounding blocks
+ * will not be read.
+ *
+ * Context: May sleep in order to allocate memory.
+ */
+static inline
+void sb_breadahead(struct super_block *sb, sector_t block)
{
__breadahead(sb->s_bdev, block, sb->s_blocksize);
}
diff --git a/include/linux/capability.h b/include/linux/capability.h
index f8532d92fcad..622137f66f09 100644
--- a/include/linux/capability.h
+++ b/include/linux/capability.h
@@ -186,9 +186,9 @@ static inline bool ns_capable_setid(struct user_namespace *ns, int cap)
}
#endif /* CONFIG_MULTIUSER */
bool privileged_wrt_inode_uidgid(struct user_namespace *ns,
- struct mnt_idmap *idmap,
+ const struct mnt_idmap *idmap,
const struct inode *inode);
-bool capable_wrt_inode_uidgid(struct mnt_idmap *idmap,
+bool capable_wrt_inode_uidgid(const struct mnt_idmap *idmap,
const struct inode *inode, int cap);
extern bool file_ns_capable(const struct file *file, struct user_namespace *ns, int cap);
extern bool ptracer_capable(struct task_struct *tsk, struct user_namespace *ns);
@@ -215,11 +215,11 @@ static inline bool checkpoint_restore_ns_capable_noaudit(struct user_namespace *
}
/* audit system wants to get cap info from files as well */
-int get_vfs_caps_from_disk(struct mnt_idmap *idmap,
+int get_vfs_caps_from_disk(const struct mnt_idmap *idmap,
const struct dentry *dentry,
struct cpu_vfs_cap_data *cpu_caps);
-int cap_convert_nscap(struct mnt_idmap *idmap, struct dentry *dentry,
+int cap_convert_nscap(const struct mnt_idmap *idmap, struct dentry *dentry,
const void **ivalue, size_t size);
#endif /* !_LINUX_CAPABILITY_H */
diff --git a/include/linux/cleanup.h b/include/linux/cleanup.h
index b1b5698cbf1b..1fb8058b897d 100644
--- a/include/linux/cleanup.h
+++ b/include/linux/cleanup.h
@@ -261,10 +261,6 @@ const volatile void * __must_check_fn(const volatile void *val)
* CLASS(name, var)(args...):
* declare the variable @var as an instance of the named class
*
- * CLASS_INIT(name, var, init_expr):
- * declare the variable @var as an instance of the named class with
- * custom initialization expression.
- *
* Ex.
*
* DEFINE_CLASS(fdget, struct fd, fdput(_T), fdget(fd), int fd)
@@ -302,9 +298,6 @@ static __always_inline class_##_name##_t class_##_name##ext##_constructor(_init_
class_##_name##_t var __cleanup(class_##_name##_destructor) = \
class_##_name##_constructor
-#define CLASS_INIT(_name, _var, _init_expr) \
- class_##_name##_t _var __cleanup(class_##_name##_destructor) = (_init_expr)
-
#define __scoped_class(_name, var, _label, args...) \
for (CLASS(_name, var)(args); ; ({ goto _label; })) \
if (0) { \
diff --git a/include/linux/coredump.h b/include/linux/coredump.h
index 7b38ee2e7913..74af57b9406b 100644
--- a/include/linux/coredump.h
+++ b/include/linux/coredump.h
@@ -6,9 +6,20 @@
#include <linux/mm.h>
#include <linux/fs.h>
#include <linux/sched/coredump.h>
+#include <uapi/linux/coredump.h>
#include <asm/siginfo.h>
#ifdef CONFIG_COREDUMP
+/**
+ * enum coredump_state - what happened while the coredump was written
+ * @COREDUMP_STATE_STARTED: the dumper committed to writing a coredump
+ * @COREDUMP_STATE_TRUNCATED: the dumper stopped before it had written all of it
+ */
+enum coredump_state {
+ COREDUMP_STATE_STARTED = (1U << 0),
+ COREDUMP_STATE_TRUNCATED = (1U << 1),
+};
+
struct core_vma_metadata {
unsigned long start, end;
vm_flags_t flags;
@@ -21,12 +32,20 @@ struct coredump_params {
const kernel_siginfo_t *siginfo;
struct file *file;
unsigned long limit;
- /* MMF_DUMP_FILTER_* bits, snapshot of mm->flags at dump start. */
- unsigned long mm_flags;
+ /* COREDUMP_MEMORY_* types to dump, the task's or the server's. */
+ u64 memory_types;
/* Snapshot of dumpable at dump start. */
enum task_dumpable dumpable;
int cpu;
+ /* COREDUMP_* options negotiated with the coredump server. */
+ u64 mask;
+ /* COREDUMP_STATE_* raised while the coredump is written. */
+ enum coredump_state state;
+ /* Record header scratch, NULL unless the coredump is a record stream. */
+ struct coredump_record_header *record_hdr;
+ /* Bytes handed to the file, record headers included. */
loff_t written;
+ /* Offset in the coredump, record headers excluded. */
loff_t pos;
loff_t to_skip;
int vma_count;
@@ -41,13 +60,13 @@ extern unsigned int core_file_note_size_limit;
* These are the only things you should do on a core-file: use only these
* functions to write out all the necessary info.
*/
-extern void dump_skip_to(struct coredump_params *cprm, unsigned long to);
-extern void dump_skip(struct coredump_params *cprm, size_t nr);
-extern int dump_emit(struct coredump_params *cprm, const void *addr, int nr);
-extern int dump_align(struct coredump_params *cprm, int align);
-int dump_user_range(struct coredump_params *cprm, unsigned long start,
- unsigned long len);
-extern void vfs_coredump(const kernel_siginfo_t *siginfo);
+void dump_skip_to(struct coredump_params *cprm, unsigned long to);
+void dump_skip(struct coredump_params *cprm, size_t nr);
+bool dump_emit(struct coredump_params *cprm, const void *addr, int nr);
+bool dump_align(struct coredump_params *cprm, int align);
+bool dump_user_range(struct coredump_params *cprm, unsigned long start,
+ unsigned long len);
+void vfs_coredump(const kernel_siginfo_t *siginfo);
/*
* Logging for the coredump code, ratelimited.
diff --git a/include/linux/dax.h b/include/linux/dax.h
index fe6c3ded1b50..f2d47975d905 100644
--- a/include/linux/dax.h
+++ b/include/linux/dax.h
@@ -155,8 +155,6 @@ int dax_writeback_mapping_range(struct address_space *mapping,
struct dax_device *dax_dev, struct writeback_control *wbc);
int dax_folio_reset_order(struct folio *folio);
-struct page *dax_layout_busy_page(struct address_space *mapping);
-struct page *dax_layout_busy_page_range(struct address_space *mapping, loff_t start, loff_t end);
dax_entry_t dax_lock_folio(struct folio *folio);
void dax_unlock_folio(struct folio *folio, dax_entry_t cookie);
dax_entry_t dax_lock_mapping_entry(struct address_space *mapping,
@@ -173,16 +171,6 @@ static inline int fs_dax_get(struct dax_device *dax_dev, void *holder,
{
return -EOPNOTSUPP;
}
-static inline struct page *dax_layout_busy_page(struct address_space *mapping)
-{
- return NULL;
-}
-
-static inline struct page *dax_layout_busy_page_range(struct address_space *mapping, pgoff_t start, pgoff_t nr_pages)
-{
- return NULL;
-}
-
static inline int dax_writeback_mapping_range(struct address_space *mapping,
struct dax_device *dax_dev, struct writeback_control *wbc)
{
diff --git a/include/linux/dcache.h b/include/linux/dcache.h
index 4b1ff99608e0..adf239f8205f 100644
--- a/include/linux/dcache.h
+++ b/include/linux/dcache.h
@@ -116,6 +116,8 @@ struct dentry {
* possible!
*/
+ /* lockdep tracking of DCACHE_PAR_LOOKUP locks */
+ struct lockdep_map lookup_map;
struct list_head d_lru; /* LRU list */
struct hlist_node d_sib; /* child of parent list */
struct hlist_head d_children; /* our children */
@@ -236,7 +238,9 @@ enum dentry_flags {
DCACHE_PAR_LOOKUP = BIT(24), /* being looked up (with parent locked shared) */
DCACHE_DENTRY_CURSOR = BIT(25),
DCACHE_NORCU = BIT(26), /* No RCU delay for freeing */
- DCACHE_PERSISTENT = BIT(27)
+ DCACHE_PERSISTENT = BIT(27),
+/* 28, 29, 30 free */
+ DCACHE_PRIVATE = BIT(31) /* fs-specific flag */
};
#define DCACHE_MANAGED_DENTRY \
@@ -257,7 +261,9 @@ extern void d_delete(struct dentry *);
extern struct dentry * d_alloc(struct dentry *, const struct qstr *);
extern struct dentry * d_alloc_anon(struct super_block *);
extern struct dentry * d_alloc_parallel(struct dentry *, const struct qstr *);
+extern struct dentry * d_alloc_trylock(struct dentry *, struct qstr *);
extern struct dentry * d_splice_alias(struct inode *, struct dentry *);
+struct dentry *d_duplicate(struct dentry *dentry);
/* weird procfs mess; *NOT* exported */
extern struct dentry * d_splice_alias_ops(struct inode *, struct dentry *,
const struct dentry_operations *);
@@ -553,6 +559,36 @@ static inline int simple_positive(const struct dentry *dentry)
unsigned long vfs_pressure_ratio(unsigned long val);
/**
+ * d_lookup_release - release ownership of DCACHE_PAR_LOOKUP lock
+ * @dentry: dentry that is locked
+ *
+ * If an in-lookup dentry is to be passed to another thread which
+ * will drop the in-lookup lock, then d_lookup_release() must be called
+ * to tell lockdep that this thread no lock holds the lock. The
+ * thread that receives the lock must call d_lookup_acquire() to
+ * acquire the lock.
+ */
+static inline void d_lookup_release(struct dentry *dentry)
+{
+ if (d_in_lookup(dentry))
+ lock_map_release(&dentry->lookup_map);
+}
+
+/**
+ * d_lookup_acquire - acquire ownership of DCACHE_PAR_LOOKUP lock
+ * @dentry: dentry that is locked
+ *
+ * If an in-lookup dentry was passed to this thread, the
+ * d_lookup_acquire() must be called to tell lockdep that this
+ * thread now owns the DCACHE_PAR_LOOKUP lock.
+ */
+static inline void d_lookup_acquire(struct dentry *dentry)
+{
+ if (d_in_lookup(dentry))
+ lock_map_acquire_try(&dentry->lookup_map);
+}
+
+/**
* d_inode - Get the actual inode of this dentry
* @dentry: The dentry to query
*
diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h
index c45306a9f007..a46781058729 100644
--- a/include/linux/fdtable.h
+++ b/include/linux/fdtable.h
@@ -25,7 +25,7 @@
struct fdtable {
unsigned int max_fds;
- struct file __rcu **fd; /* current fd array */
+ struct file __rcu **fd __counted_by_ptr(max_fds); /* current fd array */
unsigned long *close_on_exec;
unsigned long *open_fds;
unsigned long *full_fds_bits;
@@ -101,11 +101,22 @@ struct task_struct;
void put_files_struct(struct files_struct *fs);
int unshare_files(void);
+void switch_files_struct(struct task_struct *tsk, struct files_struct *files);
+int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp);
+enum fd_range_flags {
+ /* Leave behind all descriptors outside of the specified range. */
+ FD_RANGE_EXCEPT = (1U << 0),
+
+ /* Only select descriptors that have close-on-exec set. */
+ FD_RANGE_CLOEXEC_ONLY = (1U << 1),
+};
+
struct fd_range {
unsigned int from, to;
+ enum fd_range_flags flags;
};
struct files_struct *dup_fd(struct files_struct *, struct fd_range *) __latent_entropy;
-void do_close_on_exec(struct files_struct *);
+void close_cloexec_files(struct files_struct *);
int iterate_fd(struct files_struct *, unsigned,
int (*)(const void *, struct file *, unsigned),
const void *);
diff --git a/include/linux/file.h b/include/linux/file.h
index 27484b444d31..41c3c0be1064 100644
--- a/include/linux/file.h
+++ b/include/linux/file.h
@@ -12,6 +12,7 @@
#include <linux/errno.h>
#include <linux/cleanup.h>
#include <linux/err.h>
+#include <linux/vfsdebug.h>
struct file;
@@ -129,117 +130,84 @@ extern unsigned int sysctl_nr_open_min, sysctl_nr_open_max;
/*
* fd_prepare: Combined fd + file allocation cleanup class.
- * @err: Error code to indicate if allocation succeeded.
- * @__fd: Allocated fd (may not be accessed directly)
- * @__file: Allocated struct file pointer (may not be accessed directly)
+ * @fd: Allocated fd
+ * @file: Allocated struct file pointer
*
* Allocates an fd and a file together. On error paths, automatically cleans
* up whichever resource was successfully allocated. Allows flexible file
* allocation with different functions per usage.
*
- * Do not use directly.
+ * Do not declare directly, use FD_PREPARE().
*/
struct fd_prepare {
- s32 err;
- s32 __fd; /* do not access directly */
- struct file *__file; /* do not access directly */
+ int fd;
+ struct file *file;
};
-/* Typedef for fd_prepare cleanup guards. */
-typedef struct fd_prepare class_fd_prepare_t;
-
-/*
- * Accessors for fd_prepare class members.
- * _Generic() is used for zero-cost type safety.
- */
-#define fd_prepare_fd(_fdf) \
- (_Generic((_fdf), struct fd_prepare: (_fdf).__fd))
-
-#define fd_prepare_file(_fdf) \
- (_Generic((_fdf), struct fd_prepare: (_fdf).__file))
-
/* Do not use directly. */
-static inline void class_fd_prepare_destructor(const struct fd_prepare *fdf)
+static __always_inline void __fd_prepare_cleanup(const struct fd_prepare *fdf)
{
- if (unlikely(fdf->__fd >= 0))
- put_unused_fd(fdf->__fd);
- if (unlikely(!IS_ERR_OR_NULL(fdf->__file)))
- fput(fdf->__file);
+ if (unlikely(fdf->fd >= 0)) {
+ put_unused_fd(fdf->fd);
+ fput(fdf->file);
+ }
}
/* Do not use directly. */
-static inline int class_fd_prepare_lock_err(const struct fd_prepare *fdf)
+static __always_inline struct fd_prepare __fd_prepare(int fd, struct file *file)
{
- if (unlikely(fdf->err))
- return fdf->err;
- if (unlikely(fdf->__fd < 0))
- return fdf->__fd;
- if (unlikely(IS_ERR(fdf->__file)))
- return PTR_ERR(fdf->__file);
- if (unlikely(!fdf->__file))
- return -ENOMEM;
- return 0;
+ if (fd >= 0 && IS_ERR_OR_NULL(file)) {
+ int err = file ? PTR_ERR(file) : -ENOMEM;
+
+ put_unused_fd(fd);
+ fd = err;
+ file = NULL;
+ }
+ return (struct fd_prepare){ .fd = fd, .file = file };
}
/*
- * __FD_PREPARE_INIT - Helper to initialize fd_prepare class.
- * @_fd_flags: flags for get_unused_fd_flags()
- * @_file_owned: expression that returns struct file *
- *
- * Returns a struct fd_prepare with fd, file, and err set.
- * If fd allocation fails, fd will be negative and err will be set. If
- * fd succeeds but file_init_expr fails, file will be ERR_PTR and err
- * will be set. The err field is the single source of truth for error
- * checking.
- */
-#define __FD_PREPARE_INIT(_fd_flags, _file_owned) \
- ({ \
- struct fd_prepare fdf = { \
- .__fd = get_unused_fd_flags((_fd_flags)), \
- }; \
- if (likely(fdf.__fd >= 0)) \
- fdf.__file = (_file_owned); \
- fdf.err = ACQUIRE_ERR(fd_prepare, &fdf); \
- fdf; \
- })
-
-/*
- * FD_PREPARE - Macro to declare and initialize an fd_prepare variable.
+ * FD_PREPARE - Declare and initialize an fd_prepare instance.
*
- * Declares and initializes an fd_prepare variable with automatic
- * cleanup. No separate scope required - cleanup happens when variable
- * goes out of scope.
+ * This allocates a new fd and only evaluates @_file_owned if the
+ * allocation succeeded. Cleanup happens when the variable goes out of
+ * scope and the guard releases whichever of the descriptor and the file
+ * was allocated. If fd_publish() was called the fd and file are
+ * published and cleanup becomes a nop.
*
- * @_fdf: name of struct fd_prepare variable to define
+ * @_fdf: name of the const struct fd_prepare pointer to define
* @_fd_flags: flags for get_unused_fd_flags()
* @_file_owned: struct file to take ownership of (can be expression)
*/
+#define __FD_PREPARE(_guard, _fdf, _fd_flags, _file_owned) \
+ struct fd_prepare _guard __cleanup(__fd_prepare_cleanup) = ({ \
+ int __fd = get_unused_fd_flags(_fd_flags); \
+ __fd_prepare(__fd, __fd < 0 ? NULL : (_file_owned)); \
+ }); \
+ const struct fd_prepare *const _fdf = &_guard
+
#define FD_PREPARE(_fdf, _fd_flags, _file_owned) \
- CLASS_INIT(fd_prepare, _fdf, __FD_PREPARE_INIT(_fd_flags, _file_owned))
+ __FD_PREPARE(__UNIQUE_ID(fd_prepare), _fdf, _fd_flags, _file_owned)
/*
* fd_publish - Publish prepared fd and file to the fd table.
- * @_fdf: struct fd_prepare variable
+ * @fdf: struct fd_prepare pointer defined by FD_PREPARE()
*/
-#define fd_publish(_fdf) \
- ({ \
- struct fd_prepare *fdp = &(_fdf); \
- VFS_WARN_ON_ONCE(fdp->err); \
- VFS_WARN_ON_ONCE(fdp->__fd < 0); \
- VFS_WARN_ON_ONCE(IS_ERR_OR_NULL(fdp->__file)); \
- fd_install(fdp->__fd, fdp->__file); \
- retain_and_null_ptr(fdp->__file); \
- take_fd(fdp->__fd); \
- })
+static __always_inline int fd_publish(const struct fd_prepare *fdf)
+{
+ /* Callers only get a const view, the guard itself is writable. */
+ struct fd_prepare *guard = (struct fd_prepare *)fdf;
+
+ VFS_WARN_ON_ONCE(guard->fd < 0);
+ fd_install(guard->fd, guard->file);
+ return take_fd(guard->fd);
+}
/* Do not use directly. */
-#define __FD_ADD(_fdf, _fd_flags, _file_owned) \
- ({ \
- FD_PREPARE(_fdf, _fd_flags, _file_owned); \
- s32 ret = _fdf.err; \
- if (likely(!ret)) \
- ret = fd_publish(_fdf); \
- ret; \
+#define __FD_ADD(_fdf, _fd_flags, _file_owned) \
+ ({ \
+ FD_PREPARE(_fdf, _fd_flags, _file_owned); \
+ _fdf->fd < 0 ? _fdf->fd : fd_publish(_fdf); \
})
/*
diff --git a/include/linux/fileattr.h b/include/linux/fileattr.h
index 58044b598016..09e32b84e02a 100644
--- a/include/linux/fileattr.h
+++ b/include/linux/fileattr.h
@@ -74,7 +74,7 @@ static inline bool fileattr_has_fsx(const struct file_kattr *fa)
}
int vfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa);
-int vfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry,
+int vfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry,
struct file_kattr *fa);
int ioctl_getflags(struct file *file, unsigned int __user *argp);
int ioctl_setflags(struct file *file, unsigned int __user *argp);
diff --git a/include/linux/fs.h b/include/linux/fs.h
index f9d1e05e8ae6..3db90996756f 100644
--- a/include/linux/fs.h
+++ b/include/linux/fs.h
@@ -1436,10 +1436,10 @@ static inline void i_gid_write(struct inode *inode, gid_t gid)
* @idmap: idmap of the mount the inode was found from
* @inode: inode to map
*
- * Return: whe inode's i_uid mapped down according to @idmap.
+ * Return: the inode's i_uid mapped down according to @idmap.
* If the inode's i_uid has no mapping INVALID_VFSUID is returned.
*/
-static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap,
+static inline vfsuid_t i_uid_into_vfsuid(const struct mnt_idmap *idmap,
const struct inode *inode)
{
return make_vfsuid(idmap, i_user_ns(inode), inode->i_uid);
@@ -1456,7 +1456,7 @@ static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap,
*
* Return: true if @inode's i_uid field needs to be updated, false if not.
*/
-static inline bool i_uid_needs_update(struct mnt_idmap *idmap,
+static inline bool i_uid_needs_update(const struct mnt_idmap *idmap,
const struct iattr *attr,
const struct inode *inode)
{
@@ -1474,7 +1474,7 @@ static inline bool i_uid_needs_update(struct mnt_idmap *idmap,
* Safely update @inode's i_uid field translating the vfsuid of any idmapped
* mount into the filesystem kuid.
*/
-static inline void i_uid_update(struct mnt_idmap *idmap,
+static inline void i_uid_update(const struct mnt_idmap *idmap,
const struct iattr *attr,
struct inode *inode)
{
@@ -1491,7 +1491,7 @@ static inline void i_uid_update(struct mnt_idmap *idmap,
* Return: the inode's i_gid mapped down according to @idmap.
* If the inode's i_gid has no mapping INVALID_VFSGID is returned.
*/
-static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap,
+static inline vfsgid_t i_gid_into_vfsgid(const struct mnt_idmap *idmap,
const struct inode *inode)
{
return make_vfsgid(idmap, i_user_ns(inode), inode->i_gid);
@@ -1508,7 +1508,7 @@ static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap,
*
* Return: true if @inode's i_gid field needs to be updated, false if not.
*/
-static inline bool i_gid_needs_update(struct mnt_idmap *idmap,
+static inline bool i_gid_needs_update(const struct mnt_idmap *idmap,
const struct iattr *attr,
const struct inode *inode)
{
@@ -1526,7 +1526,7 @@ static inline bool i_gid_needs_update(struct mnt_idmap *idmap,
* Safely update @inode's i_gid field translating the vfsgid of any idmapped
* mount into the filesystem kgid.
*/
-static inline void i_gid_update(struct mnt_idmap *idmap,
+static inline void i_gid_update(const struct mnt_idmap *idmap,
const struct iattr *attr,
struct inode *inode)
{
@@ -1544,7 +1544,7 @@ static inline void i_gid_update(struct mnt_idmap *idmap,
* an idmapped mount map the caller's fsuid according to @idmap.
*/
static inline void inode_fsuid_set(struct inode *inode,
- struct mnt_idmap *idmap)
+ const struct mnt_idmap *idmap)
{
inode->i_uid = mapped_fsuid(idmap, i_user_ns(inode));
}
@@ -1558,7 +1558,7 @@ static inline void inode_fsuid_set(struct inode *inode,
* an idmapped mount map the caller's fsgid according to @idmap.
*/
static inline void inode_fsgid_set(struct inode *inode,
- struct mnt_idmap *idmap)
+ const struct mnt_idmap *idmap)
{
inode->i_gid = mapped_fsgid(idmap, i_user_ns(inode));
}
@@ -1575,7 +1575,7 @@ static inline void inode_fsgid_set(struct inode *inode,
* Return: true if fsuid and fsgid is mapped, false if not.
*/
static inline bool fsuidgid_has_mapping(struct super_block *sb,
- struct mnt_idmap *idmap)
+ const struct mnt_idmap *idmap)
{
struct user_namespace *fs_userns = sb->s_user_ns;
kuid_t kuid;
@@ -1755,25 +1755,25 @@ static inline bool file_write_not_started(const struct file *file)
return sb_write_not_started(file_inode(file)->i_sb);
}
-bool inode_owner_or_capable(struct mnt_idmap *idmap,
+bool inode_owner_or_capable(const struct mnt_idmap *idmap,
const struct inode *inode);
/*
* VFS helper functions..
*/
-int vfs_create(struct mnt_idmap *, struct dentry *, umode_t,
+int vfs_create(const struct mnt_idmap *, struct dentry *, umode_t,
struct delegated_inode *);
-struct dentry *vfs_mkdir(struct mnt_idmap *, struct inode *,
+struct dentry *vfs_mkdir(const struct mnt_idmap *, struct inode *,
struct dentry *, umode_t, struct delegated_inode *);
-int vfs_mknod(struct mnt_idmap *, struct inode *, struct dentry *,
+int vfs_mknod(const struct mnt_idmap *, struct inode *, struct dentry *,
umode_t, dev_t, struct delegated_inode *);
-int vfs_symlink(struct mnt_idmap *, struct inode *,
+int vfs_symlink(const struct mnt_idmap *, struct inode *,
struct dentry *, const char *, struct delegated_inode *);
-int vfs_link(struct dentry *, struct mnt_idmap *, struct inode *,
+int vfs_link(struct dentry *, const struct mnt_idmap *, struct inode *,
struct dentry *, struct delegated_inode *);
-int vfs_rmdir(struct mnt_idmap *, struct inode *, struct dentry *,
+int vfs_rmdir(const struct mnt_idmap *, struct inode *, struct dentry *,
struct delegated_inode *);
-int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *,
+int vfs_unlink(const struct mnt_idmap *, struct inode *, struct dentry *,
struct delegated_inode *);
/**
@@ -1787,7 +1787,7 @@ int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *,
* @flags: rename flags
*/
struct renamedata {
- struct mnt_idmap *mnt_idmap;
+ const struct mnt_idmap *mnt_idmap;
struct dentry *old_parent;
struct dentry *old_dentry;
struct dentry *new_parent;
@@ -1798,14 +1798,14 @@ struct renamedata {
int vfs_rename(struct renamedata *);
-static inline int vfs_whiteout(struct mnt_idmap *idmap,
+static inline int vfs_whiteout(const struct mnt_idmap *idmap,
struct inode *dir, struct dentry *dentry)
{
return vfs_mknod(idmap, dir, dentry, S_IFCHR | WHITEOUT_MODE,
WHITEOUT_DEV, NULL);
}
-struct file *kernel_tmpfile_open(struct mnt_idmap *idmap,
+struct file *kernel_tmpfile_open(const struct mnt_idmap *idmap,
const struct path *parentpath,
umode_t mode, int open_flag,
const struct cred *cred);
@@ -1830,12 +1830,12 @@ extern long compat_ptr_ioctl(struct file *file, unsigned int cmd,
/*
* VFS file helper functions.
*/
-void inode_init_owner(struct mnt_idmap *idmap, struct inode *inode,
+void inode_init_owner(const struct mnt_idmap *idmap, struct inode *inode,
const struct inode *dir, umode_t mode);
extern bool may_open_dev(const struct path *path);
-umode_t mode_strip_sgid(struct mnt_idmap *idmap,
+umode_t mode_strip_sgid(const struct mnt_idmap *idmap,
const struct inode *dir, umode_t mode);
-bool in_group_or_capable(struct mnt_idmap *idmap,
+bool in_group_or_capable(const struct mnt_idmap *idmap,
const struct inode *inode, vfsgid_t vfsgid);
/*
@@ -1994,26 +1994,26 @@ enum fs_update_time {
struct inode_operations {
struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int);
const char * (*get_link) (struct dentry *, struct inode *, struct delayed_call *);
- int (*permission) (struct mnt_idmap *, struct inode *, int);
+ int (*permission) (const struct mnt_idmap *, struct inode *, int);
struct posix_acl * (*get_inode_acl)(struct inode *, int, bool);
int (*readlink) (struct dentry *, char __user *,int);
- int (*create) (struct mnt_idmap *, struct inode *,struct dentry *,
+ int (*create) (const struct mnt_idmap *, struct inode *,struct dentry *,
umode_t);
int (*link) (struct dentry *,struct inode *,struct dentry *);
int (*unlink) (struct inode *,struct dentry *);
- int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *,
+ int (*symlink) (const struct mnt_idmap *, struct inode *,struct dentry *,
const char *);
- struct dentry *(*mkdir) (struct mnt_idmap *, struct inode *,
+ struct dentry *(*mkdir) (const struct mnt_idmap *, struct inode *,
struct dentry *, umode_t);
int (*rmdir) (struct inode *,struct dentry *);
- int (*mknod) (struct mnt_idmap *, struct inode *,struct dentry *,
+ int (*mknod) (const struct mnt_idmap *, struct inode *,struct dentry *,
umode_t,dev_t);
- int (*rename) (struct mnt_idmap *, struct inode *, struct dentry *,
+ int (*rename) (const struct mnt_idmap *, struct inode *, struct dentry *,
struct inode *, struct dentry *, unsigned int);
- int (*setattr) (struct mnt_idmap *, struct dentry *, struct iattr *);
- int (*getattr) (struct mnt_idmap *, const struct path *,
+ int (*setattr) (const struct mnt_idmap *, struct dentry *, struct iattr *);
+ int (*getattr) (const struct mnt_idmap *, const struct path *,
struct kstat *, u32, unsigned int);
ssize_t (*listxattr) (struct dentry *, char *, size_t);
int (*fiemap)(struct inode *, struct fiemap_extent_info *, u64 start,
@@ -2024,13 +2024,13 @@ struct inode_operations {
int (*atomic_open)(struct inode *, struct dentry *,
struct file *, unsigned open_flag,
umode_t create_mode);
- int (*tmpfile) (struct mnt_idmap *, struct inode *,
+ int (*tmpfile) (const struct mnt_idmap *, struct inode *,
struct file *, umode_t);
- struct posix_acl *(*get_acl)(struct mnt_idmap *, struct dentry *,
+ struct posix_acl *(*get_acl)(const struct mnt_idmap *, struct dentry *,
int);
- int (*set_acl)(struct mnt_idmap *, struct dentry *,
+ int (*set_acl)(const struct mnt_idmap *, struct dentry *,
struct posix_acl *, int);
- int (*fileattr_set)(struct mnt_idmap *idmap,
+ int (*fileattr_set)(const struct mnt_idmap *idmap,
struct dentry *dentry, struct file_kattr *fa);
int (*fileattr_get)(struct dentry *dentry, struct file_kattr *fa);
struct offset_ctx *(*get_offset_ctx)(struct inode *inode);
@@ -2173,7 +2173,7 @@ extern loff_t vfs_dedupe_file_range_one(struct file *src_file, loff_t src_pos,
(inode)->i_rdev == WHITEOUT_DEV)
#define IS_ANON_FILE(inode) ((inode)->i_flags & S_ANON_INODE)
-static inline bool HAS_UNMAPPED_ID(struct mnt_idmap *idmap,
+static inline bool HAS_UNMAPPED_ID(const struct mnt_idmap *idmap,
struct inode *inode)
{
return !vfsuid_valid(i_uid_into_vfsuid(idmap, inode)) ||
@@ -2459,7 +2459,7 @@ struct filename {
static_assert(offsetof(struct filename, iname) % sizeof(long) == 0);
static_assert(sizeof(struct filename) % 64 == 0);
-static inline struct mnt_idmap *file_mnt_idmap(const struct file *file)
+static inline const struct mnt_idmap *file_mnt_idmap(const struct file *file)
{
return mnt_idmap(file->f_path.mnt);
}
@@ -2483,7 +2483,7 @@ static inline bool is_idmapped_mnt(const struct vfsmount *mnt)
}
int vfs_truncate(const struct path *, loff_t);
-int do_truncate(struct mnt_idmap *, struct dentry *, loff_t start,
+int do_truncate(const struct mnt_idmap *, struct dentry *, loff_t start,
unsigned int time_attrs, struct file *filp);
extern int vfs_fallocate(struct file *file, int mode, loff_t offset,
loff_t len);
@@ -2707,10 +2707,10 @@ static inline int bmap(struct inode *inode, sector_t *block)
}
#endif
-int notify_change(struct mnt_idmap *, struct dentry *,
+int notify_change(const struct mnt_idmap *, struct dentry *,
struct iattr *, struct delegated_inode *);
-int inode_permission(struct mnt_idmap *, struct inode *, int);
-int generic_permission(struct mnt_idmap *, struct inode *, int);
+int inode_permission(const struct mnt_idmap *, struct inode *, int);
+int generic_permission(const struct mnt_idmap *, struct inode *, int);
static inline int file_permission(struct file *file, int mask)
{
return inode_permission(file_mnt_idmap(file),
@@ -2721,12 +2721,12 @@ static inline int path_permission(const struct path *path, int mask)
return inode_permission(mnt_idmap(path->mnt),
d_inode(path->dentry), mask);
}
-int __check_sticky(struct mnt_idmap *idmap, struct inode *dir,
+int __check_sticky(const struct mnt_idmap *idmap, struct inode *dir,
struct inode *inode);
-int may_delete_dentry(struct mnt_idmap *idmap, struct inode *dir,
+int may_delete_dentry(const struct mnt_idmap *idmap, struct inode *dir,
struct dentry *victim, bool isdir);
-int may_create_dentry(struct mnt_idmap *idmap,
+int may_create_dentry(const struct mnt_idmap *idmap,
struct inode *dir, struct dentry *child);
static inline bool execute_ok(struct inode *inode)
@@ -3045,9 +3045,9 @@ static inline struct inode *new_inode_pseudo(struct super_block *sb)
}
extern struct inode *new_inode(struct super_block *sb);
extern void free_inode_nonrcu(struct inode *inode);
-extern int setattr_should_drop_suidgid(struct mnt_idmap *, struct inode *);
+extern int setattr_should_drop_suidgid(const struct mnt_idmap *, struct inode *);
extern int file_remove_privs(struct file *);
-int setattr_should_drop_sgid(struct mnt_idmap *idmap,
+int setattr_should_drop_sgid(const struct mnt_idmap *idmap,
const struct inode *inode);
/*
@@ -3204,7 +3204,7 @@ extern int page_symlink(struct inode *inode, const char *symname, int len);
extern const struct inode_operations page_symlink_inode_operations;
extern void kfree_link(void *);
void fill_mg_cmtime(struct kstat *stat, u32 request_mask, struct inode *inode);
-void generic_fillattr(struct mnt_idmap *, u32, struct inode *, struct kstat *);
+void generic_fillattr(const struct mnt_idmap *, u32, struct inode *, struct kstat *);
void generic_fill_statx_attr(struct inode *inode, struct kstat *stat);
void generic_fill_statx_atomic_writes(struct kstat *stat,
unsigned int unit_min,
@@ -3261,9 +3261,9 @@ extern int dcache_dir_open(struct inode *, struct file *);
extern int dcache_dir_close(struct inode *, struct file *);
extern loff_t dcache_dir_lseek(struct file *, loff_t, int);
extern int dcache_readdir(struct file *, struct dir_context *);
-extern int simple_setattr(struct mnt_idmap *, struct dentry *,
+extern int simple_setattr(const struct mnt_idmap *, struct dentry *,
struct iattr *);
-extern int simple_getattr(struct mnt_idmap *, const struct path *,
+extern int simple_getattr(const struct mnt_idmap *, const struct path *,
struct kstat *, u32, unsigned int);
extern int simple_statfs(struct dentry *, struct kstatfs *);
extern int simple_open(struct inode *inode, struct file *file);
@@ -3276,7 +3276,7 @@ void simple_rename_timestamp(struct inode *old_dir, struct dentry *old_dentry,
struct inode *new_dir, struct dentry *new_dentry);
extern int simple_rename_exchange(struct inode *old_dir, struct dentry *old_dentry,
struct inode *new_dir, struct dentry *new_dentry);
-extern int simple_rename(struct mnt_idmap *, struct inode *,
+extern int simple_rename(const struct mnt_idmap *, struct inode *,
struct dentry *, struct inode *, struct dentry *,
unsigned int);
extern void simple_recursive_removal(struct dentry *,
@@ -3397,11 +3397,11 @@ static inline bool generic_ci_validate_strict_name(struct inode *dir,
}
#endif
-int may_setattr(struct mnt_idmap *idmap, struct inode *inode,
+int may_setattr(const struct mnt_idmap *idmap, struct inode *inode,
unsigned int ia_valid);
-int setattr_prepare(struct mnt_idmap *, struct dentry *, struct iattr *);
+int setattr_prepare(const struct mnt_idmap *, struct dentry *, struct iattr *);
extern int inode_newsize_ok(const struct inode *, loff_t offset);
-void setattr_copy(struct mnt_idmap *, struct inode *inode,
+void setattr_copy(const struct mnt_idmap *, struct inode *inode,
const struct iattr *attr);
extern int file_update_time(struct file *file);
@@ -3578,7 +3578,7 @@ static inline bool is_sxid(umode_t mode)
return mode & (S_ISUID | S_ISGID);
}
-static inline int check_sticky(struct mnt_idmap *idmap,
+static inline int check_sticky(const struct mnt_idmap *idmap,
struct inode *dir, struct inode *inode)
{
if (!(dir->i_mode & S_ISVTX))
@@ -3653,23 +3653,6 @@ extern int vfs_fadvise(struct file *file, loff_t offset, loff_t len,
extern int generic_fadvise(struct file *file, loff_t offset, loff_t len,
int advice);
-static inline bool vfs_empty_path(int dfd, const char __user *path)
-{
- char c;
-
- if (dfd < 0)
- return false;
-
- /* We now allow NULL to be used for empty path. */
- if (!path)
- return true;
-
- if (unlikely(get_user(c, path)))
- return false;
-
- return !c;
-}
-
int generic_atomic_write_valid(struct kiocb *iocb, struct iov_iter *iter);
static inline bool extensible_ioctl_valid(unsigned int cmd_a,
diff --git a/include/linux/fs_context.h b/include/linux/fs_context.h
index 0d6c8a6d7be2..c920aba5177c 100644
--- a/include/linux/fs_context.h
+++ b/include/linux/fs_context.h
@@ -150,6 +150,10 @@ extern int vfs_parse_fs_param_source(struct fs_context *fc,
struct fs_parameter *param);
extern void fc_drop_locked(struct fs_context *fc);
+extern int get_tree_super(struct fs_context *fc,
+ int (*test)(struct super_block *, struct fs_context *),
+ int (*fill_super)(struct super_block *sb,
+ struct fs_context *fc));
extern int get_tree_nodev(struct fs_context *fc,
int (*fill_super)(struct super_block *sb,
struct fs_context *fc));
diff --git a/include/linux/fscache-cache.h b/include/linux/fscache-cache.h
index 4c91a019972b..ee524c863fa9 100644
--- a/include/linux/fscache-cache.h
+++ b/include/linux/fscache-cache.h
@@ -67,7 +67,7 @@ struct fscache_cache_ops {
/* Change the size of a data object */
void (*resize_cookie)(struct netfs_cache_resources *cres,
- loff_t new_size);
+ uoff_t new_size);
/* Invalidate an object */
bool (*invalidate_cookie)(struct fscache_cookie *cookie);
diff --git a/include/linux/fscache.h b/include/linux/fscache.h
index 58fdb9605425..f2d958bd1f48 100644
--- a/include/linux/fscache.h
+++ b/include/linux/fscache.h
@@ -112,7 +112,7 @@ struct fscache_cookie {
struct list_head proc_link; /* Link in proc list */
struct list_head commit_link; /* Link in commit queue */
struct work_struct work; /* Commit/relinq/withdraw work */
- loff_t object_size; /* Size of the netfs object */
+ uoff_t object_size; /* Size of the netfs object */
unsigned long unused_at; /* Time at which unused (jiffies) */
unsigned long flags;
#define FSCACHE_COOKIE_RELINQUISHED 0 /* T if cookie has been relinquished */
@@ -147,6 +147,23 @@ struct fscache_cookie {
};
};
+enum fscache_extent_type {
+ FSCACHE_EXTENT_DATA,
+ FSCACHE_EXTENT_ZERO,
+} __mode(byte);
+
+/*
+ * Cache occupancy information.
+ */
+struct fscache_occupancy {
+ unsigned long long query_from; /* Point to query from */
+ unsigned long long query_to; /* Point to query to */
+ unsigned long long cached_from[2]; /* Point at which cache extents start */
+ unsigned long long cached_to[2]; /* Point at which cache extents end */
+ unsigned int granularity; /* Granularity desired */
+ enum fscache_extent_type cached_type[2]; /* Type of cache extent */
+};
+
/*
* slow-path functions for when there is actually caching available, and the
* netfs does actually have a valid token
@@ -163,22 +180,22 @@ extern struct fscache_cookie *__fscache_acquire_cookie(
u8,
const void *, size_t,
const void *, size_t,
- loff_t);
+ uoff_t);
extern void __fscache_use_cookie(struct fscache_cookie *, bool);
-extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const loff_t *);
+extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const uoff_t *);
extern void __fscache_relinquish_cookie(struct fscache_cookie *, bool);
-extern void __fscache_resize_cookie(struct fscache_cookie *, loff_t);
-extern void __fscache_invalidate(struct fscache_cookie *, const void *, loff_t, unsigned int);
+extern void __fscache_resize_cookie(struct fscache_cookie *, uoff_t);
+extern void __fscache_invalidate(struct fscache_cookie *, const void *, uoff_t, unsigned int);
extern int __fscache_begin_read_operation(struct netfs_cache_resources *, struct fscache_cookie *);
extern int __fscache_begin_write_operation(struct netfs_cache_resources *, struct fscache_cookie *);
void __fscache_write_to_cache(struct fscache_cookie *cookie,
struct address_space *mapping,
- loff_t start, size_t len, loff_t i_size,
+ uoff_t start, size_t len, uoff_t i_size,
netfs_io_terminated_t term_func,
void *term_func_priv,
bool using_pgpriv2, bool cond);
-extern void __fscache_clear_page_bits(struct address_space *, loff_t, size_t);
+extern void __fscache_clear_page_bits(struct address_space *, uoff_t, size_t);
/**
* fscache_acquire_volume - Register a volume as desiring caching services
@@ -249,7 +266,7 @@ struct fscache_cookie *fscache_acquire_cookie(struct fscache_volume *volume,
size_t index_key_len,
const void *aux_data,
size_t aux_data_len,
- loff_t object_size)
+ uoff_t object_size)
{
if (!fscache_volume_valid(volume))
return NULL;
@@ -286,7 +303,7 @@ static inline void fscache_use_cookie(struct fscache_cookie *cookie,
*/
static inline void fscache_unuse_cookie(struct fscache_cookie *cookie,
const void *aux_data,
- const loff_t *object_size)
+ const uoff_t *object_size)
{
if (fscache_cookie_valid(cookie))
__fscache_unuse_cookie(cookie, aux_data, object_size);
@@ -327,7 +344,7 @@ static inline void *fscache_get_aux(struct fscache_cookie *cookie)
*/
static inline
void fscache_update_aux(struct fscache_cookie *cookie,
- const void *aux_data, const loff_t *object_size)
+ const void *aux_data, const uoff_t *object_size)
{
void *p = fscache_get_aux(cookie);
@@ -343,7 +360,7 @@ extern atomic_t fscache_n_updates;
static inline
void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data,
- const loff_t *object_size)
+ const uoff_t *object_size)
{
#ifdef CONFIG_FSCACHE_STATS
atomic_inc(&fscache_n_updates);
@@ -369,7 +386,7 @@ void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data
*/
static inline
void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data,
- const loff_t *object_size)
+ const uoff_t *object_size)
{
if (fscache_cookie_enabled(cookie))
__fscache_update_cookie(cookie, aux_data, object_size);
@@ -386,7 +403,7 @@ void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data,
* description.
*/
static inline
-void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size)
+void fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size)
{
if (fscache_cookie_enabled(cookie))
__fscache_resize_cookie(cookie, new_size);
@@ -413,7 +430,7 @@ void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size)
*/
static inline
void fscache_invalidate(struct fscache_cookie *cookie,
- const void *aux_data, loff_t size, unsigned int flags)
+ const void *aux_data, uoff_t size, unsigned int flags)
{
if (fscache_cookie_enabled(cookie))
__fscache_invalidate(cookie, aux_data, size, flags);
@@ -502,7 +519,7 @@ static inline void fscache_end_operation(struct netfs_cache_resources *cres)
*/
static inline
int fscache_read(struct netfs_cache_resources *cres,
- loff_t start_pos,
+ uoff_t start_pos,
struct iov_iter *iter,
enum netfs_read_from_hole read_hole,
netfs_io_terminated_t term_func,
@@ -561,7 +578,7 @@ int fscache_begin_write_operation(struct netfs_cache_resources *cres,
*/
static inline
int fscache_write(struct netfs_cache_resources *cres,
- loff_t start_pos,
+ uoff_t start_pos,
struct iov_iter *iter,
netfs_io_terminated_t term_func,
void *term_func_priv)
@@ -581,7 +598,7 @@ int fscache_write(struct netfs_cache_resources *cres,
* waiting.
*/
static inline void fscache_clear_page_bits(struct address_space *mapping,
- loff_t start, size_t len,
+ uoff_t start, size_t len,
bool caching)
{
if (caching)
@@ -615,7 +632,7 @@ static inline void fscache_clear_page_bits(struct address_space *mapping,
*/
static inline void fscache_write_to_cache(struct fscache_cookie *cookie,
struct address_space *mapping,
- loff_t start, size_t len, loff_t i_size,
+ uoff_t start, size_t len, uoff_t i_size,
netfs_io_terminated_t term_func,
void *term_func_priv,
bool using_pgpriv2, bool caching)
diff --git a/include/linux/iomap.h b/include/linux/iomap.h
index bc7ae6327dbf..59718f73c15a 100644
--- a/include/linux/iomap.h
+++ b/include/linux/iomap.h
@@ -483,13 +483,35 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno,
#define IOMAP_IOEND_BOUNDARY (1U << 2)
/* is direct I/O */
#define IOMAP_IOEND_DIRECT (1U << 3)
+/* generate integrity (PI) information */
+#ifdef CONFIG_BLK_DEV_INTEGRITY
+#define IOMAP_IOEND_INTEGRITY (1U << 4)
+#else
+#define IOMAP_IOEND_INTEGRITY 0
+#endif /* CONFIG_BLK_DEV_INTEGRITY */
/*
* Flags that if set on either ioend prevent the merge of two ioends.
* (IOMAP_IOEND_BOUNDARY also prevents merges, but only one-way)
*/
#define IOMAP_IOEND_NOMERGE_FLAGS \
- (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT)
+ (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT | \
+ IOMAP_IOEND_INTEGRITY)
+
+/* ioend flags directly implied by iomap flags */
+static inline u16 iomap_ioend_flags(const struct iomap *iomap)
+{
+ unsigned int flags = 0;
+
+ if (iomap->type == IOMAP_UNWRITTEN)
+ flags |= IOMAP_IOEND_UNWRITTEN;
+ if (iomap->flags & IOMAP_F_SHARED)
+ flags |= IOMAP_IOEND_SHARED;
+ if (iomap->flags & IOMAP_F_INTEGRITY)
+ flags |= IOMAP_IOEND_INTEGRITY;
+
+ return flags;
+}
/*
* Structure for writeback I/O completions.
@@ -500,6 +522,7 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno,
struct iomap_ioend {
struct list_head io_list; /* next ioend in chain */
u16 io_flags; /* IOMAP_IOEND_* */
+ u32 io_bvec_offset; /* offset into first bvec */
struct inode *io_inode; /* file being written to */
size_t io_size; /* size of the extent */
atomic_t io_remaining; /* completetion defer count */
@@ -517,6 +540,13 @@ static inline struct iomap_ioend *iomap_ioend_from_bio(struct bio *bio)
return container_of(bio, struct iomap_ioend, io_bio);
}
+#define BVEC_ITER_IOEND(_ioend) \
+{ \
+ .bi_sector = (_ioend)->io_sector, \
+ .bi_size = (_ioend)->io_size, \
+ .bi_offset = (_ioend)->io_bvec_offset, \
+}
+
struct iomap_writeback_ops {
/*
* Performs writeback on the passed in range
@@ -565,6 +595,7 @@ void iomap_finish_ioends(struct iomap_ioend *ioend, int error);
void iomap_ioend_try_merge(struct iomap_ioend *ioend,
struct list_head *more_ioends);
void iomap_sort_ioends(struct list_head *ioend_list);
+int iomap_ioend_integrity_verify(struct iomap_ioend *ioend);
ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio,
loff_t pos, loff_t end_pos, unsigned int dirty_len);
int iomap_ioend_writeback_submit(struct iomap_writepage_ctx *wpc, int error);
@@ -577,6 +608,11 @@ void iomap_finish_folio_write(struct inode *inode, struct folio *folio,
int iomap_writeback_folio(struct iomap_writepage_ctx *wpc, struct folio *folio);
int iomap_writepages(struct iomap_writepage_ctx *wpc);
+void iomap_bounce_read(struct iomap_ioend *orig_ioend, unsigned int minsize,
+ void (*submit_ioend)(struct iomap_ioend *ioend));
+void iomap_bounce_read_end_io(struct iomap_ioend *ioend, struct bio *orig_bio,
+ int error);
+
struct iomap_read_folio_ctx {
const struct iomap_read_ops *ops;
struct folio *cur_folio;
diff --git a/include/linux/lsm_hook_defs.h b/include/linux/lsm_hook_defs.h
index 65c9609ec207..c9561564585e 100644
--- a/include/linux/lsm_hook_defs.h
+++ b/include/linux/lsm_hook_defs.h
@@ -36,6 +36,7 @@ LSM_HOOK(int, 0, binder_transfer_file, const struct cred *from,
LSM_HOOK(int, 0, ptrace_access_check, struct task_struct *child,
unsigned int mode)
LSM_HOOK(int, 0, ptrace_traceme, struct task_struct *parent)
+LSM_HOOK(int, 0, mem_foll_force, const struct cred *subject, bool opened_by_owner)
LSM_HOOK(int, 0, capget, const struct task_struct *target, kernel_cap_t *effective,
kernel_cap_t *inheritable, kernel_cap_t *permitted)
LSM_HOOK(int, 0, capset, struct cred *new, const struct cred *old,
@@ -94,7 +95,7 @@ LSM_HOOK(int, 0, path_mkdir, const struct path *dir, struct dentry *dentry,
LSM_HOOK(int, 0, path_rmdir, const struct path *dir, struct dentry *dentry)
LSM_HOOK(int, 0, path_mknod, const struct path *dir, struct dentry *dentry,
umode_t mode, unsigned int dev)
-LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, struct mnt_idmap *idmap,
+LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, const struct mnt_idmap *idmap,
struct dentry *dentry)
LSM_HOOK(int, 0, path_truncate, const struct path *path)
LSM_HOOK(int, 0, path_symlink, const struct path *dir, struct dentry *dentry,
@@ -122,7 +123,7 @@ LSM_HOOK(int, 0, inode_init_security_anon, struct inode *inode,
const struct qstr *name, const struct inode *context_inode)
LSM_HOOK(int, 0, inode_create, struct inode *dir, struct dentry *dentry,
umode_t mode)
-LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, struct mnt_idmap *idmap,
+LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, const struct mnt_idmap *idmap,
struct inode *inode)
LSM_HOOK(int, 0, inode_link, struct dentry *old_dentry, struct inode *dir,
struct dentry *new_dentry)
@@ -140,39 +141,39 @@ LSM_HOOK(int, 0, inode_readlink, struct dentry *dentry)
LSM_HOOK(int, 0, inode_follow_link, struct dentry *dentry, struct inode *inode,
bool rcu)
LSM_HOOK(int, 0, inode_permission, struct inode *inode, int mask)
-LSM_HOOK(int, 0, inode_setattr, struct mnt_idmap *idmap, struct dentry *dentry,
+LSM_HOOK(int, 0, inode_setattr, const struct mnt_idmap *idmap, struct dentry *dentry,
struct iattr *attr)
-LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, struct mnt_idmap *idmap,
+LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, const struct mnt_idmap *idmap,
struct dentry *dentry, int ia_valid)
LSM_HOOK(int, 0, inode_getattr, const struct path *path)
LSM_HOOK(int, 0, inode_xattr_skipcap, const char *name)
-LSM_HOOK(int, 0, inode_setxattr, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_setxattr, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name, const void *value,
size_t size, int flags)
LSM_HOOK(void, LSM_RET_VOID, inode_post_setxattr, struct dentry *dentry,
const char *name, const void *value, size_t size, int flags)
LSM_HOOK(int, 0, inode_getxattr, struct dentry *dentry, const char *name)
LSM_HOOK(int, 0, inode_listxattr, struct dentry *dentry)
-LSM_HOOK(int, 0, inode_removexattr, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_removexattr, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name)
LSM_HOOK(void, LSM_RET_VOID, inode_post_removexattr, struct dentry *dentry,
const char *name)
LSM_HOOK(int, 0, inode_file_setattr, struct dentry *dentry, struct file_kattr *fa)
LSM_HOOK(int, 0, inode_file_getattr, struct dentry *dentry, struct file_kattr *fa)
-LSM_HOOK(int, 0, inode_set_acl, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_set_acl, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name, struct posix_acl *kacl)
LSM_HOOK(void, LSM_RET_VOID, inode_post_set_acl, struct dentry *dentry,
const char *acl_name, struct posix_acl *kacl)
-LSM_HOOK(int, 0, inode_get_acl, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_get_acl, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name)
-LSM_HOOK(int, 0, inode_remove_acl, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_remove_acl, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name)
-LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, struct mnt_idmap *idmap,
+LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name)
LSM_HOOK(int, 0, inode_need_killpriv, struct dentry *dentry)
-LSM_HOOK(int, 0, inode_killpriv, struct mnt_idmap *idmap,
+LSM_HOOK(int, 0, inode_killpriv, const struct mnt_idmap *idmap,
struct dentry *dentry)
-LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, struct mnt_idmap *idmap,
+LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, const struct mnt_idmap *idmap,
struct inode *inode, const char *name, void **buffer, bool alloc)
LSM_HOOK(int, -EOPNOTSUPP, inode_setsecurity, struct inode *inode,
const char *name, const void *value, size_t size, int flags)
diff --git a/include/linux/mnt_idmapping.h b/include/linux/mnt_idmapping.h
index e71a6070a8f8..78eeef4c2996 100644
--- a/include/linux/mnt_idmapping.h
+++ b/include/linux/mnt_idmapping.h
@@ -8,8 +8,8 @@
struct mnt_idmap;
struct user_namespace;
-extern struct mnt_idmap nop_mnt_idmap;
-extern struct mnt_idmap invalid_mnt_idmap;
+extern const struct mnt_idmap nop_mnt_idmap;
+extern const struct mnt_idmap invalid_mnt_idmap;
extern struct user_namespace init_user_ns;
typedef struct {
@@ -121,19 +121,19 @@ static inline bool vfsgid_eq_kgid(vfsgid_t vfsgid, kgid_t kgid)
int vfsgid_in_group_p(vfsgid_t vfsgid);
-struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap);
-void mnt_idmap_put(struct mnt_idmap *idmap);
+const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap);
+void mnt_idmap_put(const struct mnt_idmap *idmap);
-vfsuid_t make_vfsuid(struct mnt_idmap *idmap,
+vfsuid_t make_vfsuid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns, kuid_t kuid);
-vfsgid_t make_vfsgid(struct mnt_idmap *idmap,
+vfsgid_t make_vfsgid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns, kgid_t kgid);
-kuid_t from_vfsuid(struct mnt_idmap *idmap,
+kuid_t from_vfsuid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns, vfsuid_t vfsuid);
-kgid_t from_vfsgid(struct mnt_idmap *idmap,
+kgid_t from_vfsgid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns, vfsgid_t vfsgid);
/**
@@ -148,7 +148,7 @@ kgid_t from_vfsgid(struct mnt_idmap *idmap,
*
* Return: true if @vfsuid has a mapping in the filesystem, false if not.
*/
-static inline bool vfsuid_has_fsmapping(struct mnt_idmap *idmap,
+static inline bool vfsuid_has_fsmapping(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns,
vfsuid_t vfsuid)
{
@@ -186,7 +186,7 @@ static inline kuid_t vfsuid_into_kuid(vfsuid_t vfsuid)
*
* Return: true if @vfsgid has a mapping in the filesystem, false if not.
*/
-static inline bool vfsgid_has_fsmapping(struct mnt_idmap *idmap,
+static inline bool vfsgid_has_fsmapping(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns,
vfsgid_t vfsgid)
{
@@ -225,7 +225,7 @@ static inline kgid_t vfsgid_into_kgid(vfsgid_t vfsgid)
*
* Return: the caller's current fsuid mapped up according to @idmap.
*/
-static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap,
+static inline kuid_t mapped_fsuid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns)
{
return from_vfsuid(idmap, fs_userns, VFSUIDT_INIT(current_fsuid()));
@@ -244,7 +244,7 @@ static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap,
*
* Return: the caller's current fsgid mapped up according to @idmap.
*/
-static inline kgid_t mapped_fsgid(struct mnt_idmap *idmap,
+static inline kgid_t mapped_fsgid(const struct mnt_idmap *idmap,
struct user_namespace *fs_userns)
{
return from_vfsgid(idmap, fs_userns, VFSGIDT_INIT(current_fsgid()));
diff --git a/include/linux/mount.h b/include/linux/mount.h
index acfe7ef86a1b..e90ccafef281 100644
--- a/include/linux/mount.h
+++ b/include/linux/mount.h
@@ -59,10 +59,10 @@ struct vfsmount {
struct dentry *mnt_root; /* root of the mounted tree */
struct super_block *mnt_sb; /* pointer to superblock */
int mnt_flags;
- struct mnt_idmap *mnt_idmap;
+ const struct mnt_idmap *mnt_idmap;
} __randomize_layout;
-static inline struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt)
+static inline const struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt)
{
/* Pairs with smp_store_release() in do_idmap_mount(). */
return READ_ONCE(mnt->mnt_idmap);
diff --git a/include/linux/namei.h b/include/linux/namei.h
index 86d657b24fc6..c4436e5c2ba6 100644
--- a/include/linux/namei.h
+++ b/include/linux/namei.h
@@ -32,8 +32,9 @@ enum { MAX_NESTED_LINKS = 8 };
#define LOOKUP_CREATE BIT(17) /* ... in object creation */
#define LOOKUP_EXCL BIT(18) /* ... in target must not exist */
#define LOOKUP_RENAME_TARGET BIT(19) /* ... in destination of rename() */
+#define LOOKUP_SHARED BIT(20) /* Parent lock is held shared */
-/* 4 spare bits for intent */
+/* 3 spare bits for intent */
/* Scoping flags for lookup. */
#define LOOKUP_NO_SYMLINKS BIT(24) /* No symlink crossing. */
@@ -70,24 +71,24 @@ extern struct dentry *try_lookup_noperm(struct qstr *, struct dentry *);
extern struct dentry *lookup_noperm(struct qstr *, struct dentry *);
extern struct dentry *lookup_noperm_unlocked(struct qstr *, struct dentry *);
extern struct dentry *lookup_noperm_positive_unlocked(struct qstr *, struct dentry *);
-struct dentry *lookup_one(struct mnt_idmap *, struct qstr *, struct dentry *);
-struct dentry *lookup_one_unlocked(struct mnt_idmap *idmap,
+struct dentry *lookup_one(const struct mnt_idmap *, struct qstr *, struct dentry *);
+struct dentry *lookup_one_unlocked(const struct mnt_idmap *idmap,
struct qstr *name, struct dentry *base);
-struct dentry *lookup_one_positive_unlocked(struct mnt_idmap *idmap,
+struct dentry *lookup_one_positive_unlocked(const struct mnt_idmap *idmap,
struct qstr *name,
struct dentry *base);
-struct dentry *lookup_one_positive_killable(struct mnt_idmap *idmap,
+struct dentry *lookup_one_positive_killable(const struct mnt_idmap *idmap,
struct qstr *name,
struct dentry *base);
-struct dentry *start_creating(struct mnt_idmap *idmap, struct dentry *parent,
+struct dentry *start_creating(const struct mnt_idmap *idmap, struct dentry *parent,
struct qstr *name);
-struct dentry *start_removing(struct mnt_idmap *idmap, struct dentry *parent,
+struct dentry *start_removing(const struct mnt_idmap *idmap, struct dentry *parent,
struct qstr *name);
-struct dentry *start_creating_killable(struct mnt_idmap *idmap,
+struct dentry *start_creating_killable(const struct mnt_idmap *idmap,
struct dentry *parent,
struct qstr *name);
-struct dentry *start_removing_killable(struct mnt_idmap *idmap,
+struct dentry *start_removing_killable(const struct mnt_idmap *idmap,
struct dentry *parent,
struct qstr *name);
struct dentry *start_creating_noperm(struct dentry *parent, struct qstr *name);
diff --git a/include/linux/netfs.h b/include/linux/netfs.h
index b4dd32863dd4..67e010b6994b 100644
--- a/include/linux/netfs.h
+++ b/include/linux/netfs.h
@@ -22,6 +22,7 @@
enum netfs_sreq_ref_trace;
typedef struct mempool mempool_t;
+struct fscache_occupancy;
struct folio_queue;
/**
@@ -62,8 +63,8 @@ struct netfs_inode {
struct fscache_cookie *cache;
#endif
struct list_head wb_queue; /* Queue of processes wanting to do writeback */
- loff_t _remote_i_size; /* Size of the remote file */
- loff_t _zero_point; /* Size after which we assume there's no data
+ uoff_t _remote_i_size; /* Size of the remote file */
+ uoff_t _zero_point; /* Size after which we assume there's no data
* on the server */
spinlock_t lock; /* Lock covering wb_queue */
atomic_t io_count; /* Number of outstanding reqs */
@@ -125,6 +126,12 @@ static inline struct netfs_group *netfs_folio_group(struct folio *folio)
return priv;
}
+enum netfs_cache_collect {
+ NETFS_CACHE_COLLECT_WRITE_GAP, /* Gap in collection, no state either way */
+ NETFS_CACHE_COLLECT_WRITE_DATA, /* Currently collecting good writes */
+ NETFS_CACHE_COLLECT_WRITE_CANCEL, /* Currently collecting cancelled writes */
+};
+
/*
* Stream of I/O subrequests going to a particular destination, such as the
* server or the local cache. This is mainly intended for writing where we may
@@ -142,7 +149,7 @@ struct netfs_io_stream {
void (*issue_write)(struct netfs_io_subrequest *subreq);
/* Collection tracking */
struct list_head subrequests; /* Contributory I/O operations */
- unsigned long long collected_to; /* Position we've collected results to */
+ uoff_t collected_to; /* Position we've collected results to */
size_t transferred; /* The amount transferred from this stream */
unsigned short error; /* Aggregate error for the stream */
enum netfs_io_source source; /* Where to read from/write to */
@@ -152,6 +159,7 @@ struct netfs_io_stream {
bool need_retry; /* T if this stream needs retrying */
bool failed; /* T if this stream failed */
bool transferred_valid; /* T is ->transferred is valid */
+ enum netfs_cache_collect cache_collect; /* Current writeback cache collect state */
};
/*
@@ -161,8 +169,11 @@ struct netfs_cache_resources {
const struct netfs_cache_ops *ops;
void *cache_priv;
void *cache_priv2;
- unsigned int debug_id; /* Cookie debug ID */
+ uoff_t cache_i_size; /* Initial size of cache file */
+ unsigned int cookie_id; /* Cache cookie debug ID */
+ unsigned int object_id; /* Cache object debug ID */
unsigned int inval_counter; /* object->inval_counter at begin_op */
+ unsigned int dio_size; /* DIO block size */
};
/*
@@ -177,7 +188,7 @@ struct netfs_io_subrequest {
struct work_struct work;
struct list_head rreq_link; /* Link in rreq->subrequests */
struct iov_iter io_iter; /* Iterator for this subrequest */
- unsigned long long start; /* Where to start the I/O */
+ uoff_t start; /* Where to start the I/O */
size_t len; /* Size of the I/O */
size_t transferred; /* Amount of data transferred */
refcount_t ref;
@@ -196,6 +207,7 @@ struct netfs_io_subrequest {
#define NETFS_SREQ_IN_PROGRESS 8 /* Unlocked when the subrequest completes */
#define NETFS_SREQ_NEED_RETRY 9 /* Set if the filesystem requests a retry */
#define NETFS_SREQ_FAILED 10 /* Set if the subreq failed unretryably */
+#define NETFS_SREQ_CANCELLED 11 /* Set if the subreq was cancelled by netfslib */
};
enum netfs_io_origin {
@@ -208,7 +220,6 @@ enum netfs_io_origin {
NETFS_DIO_READ, /* This is a direct I/O read */
NETFS_WRITEBACK, /* This write was triggered by writepages */
NETFS_WRITEBACK_SINGLE, /* This monolithic write was triggered by writepages */
- NETFS_WRITETHROUGH, /* This write was made by netfs_perform_write() */
NETFS_UNBUFFERED_WRITE, /* This is an unbuffered write */
NETFS_DIO_WRITE, /* This is a direct I/O write */
NETFS_PGPRIV2_COPY_TO_CACHE, /* [DEPRECATED] This is writing read data to the cache */
@@ -243,17 +254,18 @@ struct netfs_io_request {
void *netfs_priv; /* Private data for the netfs */
void *netfs_priv2; /* Private data for the netfs */
struct bio_vec *direct_bv; /* DIO buffer list (when handling iovec-iter) */
- unsigned long long submitted; /* Amount submitted for I/O so far */
- unsigned long long len; /* Length of the request */
+ uoff_t submitted; /* Amount submitted for I/O so far */
+ uoff_t len; /* Length of the request */
size_t transferred; /* Amount to be indicated as transferred */
size_t progress_at; /* Report read progress when hit this much read */
long error; /* 0 or error that occurred */
- unsigned long long i_size; /* Size of the file */
- unsigned long long start; /* Start position */
+ uoff_t i_size; /* Size of the file */
+ uoff_t start; /* Start position */
atomic64_t issued_to; /* Write issuer folio cursor */
- unsigned long long collected_to; /* Point we've collected to */
- unsigned long long cleaned_to; /* Position we've cleaned folios to */
- unsigned long long abandon_to; /* Position to abandon folios to */
+ uoff_t collected_to; /* Point we've collected to */
+ uoff_t cache_coll_to; /* Point the cache has collected to */
+ uoff_t cleaned_to; /* Position we've cleaned folios to */
+ uoff_t abandon_to; /* Position to abandon folios to */
const struct folio *no_unlock_folio; /* Don't unlock this folio after read */
gfp_t gfp; /* GFP flags to use */
unsigned int direct_bv_count; /* Number of elements in direct_bv[] */
@@ -273,14 +285,18 @@ struct netfs_io_request {
#define NETFS_RREQ_FAILED 3 /* The request failed */
#define NETFS_RREQ_RETRYING 4 /* Set if we're in the retry path */
#define NETFS_RREQ_SHORT_TRANSFER 5 /* Set if we have a short transfer */
-#define NETFS_RREQ_OFFLOAD_COLLECTION 8 /* Offload collection to workqueue */
-#define NETFS_RREQ_NO_UNLOCK_FOLIO 9 /* Don't unlock no_unlock_folio on completion */
+#define NETFS_RREQ_CACHE_STOP 8 /* Set to stop caching (ENOBUFS or error) */
+#define NETFS_RREQ_CACHE_ERROR 9 /* Set if we got an error from the cache */
#define NETFS_RREQ_CANCEL_CACHING 10 /* Set to cancel caching */
-#define NETFS_RREQ_UPLOAD_TO_SERVER 11 /* Need to write to the server */
-#define NETFS_RREQ_USE_IO_ITER 12 /* Use ->io_iter rather than ->i_pages */
+#define NETFS_RREQ_OFFLOAD_COLLECTION 12 /* Offload collection to workqueue */
+#define NETFS_RREQ_NO_UNLOCK_FOLIO 13 /* Don't unlock no_unlock_folio on completion */
+#define NETFS_RREQ_UPLOAD_TO_SERVER 14 /* Need to write to the server */
+#define NETFS_RREQ_USE_IO_ITER 15 /* Use ->io_iter rather than ->i_pages */
#define NETFS_RREQ_NEED_PUT_RA_REFS 17 /* Need to put the folio refs RA gave us */
+#ifdef CONFIG_NETFS_PGPRIV2
#define NETFS_RREQ_USE_PGPRIV2 31 /* [DEPRECATED] Use PG_private_2 to mark
* write to cache on read */
+#endif
const struct netfs_request_ops *netfs_ops;
};
@@ -299,12 +315,12 @@ struct netfs_request_ops {
int (*prepare_read)(struct netfs_io_subrequest *subreq);
void (*issue_read)(struct netfs_io_subrequest *subreq);
bool (*is_still_valid)(struct netfs_io_request *rreq);
- int (*check_write_begin)(struct file *file, loff_t pos, unsigned len,
+ int (*check_write_begin)(struct file *file, uoff_t pos, unsigned len,
struct folio **foliop, void **_fsdata);
void (*done)(struct netfs_io_request *rreq);
/* Modification handling */
- void (*update_i_size)(struct inode *inode, loff_t i_size);
+ void (*update_i_size)(struct inode *inode, uoff_t i_size);
void (*post_modify)(struct inode *inode);
/* Write request handling */
@@ -332,7 +348,7 @@ struct netfs_cache_ops {
/* Read data from the cache */
int (*read)(struct netfs_cache_resources *cres,
- loff_t start_pos,
+ uoff_t start_pos,
struct iov_iter *iter,
enum netfs_read_from_hole read_hole,
netfs_io_terminated_t term_func,
@@ -340,7 +356,7 @@ struct netfs_cache_ops {
/* Write data to the cache */
int (*write)(struct netfs_cache_resources *cres,
- loff_t start_pos,
+ uoff_t start_pos,
struct iov_iter *iter,
netfs_io_terminated_t term_func,
void *term_func_priv);
@@ -350,15 +366,14 @@ struct netfs_cache_ops {
/* Expand readahead request */
void (*expand_readahead)(struct netfs_cache_resources *cres,
- unsigned long long *_start,
- unsigned long long *_len,
- unsigned long long i_size);
+ uoff_t *_start,
+ uoff_t *_len,
+ uoff_t i_size);
/* Prepare a read operation, shortening it to a cached/uncached
* boundary as appropriate.
*/
- enum netfs_io_source (*prepare_read)(struct netfs_io_subrequest *subreq,
- unsigned long long i_size);
+ int (*prepare_read)(struct netfs_io_subrequest *subreq);
/* Prepare a write subrequest, working out if we're allowed to do it
* and finding out the maximum amount of data to gather before
@@ -371,15 +386,24 @@ struct netfs_cache_ops {
* actually do.
*/
int (*prepare_write)(struct netfs_cache_resources *cres,
- loff_t *_start, size_t *_len, size_t upper_len,
- loff_t i_size, bool no_space_allocated_yet);
+ uoff_t *_start, size_t *_len, size_t upper_len,
+ uoff_t i_size, bool no_space_allocated_yet);
/* Query the occupancy of the cache in a region, returning where the
* next chunk of data starts and how long it is.
*/
int (*query_occupancy)(struct netfs_cache_resources *cres,
- loff_t start, size_t len, size_t granularity,
- loff_t *_data_start, size_t *_data_len);
+ struct fscache_occupancy *occ);
+
+ /* Collect the result of buffered writeback to the cache. This
+ * includes copying a read to the cache. block_type is one of:
+ * - NETFS_CACHE_COLLECT_WRITE_DATA for a block of data
+ * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped
+ * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a cancellation gap
+ */
+ void (*collect_write)(struct netfs_io_request *wreq,
+ uoff_t start, size_t len,
+ enum netfs_cache_collect block_type);
};
/* High-level read API. */
@@ -410,7 +434,7 @@ struct readahead_control;
void netfs_readahead(struct readahead_control *);
int netfs_read_folio(struct file *, struct folio *);
int netfs_write_begin(struct netfs_inode *, struct file *,
- struct address_space *, loff_t pos, unsigned int len,
+ struct address_space *, uoff_t pos, unsigned int len,
struct folio **, void **fsdata);
int netfs_writepages(struct address_space *mapping,
struct writeback_control *wbc);
@@ -488,10 +512,10 @@ static inline struct netfs_inode *netfs_inode(struct inode *inode)
* cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit
* archs it makes no difference if preempt is enabled or not.
*/
-static inline unsigned long long netfs_read_remote_i_size(const struct inode *inode)
+static inline uoff_t netfs_read_remote_i_size(const struct inode *inode)
{
const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode);
- unsigned long long remote_i_size;
+ uoff_t remote_i_size;
#if BITS_PER_LONG==32 && defined(CONFIG_SMP)
unsigned int seq;
@@ -526,7 +550,7 @@ static inline unsigned long long netfs_read_remote_i_size(const struct inode *in
* spinning forever.
*/
static inline void netfs_write_remote_i_size(struct inode *inode,
- unsigned long long remote_i_size)
+ uoff_t remote_i_size)
{
struct netfs_inode *ictx = netfs_inode(inode);
@@ -563,10 +587,10 @@ static inline void netfs_write_remote_i_size(struct inode *inode,
* cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit
* archs it makes no difference if preempt is enabled or not.
*/
-static inline unsigned long long netfs_read_zero_point(const struct inode *inode)
+static inline uoff_t netfs_read_zero_point(const struct inode *inode)
{
struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode);
- unsigned long long zero_point;
+ uoff_t zero_point;
#if BITS_PER_LONG==32 && defined(CONFIG_SMP)
unsigned int seq;
@@ -601,7 +625,7 @@ static inline unsigned long long netfs_read_zero_point(const struct inode *inode
* forever.
*/
static inline void netfs_write_zero_point(struct inode *inode,
- unsigned long long zero_point)
+ uoff_t zero_point)
{
struct netfs_inode *ictx = netfs_inode(inode);
@@ -642,9 +666,9 @@ static inline void netfs_write_zero_point(struct inode *inode,
* archs it makes no difference if preempt is enabled or not.
*/
static inline void netfs_read_sizes(const struct inode *inode,
- unsigned long long *i_size,
- unsigned long long *remote_i_size,
- unsigned long long *zero_point)
+ uoff_t *i_size,
+ uoff_t *remote_i_size,
+ uoff_t *zero_point)
{
const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode);
#if BITS_PER_LONG==32 && defined(CONFIG_SMP)
@@ -690,9 +714,9 @@ static inline void netfs_read_sizes(const struct inode *inode,
* forever.
*/
static inline void netfs_write_sizes(struct inode *inode,
- unsigned long long i_size,
- unsigned long long remote_i_size,
- unsigned long long zero_point)
+ uoff_t i_size,
+ uoff_t remote_i_size,
+ uoff_t zero_point)
{
struct netfs_inode *ictx = netfs_inode(inode);
@@ -760,7 +784,7 @@ static inline void netfs_inode_init(struct netfs_inode *ctx,
* Inform the netfs lib that a file got resized so that it can adjust its state.
*/
static inline void netfs_resize_file(struct netfs_inode *ictx,
- unsigned long long new_i_size,
+ uoff_t new_i_size,
bool changed_on_server)
{
#if BITS_PER_LONG==32 && defined(CONFIG_SMP)
diff --git a/include/linux/nfs_fs.h b/include/linux/nfs_fs.h
index b85a73ae7919..d2c716322c6f 100644
--- a/include/linux/nfs_fs.h
+++ b/include/linux/nfs_fs.h
@@ -437,11 +437,11 @@ extern int nfs_refresh_inode(struct inode *, struct nfs_fattr *);
extern int nfs_post_op_update_inode(struct inode *inode, struct nfs_fattr *fattr);
extern int nfs_post_op_update_inode_force_wcc(struct inode *inode, struct nfs_fattr *fattr);
extern int nfs_post_op_update_inode_force_wcc_locked(struct inode *inode, struct nfs_fattr *fattr);
-extern int nfs_getattr(struct mnt_idmap *, const struct path *,
+extern int nfs_getattr(const struct mnt_idmap *, const struct path *,
struct kstat *, u32, unsigned int);
extern void nfs_access_add_cache(struct inode *, struct nfs_access_entry *, const struct cred *);
extern void nfs_access_set_mask(struct nfs_access_entry *, u32);
-extern int nfs_permission(struct mnt_idmap *, struct inode *, int);
+extern int nfs_permission(const struct mnt_idmap *, struct inode *, int);
extern int nfs_open(struct inode *, struct file *);
extern int nfs_attribute_cache_expired(struct inode *inode);
extern int nfs_revalidate_inode(struct inode *inode, unsigned long flags);
@@ -450,7 +450,7 @@ extern int nfs_clear_invalid_mapping(struct address_space *mapping);
extern bool nfs_mapping_need_revalidate_inode(struct inode *inode);
extern int nfs_revalidate_mapping(struct inode *inode, struct address_space *mapping);
extern int nfs_revalidate_mapping_rcu(struct inode *inode);
-extern int nfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *);
+extern int nfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *);
extern void nfs_setattr_update_inode(struct inode *inode, struct iattr *attr, struct nfs_fattr *);
extern void nfs_setsecurity(struct inode *inode, struct nfs_fattr *fattr);
extern struct nfs_open_context *get_nfs_open_context(struct nfs_open_context *ctx);
diff --git a/include/linux/posix_acl.h b/include/linux/posix_acl.h
index 62d497763e25..caf500bed993 100644
--- a/include/linux/posix_acl.h
+++ b/include/linux/posix_acl.h
@@ -74,20 +74,20 @@ extern int __posix_acl_create(struct posix_acl **, gfp_t, umode_t *);
extern int __posix_acl_chmod(struct posix_acl **, gfp_t, umode_t);
extern struct posix_acl *get_posix_acl(struct inode *, int);
-int set_posix_acl(struct mnt_idmap *, struct dentry *, int,
+int set_posix_acl(const struct mnt_idmap *, struct dentry *, int,
struct posix_acl *);
struct posix_acl *get_cached_acl_rcu(struct inode *inode, int type);
struct posix_acl *posix_acl_clone(const struct posix_acl *acl, gfp_t flags);
#ifdef CONFIG_FS_POSIX_ACL
-int posix_acl_chmod(struct mnt_idmap *, struct dentry *, umode_t);
+int posix_acl_chmod(const struct mnt_idmap *, struct dentry *, umode_t);
extern int posix_acl_create(struct inode *, umode_t *, struct posix_acl **,
struct posix_acl **);
-int posix_acl_update_mode(struct mnt_idmap *, struct inode *, umode_t *,
+int posix_acl_update_mode(const struct mnt_idmap *, struct inode *, umode_t *,
struct posix_acl **);
-int simple_set_acl(struct mnt_idmap *, struct dentry *,
+int simple_set_acl(const struct mnt_idmap *, struct dentry *,
struct posix_acl *, int);
extern int simple_acl_create(struct inode *, struct inode *);
@@ -96,7 +96,7 @@ void set_cached_acl(struct inode *inode, int type, struct posix_acl *acl);
void forget_cached_acl(struct inode *inode, int type);
void forget_all_cached_acls(struct inode *inode);
int posix_acl_valid(struct user_namespace *, const struct posix_acl *);
-int posix_acl_permission(struct mnt_idmap *, struct inode *,
+int posix_acl_permission(const struct mnt_idmap *, struct inode *,
const struct posix_acl *, int);
static inline void cache_no_acl(struct inode *inode)
@@ -105,16 +105,16 @@ static inline void cache_no_acl(struct inode *inode)
inode->i_default_acl = NULL;
}
-int vfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry,
+int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry,
const char *acl_name, struct posix_acl *kacl);
-struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap,
+struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name);
-int vfs_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry,
+int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry,
const char *acl_name);
int posix_acl_listxattr(struct inode *inode, char **buffer,
ssize_t *remaining_size);
#else
-static inline int posix_acl_chmod(struct mnt_idmap *idmap,
+static inline int posix_acl_chmod(const struct mnt_idmap *idmap,
struct dentry *dentry, umode_t mode)
{
return 0;
@@ -141,21 +141,21 @@ static inline void forget_all_cached_acls(struct inode *inode)
{
}
-static inline int vfs_set_acl(struct mnt_idmap *idmap,
+static inline int vfs_set_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name,
struct posix_acl *acl)
{
return -EOPNOTSUPP;
}
-static inline struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap,
+static inline struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name)
{
return ERR_PTR(-EOPNOTSUPP);
}
-static inline int vfs_remove_acl(struct mnt_idmap *idmap,
+static inline int vfs_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name)
{
return -EOPNOTSUPP;
diff --git a/include/linux/quotaops.h b/include/linux/quotaops.h
index f9c0f9d7c9d9..0c64ca674e77 100644
--- a/include/linux/quotaops.h
+++ b/include/linux/quotaops.h
@@ -20,7 +20,7 @@ static inline struct quota_info *sb_dqopt(struct super_block *sb)
}
/* i_rwsem must being held */
-static inline bool is_quota_modification(struct mnt_idmap *idmap,
+static inline bool is_quota_modification(const struct mnt_idmap *idmap,
struct inode *inode, struct iattr *ia)
{
return ((ia->ia_valid & ATTR_SIZE) ||
@@ -109,7 +109,7 @@ int dquot_set_dqblk(struct super_block *sb, struct kqid id,
struct qc_dqblk *di);
int __dquot_transfer(struct inode *inode, struct dquot **transfer_to);
-int dquot_transfer(struct mnt_idmap *idmap, struct inode *inode,
+int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode,
struct iattr *iattr);
static inline struct mem_dqinfo *sb_dqinfo(struct super_block *sb, int type)
@@ -229,7 +229,7 @@ static inline void dquot_free_inode(struct inode *inode)
{
}
-static inline int dquot_transfer(struct mnt_idmap *idmap,
+static inline int dquot_transfer(const struct mnt_idmap *idmap,
struct inode *inode, struct iattr *iattr)
{
return 0;
diff --git a/include/linux/sched.h b/include/linux/sched.h
index d35ae49a991f..78bfc0e32e56 100644
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -1817,7 +1817,7 @@ extern struct pid __rcu *cad_pid;
* I am cleaning dirty pages from some other bdi. */
#define PF_KTHREAD 0x00200000 /* I am a kernel thread */
#define PF_RANDOMIZE 0x00400000 /* Randomize virtual address space */
-#define PF__HOLE__00800000 0x00800000
+#define PF_NO_NOTIFY_SIGNAL 0x00800000 /* see no_notify_signal_save() */
#define PF__HOLE__01000000 0x01000000
#define PF__HOLE__02000000 0x02000000
#define PF_NO_SETAFFINITY 0x04000000 /* Userland is not allowed to meddle with cpus_mask */
diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h
index d45a5476b97d..d9419dc902f6 100644
--- a/include/linux/sched/signal.h
+++ b/include/linux/sched/signal.h
@@ -2,6 +2,7 @@
#ifndef _LINUX_SCHED_SIGNAL_H
#define _LINUX_SCHED_SIGNAL_H
+#include <linux/cleanup.h>
#include <linux/rculist.h>
#include <linux/signal.h>
#include <linux/sched.h>
@@ -79,9 +80,9 @@ struct core_thread {
};
struct core_state {
- atomic_t nr_threads;
- struct core_thread dumper;
- struct completion startup;
+ /* Threads the dumper still waits for. */
+ atomic_t threads_remaining;
+ struct core_thread *tasks;
};
/*
@@ -384,14 +385,36 @@ static inline int task_sigpending(struct task_struct *p)
return unlikely(test_tsk_thread_flag(p,TIF_SIGPENDING));
}
+/* Prevent TIF_NOTIFY_SIGNAL from interrupting this task. */
+static inline unsigned int no_notify_signal_save(void)
+{
+ unsigned int flags = current->flags;
+
+ current->flags |= PF_NO_NOTIFY_SIGNAL;
+ return flags;
+}
+
+/* Restore the previous PF_NO_NOTIFY_SIGNAL state. */
+static inline void no_notify_signal_restore(unsigned int flags)
+{
+ current_restore_flags(flags, PF_NO_NOTIFY_SIGNAL);
+}
+
+DEFINE_LOCK_GUARD_0(no_notify_signal,
+ _T->flags = no_notify_signal_save(),
+ no_notify_signal_restore(_T->flags),
+ unsigned int flags)
+
static inline int signal_pending(struct task_struct *p)
{
/*
* TIF_NOTIFY_SIGNAL isn't really a signal, but it requires the same
* behavior in terms of ensuring that we break out of wait loops
- * so that notify signal callbacks can be processed.
+ * so that notify signal callbacks can be processed. Not for a task
+ * that asked not to be interrupted by it, see no_notify_signal_save().
*/
- if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL)))
+ if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL)) &&
+ likely(!(READ_ONCE(p->flags) & PF_NO_NOTIFY_SIGNAL)))
return 1;
return task_sigpending(p);
}
diff --git a/include/linux/security.h b/include/linux/security.h
index 153e9043058f..f7ff72ff956b 100644
--- a/include/linux/security.h
+++ b/include/linux/security.h
@@ -185,11 +185,11 @@ extern int cap_capset(struct cred *new, const struct cred *old,
extern int cap_bprm_creds_from_file(struct linux_binprm *bprm, const struct file *file);
int cap_inode_setxattr(struct dentry *dentry, const char *name,
const void *value, size_t size, int flags);
-int cap_inode_removexattr(struct mnt_idmap *idmap,
+int cap_inode_removexattr(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name);
int cap_inode_need_killpriv(struct dentry *dentry);
-int cap_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry);
-int cap_inode_getsecurity(struct mnt_idmap *idmap,
+int cap_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry);
+int cap_inode_getsecurity(const struct mnt_idmap *idmap,
struct inode *inode, const char *name, void **buffer,
bool alloc);
extern int cap_mmap_addr(unsigned long addr);
@@ -338,6 +338,7 @@ int security_binder_transfer_file(const struct cred *from,
const struct cred *to, const struct file *file);
int security_ptrace_access_check(struct task_struct *child, unsigned int mode);
int security_ptrace_traceme(struct task_struct *parent);
+int security_mem_foll_force(const struct cred *subject, bool opened_by_owner);
int security_capget(const struct task_struct *target,
kernel_cap_t *effective,
kernel_cap_t *inheritable,
@@ -405,7 +406,7 @@ int security_inode_init_security_anon(struct inode *inode,
const struct qstr *name,
const struct inode *context_inode);
int security_inode_create(struct inode *dir, struct dentry *dentry, umode_t mode);
-void security_inode_post_create_tmpfile(struct mnt_idmap *idmap,
+void security_inode_post_create_tmpfile(const struct mnt_idmap *idmap,
struct inode *inode);
int security_inode_link(struct dentry *old_dentry, struct inode *dir,
struct dentry *new_dentry);
@@ -422,31 +423,31 @@ int security_inode_readlink(struct dentry *dentry);
int security_inode_follow_link(struct dentry *dentry, struct inode *inode,
bool rcu);
int security_inode_permission(struct inode *inode, int mask);
-int security_inode_setattr(struct mnt_idmap *idmap,
+int security_inode_setattr(const struct mnt_idmap *idmap,
struct dentry *dentry, struct iattr *attr);
-void security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
+void security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry,
int ia_valid);
int security_inode_getattr(const struct path *path);
-int security_inode_setxattr(struct mnt_idmap *idmap,
+int security_inode_setxattr(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name,
const void *value, size_t size, int flags);
-int security_inode_set_acl(struct mnt_idmap *idmap,
+int security_inode_set_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name,
struct posix_acl *kacl);
void security_inode_post_set_acl(struct dentry *dentry, const char *acl_name,
struct posix_acl *kacl);
-int security_inode_get_acl(struct mnt_idmap *idmap,
+int security_inode_get_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name);
-int security_inode_remove_acl(struct mnt_idmap *idmap,
+int security_inode_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *acl_name);
-void security_inode_post_remove_acl(struct mnt_idmap *idmap,
+void security_inode_post_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name);
void security_inode_post_setxattr(struct dentry *dentry, const char *name,
const void *value, size_t size, int flags);
int security_inode_getxattr(struct dentry *dentry, const char *name);
int security_inode_listxattr(struct dentry *dentry);
-int security_inode_removexattr(struct mnt_idmap *idmap,
+int security_inode_removexattr(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name);
void security_inode_post_removexattr(struct dentry *dentry, const char *name);
int security_inode_file_setattr(struct dentry *dentry,
@@ -454,8 +455,8 @@ int security_inode_file_setattr(struct dentry *dentry,
int security_inode_file_getattr(struct dentry *dentry,
struct file_kattr *fa);
int security_inode_need_killpriv(struct dentry *dentry);
-int security_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry);
-int security_inode_getsecurity(struct mnt_idmap *idmap,
+int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry);
+int security_inode_getsecurity(const struct mnt_idmap *idmap,
struct inode *inode, const char *name,
void **buffer, bool alloc);
int security_inode_setsecurity(struct inode *inode, const char *name, const void *value, size_t size, int flags);
@@ -676,6 +677,12 @@ static inline int security_ptrace_traceme(struct task_struct *parent)
return cap_ptrace_traceme(parent);
}
+static inline int security_mem_foll_force(const struct cred *subject,
+ bool opened_by_owner)
+{
+ return 0;
+}
+
static inline int security_capget(const struct task_struct *target,
kernel_cap_t *effective,
kernel_cap_t *inheritable,
@@ -910,7 +917,7 @@ static inline int security_inode_create(struct inode *dir,
}
static inline void
-security_inode_post_create_tmpfile(struct mnt_idmap *idmap, struct inode *inode)
+security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode)
{ }
static inline int security_inode_link(struct dentry *old_dentry,
@@ -979,7 +986,7 @@ static inline int security_inode_permission(struct inode *inode, int mask)
return 0;
}
-static inline int security_inode_setattr(struct mnt_idmap *idmap,
+static inline int security_inode_setattr(const struct mnt_idmap *idmap,
struct dentry *dentry,
struct iattr *attr)
{
@@ -987,7 +994,7 @@ static inline int security_inode_setattr(struct mnt_idmap *idmap,
}
static inline void
-security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
+security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry,
int ia_valid)
{ }
@@ -996,14 +1003,14 @@ static inline int security_inode_getattr(const struct path *path)
return 0;
}
-static inline int security_inode_setxattr(struct mnt_idmap *idmap,
+static inline int security_inode_setxattr(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name, const void *value,
size_t size, int flags)
{
return cap_inode_setxattr(dentry, name, value, size, flags);
}
-static inline int security_inode_set_acl(struct mnt_idmap *idmap,
+static inline int security_inode_set_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name,
struct posix_acl *kacl)
@@ -1016,21 +1023,21 @@ static inline void security_inode_post_set_acl(struct dentry *dentry,
struct posix_acl *kacl)
{ }
-static inline int security_inode_get_acl(struct mnt_idmap *idmap,
+static inline int security_inode_get_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name)
{
return 0;
}
-static inline int security_inode_remove_acl(struct mnt_idmap *idmap,
+static inline int security_inode_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name)
{
return 0;
}
-static inline void security_inode_post_remove_acl(struct mnt_idmap *idmap,
+static inline void security_inode_post_remove_acl(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *acl_name)
{ }
@@ -1050,7 +1057,7 @@ static inline int security_inode_listxattr(struct dentry *dentry)
return 0;
}
-static inline int security_inode_removexattr(struct mnt_idmap *idmap,
+static inline int security_inode_removexattr(const struct mnt_idmap *idmap,
struct dentry *dentry,
const char *name)
{
@@ -1078,13 +1085,13 @@ static inline int security_inode_need_killpriv(struct dentry *dentry)
return cap_inode_need_killpriv(dentry);
}
-static inline int security_inode_killpriv(struct mnt_idmap *idmap,
+static inline int security_inode_killpriv(const struct mnt_idmap *idmap,
struct dentry *dentry)
{
return cap_inode_killpriv(idmap, dentry);
}
-static inline int security_inode_getsecurity(struct mnt_idmap *idmap,
+static inline int security_inode_getsecurity(const struct mnt_idmap *idmap,
struct inode *inode,
const char *name, void **buffer,
bool alloc)
@@ -2085,7 +2092,7 @@ int security_path_mkdir(const struct path *dir, struct dentry *dentry, umode_t m
int security_path_rmdir(const struct path *dir, struct dentry *dentry);
int security_path_mknod(const struct path *dir, struct dentry *dentry, umode_t mode,
unsigned int dev);
-void security_path_post_mknod(struct mnt_idmap *idmap, struct dentry *dentry);
+void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry);
int security_path_truncate(const struct path *path);
int security_path_symlink(const struct path *dir, struct dentry *dentry,
const char *old_name);
@@ -2120,7 +2127,7 @@ static inline int security_path_mknod(const struct path *dir, struct dentry *den
return 0;
}
-static inline void security_path_post_mknod(struct mnt_idmap *idmap,
+static inline void security_path_post_mknod(const struct mnt_idmap *idmap,
struct dentry *dentry)
{ }
diff --git a/include/linux/splice.h b/include/linux/splice.h
index 9dec4861d09f..0e6c955dc6ff 100644
--- a/include/linux/splice.h
+++ b/include/linux/splice.h
@@ -79,8 +79,8 @@ ssize_t add_to_pipe(struct pipe_inode_info *pipe, struct pipe_buffer *buf);
ssize_t vfs_splice_read(struct file *in, loff_t *ppos,
struct pipe_inode_info *pipe, size_t len,
unsigned int flags);
-ssize_t splice_direct_to_actor(struct file *file, struct splice_desc *sd,
- splice_direct_actor *actor);
+ssize_t vfs_splice_to_actor(struct file *file, loff_t pos, size_t count,
+ splice_direct_actor *actor, void *private);
ssize_t do_splice(struct file *in, loff_t *off_in, struct file *out,
loff_t *off_out, size_t len, unsigned int flags);
ssize_t do_splice_direct(struct file *in, loff_t *ppos, struct file *out,
diff --git a/include/linux/uidgid.h b/include/linux/uidgid.h
index 2dc767e08f54..02403629b49f 100644
--- a/include/linux/uidgid.h
+++ b/include/linux/uidgid.h
@@ -130,9 +130,9 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid)
return from_kgid(ns, gid) != (gid_t) -1;
}
-u32 map_id_down(struct uid_gid_map *map, u32 id);
-u32 map_id_up(struct uid_gid_map *map, u32 id);
-u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count);
+u32 map_id_down(const struct uid_gid_map *map, u32 id);
+u32 map_id_up(const struct uid_gid_map *map, u32 id);
+u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count);
#else
@@ -182,17 +182,17 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid)
return gid_valid(gid);
}
-static inline u32 map_id_down(struct uid_gid_map *map, u32 id)
+static inline u32 map_id_down(const struct uid_gid_map *map, u32 id)
{
return id;
}
-static inline u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count)
+static inline u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count)
{
return id;
}
-static inline u32 map_id_up(struct uid_gid_map *map, u32 id)
+static inline u32 map_id_up(const struct uid_gid_map *map, u32 id)
{
return id;
}
diff --git a/include/linux/user_namespace.h b/include/linux/user_namespace.h
index e38d9e60569f..91232053775d 100644
--- a/include/linux/user_namespace.h
+++ b/include/linux/user_namespace.h
@@ -29,8 +29,8 @@ struct uid_gid_map { /* 64 bytes -- 1 cache line */
u32 nr_extents;
};
struct {
- struct uid_gid_extent *forward;
- struct uid_gid_extent *reverse;
+ struct uid_gid_extent *forward __counted_by_ptr(nr_extents);
+ struct uid_gid_extent *reverse __counted_by_ptr(nr_extents);
};
};
};
@@ -207,6 +207,13 @@ extern bool in_userns(const struct user_namespace *ancestor,
const struct user_namespace *child);
extern bool current_in_userns(const struct user_namespace *target_ns);
struct ns_common *ns_get_owner(struct ns_common *ns);
+
+#if IS_ENABLED(CONFIG_KUNIT)
+extern int uid_gid_map_insert_extent(struct uid_gid_map *map,
+ struct uid_gid_extent *extent);
+extern int uid_gid_map_sort(struct uid_gid_map *map);
+#endif /* CONFIG_KUNIT */
+
#else
static inline struct user_namespace *get_user_ns(struct user_namespace *ns)
diff --git a/include/linux/wait_bit.h b/include/linux/wait_bit.h
index 553d7b23e3ad..af077ed4caf6 100644
--- a/include/linux/wait_bit.h
+++ b/include/linux/wait_bit.h
@@ -433,6 +433,32 @@ do { \
})
/**
+ * wait_var_event_state - wait for a variable to be updated and notified
+ * @var: the address of variable being waited on
+ * @condition: the condition to wait for
+ * @state: the task state to sleep in, %TASK_UNINTERRUPTIBLE etc.
+ *
+ * Wait for a @condition to be true, only re-checking when a wake up is
+ * received for the given @var (an arbitrary kernel address which need
+ * not be directly related to the given condition, but usually is).
+ *
+ * Returns 0 if the condition became true, or %-ERESTARTSYS if a signal
+ * arrived which @state allows to interrupt.
+ *
+ * The condition should normally use smp_load_acquire() or a similarly
+ * ordered access to ensure that any changes to memory made before the
+ * condition became true will be visible after the wait completes.
+ */
+#define wait_var_event_state(var, condition, state) \
+({ \
+ int __ret = 0; \
+ might_sleep(); \
+ if (!(condition)) \
+ __ret = ___wait_var_event(var, condition, (state), 0, 0, schedule()); \
+ __ret; \
+})
+
+/**
* wait_var_event_any_lock - wait for a variable to be updated under a lock
* @var: the address of the variable being waited on
* @condition: condition to wait for
diff --git a/include/linux/xattr.h b/include/linux/xattr.h
index 54ac3cbc133f..4cc4257de084 100644
--- a/include/linux/xattr.h
+++ b/include/linux/xattr.h
@@ -47,7 +47,7 @@ struct xattr_handler {
struct inode *inode, const char *name, void *buffer,
size_t size);
int (*set)(const struct xattr_handler *,
- struct mnt_idmap *idmap, struct dentry *dentry,
+ const struct mnt_idmap *idmap, struct dentry *dentry,
struct inode *inode, const char *name, const void *buffer,
size_t size, int flags);
};
@@ -77,25 +77,25 @@ struct xattr {
};
ssize_t __vfs_getxattr(struct dentry *, struct inode *, const char *, void *, size_t);
-ssize_t vfs_getxattr(struct mnt_idmap *, struct dentry *, const char *,
+ssize_t vfs_getxattr(const struct mnt_idmap *, struct dentry *, const char *,
void *, size_t);
ssize_t vfs_listxattr(struct dentry *d, char *list, size_t size);
-int __vfs_setxattr(struct mnt_idmap *, struct dentry *, struct inode *,
+int __vfs_setxattr(const struct mnt_idmap *, struct dentry *, struct inode *,
const char *, const void *, size_t, int);
-int __vfs_setxattr_noperm(struct mnt_idmap *, struct dentry *,
+int __vfs_setxattr_noperm(const struct mnt_idmap *, struct dentry *,
const char *, const void *, size_t, int);
-int __vfs_setxattr_locked(struct mnt_idmap *, struct dentry *,
+int __vfs_setxattr_locked(const struct mnt_idmap *, struct dentry *,
const char *, const void *, size_t, int,
struct delegated_inode *);
-int vfs_setxattr(struct mnt_idmap *, struct dentry *, const char *,
+int vfs_setxattr(const struct mnt_idmap *, struct dentry *, const char *,
const void *, size_t, int);
-int __vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *);
-int __vfs_removexattr_locked(struct mnt_idmap *, struct dentry *,
+int __vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *);
+int __vfs_removexattr_locked(const struct mnt_idmap *, struct dentry *,
const char *, struct delegated_inode *);
-int vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *);
+int vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *);
ssize_t generic_listxattr(struct dentry *dentry, char *buffer, size_t buffer_size);
-int vfs_getxattr_alloc(struct mnt_idmap *idmap,
+int vfs_getxattr_alloc(const struct mnt_idmap *idmap,
struct dentry *dentry, const char *name,
char **xattr_value, size_t size, gfp_t flags);
diff --git a/include/trace/events/cachefiles.h b/include/trace/events/cachefiles.h
index e3101410e8b2..a19233e8ae78 100644
--- a/include/trace/events/cachefiles.h
+++ b/include/trace/events/cachefiles.h
@@ -52,6 +52,8 @@ enum cachefiles_coherency_trace {
cachefiles_coherency_check_ok,
cachefiles_coherency_check_type,
cachefiles_coherency_check_xattr,
+ cachefiles_coherency_discontiguous,
+ cachefiles_coherency_remove,
cachefiles_coherency_set_fail,
cachefiles_coherency_set_ok,
cachefiles_coherency_vol_check_cmp,
@@ -63,9 +65,11 @@ enum cachefiles_coherency_trace {
};
enum cachefiles_trunc_trace {
+ cachefiles_trunc_clear_padding,
cachefiles_trunc_dio_adjust,
cachefiles_trunc_expand_tmpfile,
cachefiles_trunc_shrink,
+ cachefiles_trunc_zap,
};
enum cachefiles_prepare_read_trace {
@@ -80,11 +84,14 @@ enum cachefiles_prepare_read_trace {
};
enum cachefiles_error_trace {
+ cachefiles_trace_alignment_error,
+ cachefiles_trace_create_nospace,
cachefiles_trace_fallocate_error,
cachefiles_trace_getxattr_error,
cachefiles_trace_link_error,
cachefiles_trace_lookup_error,
cachefiles_trace_mkdir_error,
+ cachefiles_trace_mkdir_nospace,
cachefiles_trace_notify_change_error,
cachefiles_trace_open_error,
cachefiles_trace_read_error,
@@ -97,6 +104,8 @@ enum cachefiles_error_trace {
cachefiles_trace_trunc_error,
cachefiles_trace_unlink_error,
cachefiles_trace_write_error,
+ cachefiles_trace_write_nospace,
+ cachefiles_trace_write_nospace_2,
};
#endif
@@ -136,6 +145,8 @@ enum cachefiles_error_trace {
EM(cachefiles_coherency_check_ok, "OK ") \
EM(cachefiles_coherency_check_type, "BAD type") \
EM(cachefiles_coherency_check_xattr, "BAD xatt") \
+ EM(cachefiles_coherency_discontiguous, "--- gap ") \
+ EM(cachefiles_coherency_remove, "REMOVE ") \
EM(cachefiles_coherency_set_fail, "SET fail") \
EM(cachefiles_coherency_set_ok, "SET ok ") \
EM(cachefiles_coherency_vol_check_cmp, "VOL BAD cmp ") \
@@ -146,9 +157,11 @@ enum cachefiles_error_trace {
E_(cachefiles_coherency_vol_set_ok, "VOL SET ok ")
#define cachefiles_trunc_traces \
+ EM(cachefiles_trunc_clear_padding, "CLRPAD") \
EM(cachefiles_trunc_dio_adjust, "DIOADJ") \
EM(cachefiles_trunc_expand_tmpfile, "EXPTMP") \
- E_(cachefiles_trunc_shrink, "SHRINK")
+ EM(cachefiles_trunc_shrink, "SHRINK") \
+ E_(cachefiles_trunc_zap, "ZAP ")
#define cachefiles_prepare_read_traces \
EM(cachefiles_trace_read_after_eof, "after-eof ") \
@@ -161,11 +174,14 @@ enum cachefiles_error_trace {
E_(cachefiles_trace_read_seek_nxio, "seek-enxio")
#define cachefiles_error_traces \
+ EM(cachefiles_trace_alignment_error, "align") \
+ EM(cachefiles_trace_create_nospace, "create-nospace") \
EM(cachefiles_trace_fallocate_error, "fallocate") \
EM(cachefiles_trace_getxattr_error, "getxattr") \
EM(cachefiles_trace_link_error, "link") \
EM(cachefiles_trace_lookup_error, "lookup") \
EM(cachefiles_trace_mkdir_error, "mkdir") \
+ EM(cachefiles_trace_mkdir_nospace, "mkdir-nospace") \
EM(cachefiles_trace_notify_change_error, "notify_change") \
EM(cachefiles_trace_open_error, "open") \
EM(cachefiles_trace_read_error, "read") \
@@ -177,7 +193,9 @@ enum cachefiles_error_trace {
EM(cachefiles_trace_tmpfile_error, "tmpfile") \
EM(cachefiles_trace_trunc_error, "trunc") \
EM(cachefiles_trace_unlink_error, "unlink") \
- E_(cachefiles_trace_write_error, "write")
+ EM(cachefiles_trace_write_error, "write") \
+ EM(cachefiles_trace_write_nospace, "write-nospace") \
+ E_(cachefiles_trace_write_nospace_2, "write-nospace-2")
/*
@@ -371,12 +389,12 @@ TRACE_EVENT(cachefiles_rename,
TRACE_EVENT(cachefiles_coherency,
TP_PROTO(struct cachefiles_object *obj,
- ino_t ino,
+ ino_t ino, uoff_t obj_size,
const void *disk_aux,
enum cachefiles_content content,
enum cachefiles_coherency_trace why),
- TP_ARGS(obj, ino, disk_aux, content, why),
+ TP_ARGS(obj, ino, obj_size, disk_aux, content, why),
/* Note that obj may be NULL */
TP_STRUCT__entry(
@@ -384,6 +402,7 @@ TRACE_EVENT(cachefiles_coherency,
__field(enum cachefiles_coherency_trace, why)
__field(enum cachefiles_content, content)
__field(u64, ino)
+ __field(u64, obj_size)
__field(u64, aux)
__field(u64, disk_aux)
),
@@ -398,6 +417,7 @@ TRACE_EVENT(cachefiles_coherency,
__entry->why = why;
__entry->content = content;
__entry->ino = ino;
+ __entry->obj_size = obj_size;
__entry->aux = be64_to_cpup((__be64 *)obj->cookie->inline_aux);
/* cachefiles_xattr::data is 2-byte aligned but not 8-byte aligned. */
@@ -412,10 +432,11 @@ TRACE_EVENT(cachefiles_coherency,
}
),
- TP_printk("o=%08x %s B=%llx c=%u aux=%llx dsk=%llx",
+ TP_printk("o=%08x %s B=%llx oz=%llx c=%u aux=%llx dsk=%llx",
__entry->obj,
__print_symbolic(__entry->why, cachefiles_coherency_traces),
__entry->ino,
+ __entry->obj_size,
__entry->content,
__entry->aux,
__entry->disk_aux)
@@ -449,7 +470,7 @@ TRACE_EVENT(cachefiles_vol_coherency,
TRACE_EVENT(cachefiles_prep_read,
TP_PROTO(struct cachefiles_object *obj,
- loff_t start,
+ uoff_t start,
size_t len,
unsigned short flags,
enum netfs_io_source source,
@@ -464,7 +485,7 @@ TRACE_EVENT(cachefiles_prep_read,
__field(enum netfs_io_source, source)
__field(enum cachefiles_prepare_read_trace, why)
__field(size_t, len)
- __field(loff_t, start)
+ __field(uoff_t, start)
__field(unsigned int, netfs_inode)
__field(unsigned int, cache_inode)
),
@@ -492,16 +513,16 @@ TRACE_EVENT(cachefiles_prep_read,
TRACE_EVENT(cachefiles_read,
TP_PROTO(struct cachefiles_object *obj,
struct inode *backer,
- loff_t start,
+ uoff_t start,
size_t len),
TP_ARGS(obj, backer, start, len),
TP_STRUCT__entry(
- __field(unsigned int, obj)
- __field(unsigned int, backer)
- __field(size_t, len)
- __field(loff_t, start)
+ __field(unsigned int, obj)
+ __field(unsigned int, backer)
+ __field(size_t, len)
+ __field(uoff_t, start)
),
TP_fast_assign(
@@ -521,16 +542,16 @@ TRACE_EVENT(cachefiles_read,
TRACE_EVENT(cachefiles_write,
TP_PROTO(struct cachefiles_object *obj,
struct inode *backer,
- loff_t start,
+ uoff_t start,
size_t len),
TP_ARGS(obj, backer, start, len),
TP_STRUCT__entry(
- __field(unsigned int, obj)
- __field(unsigned int, backer)
- __field(size_t, len)
- __field(loff_t, start)
+ __field(unsigned int, obj)
+ __field(unsigned int, backer)
+ __field(size_t, len)
+ __field(uoff_t, start)
),
TP_fast_assign(
@@ -549,7 +570,7 @@ TRACE_EVENT(cachefiles_write,
TRACE_EVENT(cachefiles_trunc,
TP_PROTO(struct cachefiles_object *obj, struct inode *backer,
- loff_t from, loff_t to, enum cachefiles_trunc_trace why),
+ uoff_t from, uoff_t to, enum cachefiles_trunc_trace why),
TP_ARGS(obj, backer, from, to, why),
@@ -557,8 +578,8 @@ TRACE_EVENT(cachefiles_trunc,
__field(unsigned int, obj)
__field(unsigned int, backer)
__field(enum cachefiles_trunc_trace, why)
- __field(loff_t, from)
- __field(loff_t, to)
+ __field(uoff_t, from)
+ __field(uoff_t, to)
),
TP_fast_assign(
@@ -694,6 +715,26 @@ TRACE_EVENT(cachefiles_io_error,
__entry->error)
);
+TRACE_EVENT(cachefiles_no_space,
+ TP_PROTO(struct cachefiles_object *obj, enum cachefiles_error_trace trace),
+
+ TP_ARGS(obj, trace),
+
+ TP_STRUCT__entry(
+ __field(unsigned int, obj)
+ __field(enum cachefiles_error_trace, trace)
+ ),
+
+ TP_fast_assign(
+ __entry->obj = obj ? obj->debug_id : 0;
+ __entry->trace = trace;
+ ),
+
+ TP_printk("o=%08x %s",
+ __entry->obj,
+ __print_symbolic(__entry->trace, cachefiles_error_traces))
+ );
+
#endif /* _TRACE_CACHEFILES_H */
/* This part must be outside protection */
diff --git a/include/trace/events/fscache.h b/include/trace/events/fscache.h
index f1a73aa83fbb..8735d428ebd9 100644
--- a/include/trace/events/fscache.h
+++ b/include/trace/events/fscache.h
@@ -460,13 +460,13 @@ TRACE_EVENT(fscache_relinquish,
);
TRACE_EVENT(fscache_invalidate,
- TP_PROTO(struct fscache_cookie *cookie, loff_t new_size),
+ TP_PROTO(struct fscache_cookie *cookie, uoff_t new_size),
TP_ARGS(cookie, new_size),
TP_STRUCT__entry(
__field(unsigned int, cookie )
- __field(loff_t, new_size )
+ __field(uoff_t, new_size )
),
TP_fast_assign(
@@ -479,14 +479,14 @@ TRACE_EVENT(fscache_invalidate,
);
TRACE_EVENT(fscache_resize,
- TP_PROTO(struct fscache_cookie *cookie, loff_t new_size),
+ TP_PROTO(struct fscache_cookie *cookie, uoff_t new_size),
TP_ARGS(cookie, new_size),
TP_STRUCT__entry(
__field(unsigned int, cookie )
- __field(loff_t, old_size )
- __field(loff_t, new_size )
+ __field(uoff_t, old_size )
+ __field(uoff_t, new_size )
),
TP_fast_assign(
diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h
index 3fec3e8f91c8..bf1e1f185b05 100644
--- a/include/trace/events/netfs.h
+++ b/include/trace/events/netfs.h
@@ -30,8 +30,7 @@
EM(netfs_write_trace_dio_write, "DIO-WRITE") \
EM(netfs_write_trace_unbuffered_write, "UNB-WRITE") \
EM(netfs_write_trace_writeback, "WRITEBACK") \
- EM(netfs_write_trace_writeback_single, "WB-SINGLE") \
- E_(netfs_write_trace_writethrough, "WRITETHRU")
+ E_(netfs_write_trace_writeback_single, "WB-SINGLE")
#define netfs_rreq_origins \
EM(NETFS_READAHEAD, "RA") \
@@ -43,13 +42,17 @@
EM(NETFS_DIO_READ, "DR") \
EM(NETFS_WRITEBACK, "WB") \
EM(NETFS_WRITEBACK_SINGLE, "W1") \
- EM(NETFS_WRITETHROUGH, "WT") \
EM(NETFS_UNBUFFERED_WRITE, "UW") \
EM(NETFS_DIO_WRITE, "DW") \
E_(NETFS_PGPRIV2_COPY_TO_CACHE, "2C")
#define netfs_rreq_traces \
+ EM(netfs_rreq_trace_all_queued, "ALL-Q ") \
EM(netfs_rreq_trace_assess, "ASSESS ") \
+ EM(netfs_rreq_trace_cache_cancelled, "CA-CNCL") \
+ EM(netfs_rreq_trace_cache_failed, "CA-FAIL") \
+ EM(netfs_rreq_trace_cache_fail_collect, "CA-F-CO") \
+ EM(netfs_rreq_trace_cache_no_space, "CA-NOSP") \
EM(netfs_rreq_trace_collect, "COLLECT") \
EM(netfs_rreq_trace_complete, "COMPLET") \
EM(netfs_rreq_trace_copy, "COPY ") \
@@ -58,11 +61,14 @@
EM(netfs_rreq_trace_end_copy_to_cache, "END-C2C") \
EM(netfs_rreq_trace_free, "FREE ") \
EM(netfs_rreq_trace_intr, "INTR ") \
+ EM(netfs_rreq_trace_inval_cache, "INVL-CA") \
EM(netfs_rreq_trace_ki_complete, "KI-CMPL") \
EM(netfs_rreq_trace_ra_put_ref, "RA-PUT ") \
EM(netfs_rreq_trace_recollect, "RECLLCT") \
EM(netfs_rreq_trace_redirty, "REDIRTY") \
EM(netfs_rreq_trace_resubmit, "RESUBMT") \
+ EM(netfs_rreq_trace_retry_begin, "RETRY-BEGIN") \
+ EM(netfs_rreq_trace_retry_end, "RETRY-END") \
EM(netfs_rreq_trace_set_abandon, "S-ABNDN") \
EM(netfs_rreq_trace_set_pause, "PAUSE ") \
EM(netfs_rreq_trace_unlock, "UNLOCK ") \
@@ -94,8 +100,10 @@
EM(netfs_sreq_trace_abandoned, "ABNDN") \
EM(netfs_sreq_trace_add_donations, "+DON ") \
EM(netfs_sreq_trace_added, "ADD ") \
+ EM(netfs_sreq_trace_cache_nofile, "CA-!F") \
EM(netfs_sreq_trace_cache_nowrite, "CA-NW") \
EM(netfs_sreq_trace_cache_prepare, "CA-PR") \
+ EM(netfs_sreq_trace_cache_waitfail, "CA-!W") \
EM(netfs_sreq_trace_cache_write, "CA-WR") \
EM(netfs_sreq_trace_cancel, "CANCL") \
EM(netfs_sreq_trace_clear, "CLEAR") \
@@ -134,12 +142,12 @@
#define netfs_failures \
EM(netfs_fail_check_write_begin, "check-write-begin") \
- EM(netfs_fail_copy_to_cache, "copy-to-cache") \
EM(netfs_fail_dio_read_short, "dio-read-short") \
EM(netfs_fail_dio_read_zero, "dio-read-zero") \
EM(netfs_fail_read, "read") \
EM(netfs_fail_short_read, "short-read") \
EM(netfs_fail_prepare_write, "prep-write") \
+ EM(netfs_fail_upload, "upload") \
E_(netfs_fail_write, "write")
#define netfs_rreq_ref_traces \
@@ -194,11 +202,11 @@
EM(netfs_folio_trace_alloc_buffer, "alloc-buf") \
EM(netfs_folio_trace_cancel_copy, "cancel-copy") \
EM(netfs_folio_trace_cancel_store, "cancel-store") \
- EM(netfs_folio_trace_clear, "clear") \
- EM(netfs_folio_trace_clear_cc, "clear-cc") \
- EM(netfs_folio_trace_clear_g, "clear-g") \
- EM(netfs_folio_trace_clear_s, "clear-s") \
EM(netfs_folio_trace_end_copy, "end-copy") \
+ EM(netfs_folio_trace_endwb, "endwb") \
+ EM(netfs_folio_trace_endwb_cc, "endwb-cc") \
+ EM(netfs_folio_trace_endwb_g, "endwb-g") \
+ EM(netfs_folio_trace_endwb_s, "endwb-s") \
EM(netfs_folio_trace_filled_gaps, "filled-gaps") \
EM(netfs_folio_trace_invalidate_all, "inval-all") \
EM(netfs_folio_trace_invalidate_front, "inval-front") \
@@ -223,9 +231,7 @@
EM(netfs_folio_trace_sched_copy, "sched-copy") \
EM(netfs_folio_trace_store, "store") \
EM(netfs_folio_trace_store_copy, "store-copy") \
- EM(netfs_folio_trace_store_plus, "store+") \
- EM(netfs_folio_trace_wthru, "wthru") \
- E_(netfs_folio_trace_wthru_plus, "wthru+")
+ E_(netfs_folio_trace_store_plus, "store+")
#define netfs_collect_contig_traces \
EM(netfs_contig_trace_collect, "Collect") \
@@ -301,7 +307,7 @@ netfs_folioq_traces;
TRACE_EVENT(netfs_read,
TP_PROTO(struct netfs_io_request *rreq,
- loff_t start, size_t len,
+ uoff_t start, size_t len,
enum netfs_read_trace what),
TP_ARGS(rreq, start, len, what),
@@ -309,8 +315,9 @@ TRACE_EVENT(netfs_read,
TP_STRUCT__entry(
__field(unsigned int, rreq)
__field(unsigned int, cookie)
- __field(loff_t, i_size)
- __field(loff_t, start)
+ __field(unsigned int, object)
+ __field(uoff_t, i_size)
+ __field(uoff_t, start)
__field(size_t, len)
__field(enum netfs_read_trace, what)
__field(u64, netfs_inode)
@@ -318,7 +325,8 @@ TRACE_EVENT(netfs_read,
TP_fast_assign(
__entry->rreq = rreq->debug_id;
- __entry->cookie = rreq->cache_resources.debug_id;
+ __entry->cookie = rreq->cache_resources.cookie_id;
+ __entry->object = rreq->cache_resources.object_id;
__entry->i_size = rreq->i_size;
__entry->start = start;
__entry->len = len;
@@ -326,10 +334,10 @@ TRACE_EVENT(netfs_read,
__entry->netfs_inode = rreq->inode->i_ino;
),
- TP_printk("R=%08x %s c=%08x ni=%llx s=%llx l=%zx sz=%llx",
+ TP_printk("R=%08x %s c=%08x o=%08x ni=%llx s=%llx l=%zx sz=%llx",
__entry->rreq,
__print_symbolic(__entry->what, netfs_read_traces),
- __entry->cookie,
+ __entry->cookie, __entry->object,
__entry->netfs_inode,
__entry->start, __entry->len, __entry->i_size)
);
@@ -377,7 +385,7 @@ TRACE_EVENT(netfs_sreq,
__field(u8, slot)
__field(size_t, len)
__field(size_t, transferred)
- __field(loff_t, start)
+ __field(uoff_t, start)
),
TP_fast_assign(
@@ -418,7 +426,7 @@ TRACE_EVENT(netfs_failure,
__field(enum netfs_failure, what)
__field(size_t, len)
__field(size_t, transferred)
- __field(loff_t, start)
+ __field(uoff_t, start)
),
TP_fast_assign(
@@ -501,6 +509,7 @@ TRACE_EVENT(netfs_folio,
TP_STRUCT__entry(
__field(u64, ino)
__field(pgoff_t, index)
+ __field(unsigned long, pfn)
__field(unsigned int, nr)
__field(enum netfs_folio_trace, why)
),
@@ -511,9 +520,11 @@ TRACE_EVENT(netfs_folio,
__entry->why = why;
__entry->index = folio->index;
__entry->nr = folio_nr_pages(folio);
+ __entry->pfn = folio_pfn(folio);
),
- TP_printk("i=%05llx ix=%05lx-%05lx %s",
+ TP_printk("p=%lx i=%05llx ix=%05lx-%05lx %s",
+ __entry->pfn,
__entry->ino, __entry->index, __entry->index + __entry->nr - 1,
__print_symbolic(__entry->why, netfs_folio_traces))
);
@@ -524,10 +535,10 @@ TRACE_EVENT(netfs_write_iter,
TP_ARGS(iocb, from),
TP_STRUCT__entry(
- __field(unsigned long long, start)
- __field(size_t, len)
- __field(unsigned int, flags)
- __field(unsigned int, ino)
+ __field(uoff_t, start)
+ __field(size_t, len)
+ __field(unsigned int, flags)
+ __field(unsigned int, ino)
),
TP_fast_assign(
@@ -550,27 +561,27 @@ TRACE_EVENT(netfs_write,
TP_STRUCT__entry(
__field(unsigned int, wreq)
__field(unsigned int, cookie)
+ __field(unsigned int, object)
__field(unsigned int, ino)
__field(enum netfs_write_trace, what)
- __field(unsigned long long, start)
- __field(unsigned long long, len)
+ __field(uoff_t, start)
+ __field(uoff_t, len)
),
TP_fast_assign(
- struct netfs_inode *__ctx = netfs_inode(wreq->inode);
- struct fscache_cookie *__cookie = netfs_i_cookie(__ctx);
__entry->wreq = wreq->debug_id;
- __entry->cookie = __cookie ? __cookie->debug_id : 0;
+ __entry->cookie = wreq->cache_resources.cookie_id;
+ __entry->object = wreq->cache_resources.object_id;
__entry->ino = wreq->inode->i_ino;
__entry->what = what;
__entry->start = wreq->start;
__entry->len = wreq->len;
),
- TP_printk("R=%08x %s c=%08x i=%x by=%llx-%llx",
+ TP_printk("R=%08x %s c=%08x o=%08x i=%x by=%llx-%llx",
__entry->wreq,
__print_symbolic(__entry->what, netfs_write_traces),
- __entry->cookie,
+ __entry->cookie, __entry->object,
__entry->ino,
__entry->start, __entry->start + __entry->len - 1)
);
@@ -582,25 +593,26 @@ TRACE_EVENT(netfs_copy2cache,
TP_ARGS(rreq, creq),
TP_STRUCT__entry(
- __field(unsigned int, rreq)
- __field(unsigned int, creq)
- __field(unsigned int, cookie)
- __field(unsigned int, ino)
+ __field(unsigned int, rreq)
+ __field(unsigned int, creq)
+ __field(unsigned int, cookie)
+ __field(unsigned int, object)
+ __field(unsigned int, ino)
),
TP_fast_assign(
- struct netfs_inode *__ctx = netfs_inode(rreq->inode);
- struct fscache_cookie *__cookie = netfs_i_cookie(__ctx);
__entry->rreq = rreq->debug_id;
__entry->creq = creq->debug_id;
- __entry->cookie = __cookie ? __cookie->debug_id : 0;
+ __entry->cookie = rreq->cache_resources.cookie_id;
+ __entry->object = rreq->cache_resources.object_id;
__entry->ino = rreq->inode->i_ino;
),
- TP_printk("R=%08x CR=%08x c=%08x i=%x ",
+ TP_printk("R=%08x CR=%08x c=%08x o=%08x i=%x ",
__entry->rreq,
__entry->creq,
__entry->cookie,
+ __entry->object,
__entry->ino)
);
@@ -610,10 +622,10 @@ TRACE_EVENT(netfs_collect,
TP_ARGS(wreq),
TP_STRUCT__entry(
- __field(unsigned int, wreq)
- __field(unsigned int, len)
- __field(unsigned long long, transferred)
- __field(unsigned long long, start)
+ __field(unsigned int, wreq)
+ __field(unsigned int, len)
+ __field(uoff_t, transferred)
+ __field(uoff_t, start)
),
TP_fast_assign(
@@ -636,12 +648,12 @@ TRACE_EVENT(netfs_collect_sreq,
TP_ARGS(wreq, subreq),
TP_STRUCT__entry(
- __field(unsigned int, wreq)
- __field(unsigned int, subreq)
- __field(unsigned int, stream)
- __field(unsigned int, len)
- __field(unsigned int, transferred)
- __field(unsigned long long, start)
+ __field(unsigned int, wreq)
+ __field(unsigned int, subreq)
+ __field(unsigned int, stream)
+ __field(unsigned int, len)
+ __field(unsigned int, transferred)
+ __field(uoff_t, start)
),
TP_fast_assign(
@@ -660,37 +672,30 @@ TRACE_EVENT(netfs_collect_sreq,
TRACE_EVENT(netfs_collect_folio,
TP_PROTO(const struct netfs_io_request *wreq,
- const struct folio *folio,
- unsigned long long fend,
- unsigned long long collected_to),
+ const struct folio *folio),
- TP_ARGS(wreq, folio, fend, collected_to),
+ TP_ARGS(wreq, folio),
TP_STRUCT__entry(
__field(unsigned int, wreq)
__field(unsigned long, index)
- __field(unsigned long long, fend)
- __field(unsigned long long, cleaned_to)
- __field(unsigned long long, collected_to)
+ __field(unsigned int, nr)
),
TP_fast_assign(
__entry->wreq = wreq->debug_id;
__entry->index = folio->index;
- __entry->fend = fend;
- __entry->cleaned_to = wreq->cleaned_to;
- __entry->collected_to = collected_to;
+ __entry->nr = folio_nr_pages(folio);
),
- TP_printk("R=%08x ix=%05lx r=%llx-%llx t=%llx/%llx",
+ TP_printk("R=%08x ix=%05lx-%05lx",
__entry->wreq, __entry->index,
- (unsigned long long)__entry->index * PAGE_SIZE, __entry->fend,
- __entry->cleaned_to, __entry->collected_to)
+ __entry->index + __entry->nr - 1)
);
TRACE_EVENT(netfs_collect_state,
TP_PROTO(const struct netfs_io_request *wreq,
- unsigned long long collected_to,
+ uoff_t collected_to,
unsigned int notes),
TP_ARGS(wreq, collected_to, notes),
@@ -698,8 +703,8 @@ TRACE_EVENT(netfs_collect_state,
TP_STRUCT__entry(
__field(unsigned int, wreq)
__field(unsigned int, notes)
- __field(unsigned long long, collected_to)
- __field(unsigned long long, cleaned_to)
+ __field(uoff_t, collected_to)
+ __field(uoff_t, cleaned_to)
),
TP_fast_assign(
@@ -718,7 +723,7 @@ TRACE_EVENT(netfs_collect_state,
TRACE_EVENT(netfs_collect_gap,
TP_PROTO(const struct netfs_io_request *wreq,
const struct netfs_io_stream *stream,
- unsigned long long jump_to, char type),
+ uoff_t jump_to, char type),
TP_ARGS(wreq, stream, jump_to, type),
@@ -726,8 +731,8 @@ TRACE_EVENT(netfs_collect_gap,
__field(unsigned int, wreq)
__field(unsigned char, stream)
__field(unsigned char, type)
- __field(unsigned long long, from)
- __field(unsigned long long, to)
+ __field(uoff_t, from)
+ __field(uoff_t, to)
),
TP_fast_assign(
@@ -752,8 +757,8 @@ TRACE_EVENT(netfs_collect_stream,
TP_STRUCT__entry(
__field(unsigned int, wreq)
__field(unsigned char, stream)
- __field(unsigned long long, collected_to)
- __field(unsigned long long, issued_to)
+ __field(uoff_t, collected_to)
+ __field(uoff_t, issued_to)
),
TP_fast_assign(
diff --git a/include/uapi/linux/close_range.h b/include/uapi/linux/close_range.h
index 2d804281554c..7da9ed95258a 100644
--- a/include/uapi/linux/close_range.h
+++ b/include/uapi/linux/close_range.h
@@ -2,11 +2,34 @@
#ifndef _UAPI_LINUX_CLOSE_RANGE_H
#define _UAPI_LINUX_CLOSE_RANGE_H
-/* Unshare the file descriptor table before closing file descriptors. */
-#define CLOSE_RANGE_UNSHARE (1U << 1)
+/*
+ * A macro of one of these names defined before this header is parsed, by
+ * a libc or by a program's own fallback, would replace the enumerator.
+ */
+#undef CLOSE_RANGE_UNSHARE
+#undef CLOSE_RANGE_CLOEXEC
+#undef CLOSE_RANGE_EXCEPT
+#undef CLOSE_RANGE_CLOEXEC_ONLY
-/* Set the FD_CLOEXEC bit instead of closing the file descriptor. */
-#define CLOSE_RANGE_CLOEXEC (1U << 2)
+enum close_range_flags {
+ /* Unshare the file descriptor table before closing file descriptors. */
+ CLOSE_RANGE_UNSHARE = (1U << 1),
+
+ /* Set the FD_CLOEXEC bit instead of closing the file descriptor. */
+ CLOSE_RANGE_CLOEXEC = (1U << 2),
+
+ /* Act on every file descriptor outside of the given range instead. */
+ CLOSE_RANGE_EXCEPT = (1U << 3),
+
+ /* Only close file descriptors that have the FD_CLOEXEC bit set. */
+ CLOSE_RANGE_CLOEXEC_ONLY = (1U << 4),
+};
+
+/* Keep #ifdef working and let glibc skip its own definitions. */
+#define CLOSE_RANGE_UNSHARE CLOSE_RANGE_UNSHARE
+#define CLOSE_RANGE_CLOEXEC CLOSE_RANGE_CLOEXEC
+#define CLOSE_RANGE_EXCEPT CLOSE_RANGE_EXCEPT
+#define CLOSE_RANGE_CLOEXEC_ONLY CLOSE_RANGE_CLOEXEC_ONLY
#endif /* _UAPI_LINUX_CLOSE_RANGE_H */
diff --git a/include/uapi/linux/coredump.h b/include/uapi/linux/coredump.h
index dc3789b78af0..6d0c53b534ea 100644
--- a/include/uapi/linux/coredump.h
+++ b/include/uapi/linux/coredump.h
@@ -11,12 +11,53 @@
* @COREDUMP_USERSPACE: userspace writes coredump
* @COREDUMP_REJECT: don't generate coredump
* @COREDUMP_WAIT: wait for coredump server
+ * @COREDUMP_RECORDS: send the coredump as a sequence of records instead of
+ * as a plain byte stream, see struct coredump_record_header;
+ * requires COREDUMP_KERNEL
+ * @COREDUMP_SPARSE: describe the holes in the coredump as zero records
+ * instead of transferring them; requires COREDUMP_RECORDS
+ * @COREDUMP_MEMORY_TYPES: dump the memory types in
+ * coredump_ack->memory_types instead of the ones
+ * the task selected; requires COREDUMP_KERNEL
*/
enum {
COREDUMP_KERNEL = (1ULL << 0),
COREDUMP_USERSPACE = (1ULL << 1),
COREDUMP_REJECT = (1ULL << 2),
COREDUMP_WAIT = (1ULL << 3),
+ COREDUMP_RECORDS = (1ULL << 4),
+ COREDUMP_SPARSE = (1ULL << 5),
+ COREDUMP_MEMORY_TYPES = (1ULL << 6),
+};
+
+/**
+ * coredump memory types
+ * @COREDUMP_MEMORY_ANON_PRIVATE: anonymous private memory
+ * @COREDUMP_MEMORY_ANON_SHARED: anonymous shared memory
+ * @COREDUMP_MEMORY_FILE_PRIVATE: file-backed private memory
+ * @COREDUMP_MEMORY_FILE_SHARED: file-backed shared memory
+ * @COREDUMP_MEMORY_ELF_HEADERS: the first page of a file-backed private
+ * mapping that starts an ELF file
+ * @COREDUMP_MEMORY_HUGETLB_PRIVATE: hugetlb private memory
+ * @COREDUMP_MEMORY_HUGETLB_SHARED: hugetlb shared memory
+ * @COREDUMP_MEMORY_DAX_PRIVATE: DAX private memory
+ * @COREDUMP_MEMORY_DAX_SHARED: DAX shared memory
+ *
+ * A bitmask of memory types a coredump may request to be included. New
+ * memory type bits must ensure that they do not steal memory from an
+ * existing one so a coredump server will continue to get the same
+ * coredumps even if a new bit is introduced.
+ */
+enum {
+ COREDUMP_MEMORY_ANON_PRIVATE = (1ULL << 0),
+ COREDUMP_MEMORY_ANON_SHARED = (1ULL << 1),
+ COREDUMP_MEMORY_FILE_PRIVATE = (1ULL << 2),
+ COREDUMP_MEMORY_FILE_SHARED = (1ULL << 3),
+ COREDUMP_MEMORY_ELF_HEADERS = (1ULL << 4),
+ COREDUMP_MEMORY_HUGETLB_PRIVATE = (1ULL << 5),
+ COREDUMP_MEMORY_HUGETLB_SHARED = (1ULL << 6),
+ COREDUMP_MEMORY_DAX_PRIVATE = (1ULL << 7),
+ COREDUMP_MEMORY_DAX_SHARED = (1ULL << 8),
};
/**
@@ -24,17 +65,19 @@ enum {
* @size: size of struct coredump_req
* @size_ack: known size of struct coredump_ack on this kernel
* @mask: supported features
+ * @memory_types: the memory types the task selected
+ * @memory_types_mask: the memory types this kernel knows
*
* When a coredump happens the kernel will connect to the coredump
* socket and send a coredump request to the coredump server. The @size
* member is set to the size of struct coredump_req and provides a hint
* to userspace how much data can be read. Userspace may use MSG_PEEK to
* peek the size of struct coredump_req and then choose to consume it in
- * one go. Userspace may also simply read a COREDUMP_ACK_SIZE_VER0
+ * one go. Userspace may also simply read a COREDUMP_REQ_SIZE_VER0
* request. If the size the kernel sends is larger userspace simply
* discards any remaining data.
*
- * The coredump_req->mask member is set to the currently know features.
+ * The coredump_req->mask member is set to the currently known features.
* Userspace may only set coredump_ack->mask to the bits raised by the
* kernel in coredump_req->mask.
*
@@ -42,15 +85,27 @@ enum {
* struct coredump_ack the kernel knows. Userspace may only send up to
* coredump_req->size_ack bytes to the kernel and must set
* coredump_ack->size accordingly.
+ *
+ * @memory_types is set to the default memory types that are included in
+ * the coredump. This can be overridden by raising bits in
+ * coredump_ack->memory_types.
+ *
+ * @memory_types_mask contains a bitmask of all memory types the kernel
+ * knows about. A coredump server may only raise bits in
+ * coredump_ack->memory_types that are raised in
+ * coredump_req->memory_types_mask.
*/
struct coredump_req {
__u32 size;
__u32 size_ack;
__u64 mask;
+ __u64 memory_types;
+ __u64 memory_types_mask;
};
enum {
COREDUMP_REQ_SIZE_VER0 = 16U, /* size of first published struct */
+ COREDUMP_REQ_SIZE_VER1 = 32U, /* memory_types and memory_types_mask added */
};
/**
@@ -58,6 +113,8 @@ enum {
* @size: size of the struct
* @spare: unused
* @mask: features kernel is supposed to use
+ * @memory_types: memory types to dump, only with COREDUMP_MEMORY_TYPES
+ * in @mask
*
* The @size member must be set to the size of struct coredump_ack. It
* may never exceed what the kernel returned in coredump_req->size_ack
@@ -67,15 +124,30 @@ enum {
* The @mask member must be set to the features the coredump server
* wants the kernel to use. Only bits the kernel returned in
* coredump_req->mask may be set.
+ *
+ * If COREDUMP_MEMORY_TYPES is raised in @mask the kernel dumps the
+ * memory types set in the @memory_types mask. Zero is valid and dumps
+ * no memory apart from the mappings that are always dumped.
+ *
+ * Note that memory a task excluded via MADV_DONTDUMP is always left
+ * out. A coredump server wanting to add or drop memory types instead of
+ * outright replacing it should simply copy coredump_req->memory_types
+ * and then mask off or raise types as needed.
+ *
+ * Note that @memory_types must be zero if COREDUMP_MEMORY_TYPES isn't
+ * raised. COREDUMP_MEMORY_TYPES requires COREDUMP_KERNEL and an ack of
+ * at least COREDUMP_ACK_SIZE_VER1 bytes.
*/
struct coredump_ack {
__u32 size;
__u32 spare;
__u64 mask;
+ __u64 memory_types;
};
enum {
COREDUMP_ACK_SIZE_VER0 = 16U, /* size of first published struct */
+ COREDUMP_ACK_SIZE_VER1 = 24U, /* memory_types added */
};
/**
@@ -83,11 +155,12 @@ enum {
*
* The kernel will place a single byte on the coredump socket. The
* markers notify userspace whether the coredump ack succeeded or
- * failed.
+ * failed. After any marker other than COREDUMP_MARK_REQACK the kernel
+ * closes the connection and no coredump is generated.
*
* @COREDUMP_MARK_MINSIZE: the provided coredump_ack size was too small
* @COREDUMP_MARK_MAXSIZE: the provided coredump_ack size was too big
- * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask was invalid
+ * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask or memory types were invalid
* @COREDUMP_MARK_CONFLICTING: the provided coredump_ack mask has conflicting options
* @COREDUMP_MARK_REQACK: the coredump request and ack was successful
* @__COREDUMP_MARK_MAX: the maximum coredump mark value
@@ -101,4 +174,72 @@ enum coredump_mark {
__COREDUMP_MARK_MAX = (1U << 31),
};
+/**
+ * enum coredump_record_type - Type of a coredump record
+ *
+ * @COREDUMP_RECORD_DATA: the header is followed by ->len bytes of data
+ * @COREDUMP_RECORD_END: the coredump ends here, the header is not followed
+ * by any data and no further record is sent
+ * @COREDUMP_RECORD_ZERO: the header stands for ->len zero bytes and is not
+ * followed by any data
+ * @__COREDUMP_RECORD_TYPE_MAX: the maximum coredump record type value
+ */
+enum coredump_record_type {
+ COREDUMP_RECORD_DATA = 0U,
+ COREDUMP_RECORD_END = 1U,
+ COREDUMP_RECORD_ZERO = 2U,
+ __COREDUMP_RECORD_TYPE_MAX = (1U << 31),
+};
+
+/**
+ * struct coredump_record_header - header of a coredump record
+ * @size: size of struct coredump_record_header
+ * @type: one of enum coredump_record_type
+ * @flags: modifiers for this record
+ * @offset: offset in the coredump this record starts at
+ * @len: number of coredump bytes this record accounts for
+ *
+ * If the coredump server raises COREDUMP_RECORDS in coredump_ack->mask
+ * the kernel doesn't send the coredump as a plain byte stream. It sends
+ * a sequence of records instead. A COREDUMP_RECORD_DATA record is
+ * followed by @len bytes of actual coredump data. A
+ * COREDUMP_RECORD_ZERO record is followed by nothing and stands for
+ * @len zero bytes. A server that didn't raise COREDUMP_SPARSE never
+ * sees a zero record. Records arrive in order and leave no gaps. So
+ * @offset is the sum of the @len of all records before it.
+ *
+ * The last record is a COREDUMP_RECORD_END record. It is followed by
+ * nothing. Its @len is zero. Its @offset is the size of the coredump.
+ * The kernel only sends it once it has written the whole coredump. A
+ * server that hits end-of-file without having seen an end record must
+ * treat the coredump as incomplete.
+ *
+ * The @size member is set to the size of struct coredump_record_header
+ * the kernel knows and lets the header grow later. It comes first so it
+ * can be peeked. Userspace must consume @size bytes and discard
+ * anything beyond what it knows. It must refuse a @size smaller than
+ * COREDUMP_RECORD_HEADER_SIZE_VER0. @size covers the header alone.
+ * @offset and @len count coredump bytes.
+ *
+ * The @flags member carries modifiers that change how the record is to
+ * be interpreted. No flag is defined yet. Userspace must refuse a
+ * record carrying a flag or a type it doesn't know. Every new record
+ * type is raised in coredump_req->mask as a feature of its own. A
+ * server only ever sees the types it asked for.
+ *
+ * COREDUMP_RECORDS must be combined with COREDUMP_KERNEL, and
+ * COREDUMP_SPARSE with COREDUMP_RECORDS.
+ */
+struct coredump_record_header {
+ __u32 size;
+ __u32 type;
+ __u64 flags;
+ __u64 offset;
+ __u64 len;
+};
+
+enum {
+ COREDUMP_RECORD_HEADER_SIZE_VER0 = 32U, /* size of first published struct */
+};
+
#endif /* _UAPI_LINUX_COREDUMP_H */
diff --git a/include/uapi/linux/fs.h b/include/uapi/linux/fs.h
index 34c6f219462a..a46c33692aa2 100644
--- a/include/uapi/linux/fs.h
+++ b/include/uapi/linux/fs.h
@@ -88,7 +88,7 @@ struct fstrim_range {
* We include a length field because some filesystems (vfat) have an identifier
* that we do want to expose as a UUID, but doesn't have the standard length.
*
- * We use a fixed size buffer beacuse this interface will, by fiat, never
+ * We use a fixed size buffer because this interface will, by fiat, never
* support "UUIDs" longer than 16 bytes; we don't want to force all downstream
* users to have to deal with that.
*/