diff options
| author | Mark Brown <broonie@kernel.org> | 2026-09-30 11:58:30 +0100 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-09-30 11:58:30 +0100 |
| commit | 5eb3fdca6ca7667a73d52adfd047d94e8b8b105e (patch) | |
| tree | c935b3fa5b47c9d70e0e66c7cc7830e74988b13f /include | |
| parent | f60deca1c3741ab8fcf318a7bb82a3d026ea7c49 (diff) | |
| parent | 84086827932b58e7645d93d970bbc566c4ee408b (diff) | |
| download | linux-next-5eb3fdca6ca7667a73d52adfd047d94e8b8b105e.tar.gz linux-next-5eb3fdca6ca7667a73d52adfd047d94e8b8b105e.zip | |
Merge branch 'vfs.all' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git
# Conflicts:
# fs/smb/server/smb2pdu.c
# fs/smb/server/vfs.c
# fs/smb/server/vfs.h
Diffstat (limited to 'include')
40 files changed, 907 insertions, 484 deletions
diff --git a/include/linux/binfmts.h b/include/linux/binfmts.h index f686a37f7a0a..2e87faf9a8c2 100644 --- a/include/linux/binfmts.h +++ b/include/linux/binfmts.h @@ -128,7 +128,8 @@ struct linux_binfmt { struct module *module; int (*load_binary)(struct linux_binprm *); #ifdef CONFIG_COREDUMP - int (*core_dump)(struct coredump_params *cprm); + /* Returns true if the whole coredump was written. */ + bool (*core_dump)(struct coredump_params *cprm); unsigned long min_coredump; /* minimal dump size */ #endif } __randomize_layout; diff --git a/include/linux/bio-integrity.h b/include/linux/bio-integrity.h index 0ea2a8bf7efb..a954c97be0b3 100644 --- a/include/linux/bio-integrity.h +++ b/include/linux/bio-integrity.h @@ -151,7 +151,6 @@ void bio_integrity_setup_default(struct bio *bio); unsigned int fs_bio_integrity_alloc(struct bio *bio); void fs_bio_integrity_free(struct bio *bio); void fs_bio_integrity_generate(struct bio *bio); -int fs_bio_integrity_verify(struct bio *bio, sector_t sector, - unsigned int size); +int fs_bio_integrity_verify(struct bio *bio, struct bvec_iter *data_iter); #endif /* _LINUX_BIO_INTEGRITY_H */ diff --git a/include/linux/bio.h b/include/linux/bio.h index bb3235497e67..17944e44b584 100644 --- a/include/linux/bio.h +++ b/include/linux/bio.h @@ -479,6 +479,7 @@ static inline void bio_init_inline(struct bio *bio, struct block_device *bdev, extern void bio_uninit(struct bio *); void bio_reset(struct bio *bio, struct block_device *bdev, blk_opf_t opf); void bio_reuse(struct bio *bio, blk_opf_t opf); +void bio_prepare_reissue(struct bio *bio, struct block_device *bdev); void bio_chain(struct bio *, struct bio *); void bio_await(struct bio *bio, void *priv, void (*submit)(struct bio *bio, void *priv)); @@ -516,16 +517,18 @@ int bdev_rw_virt(struct block_device *bdev, sector_t sector, void *data, size_t len, enum req_op op); int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter, - unsigned mem_align_mask, unsigned len_align_mask); + unsigned maxlen, unsigned mem_align_mask, + unsigned len_align_mask); bool bio_iov_iter_set(struct bio *bio, const struct iov_iter *iter); void __bio_release_pages(struct bio *bio, bool mark_dirty); extern void bio_set_pages_dirty(struct bio *bio); extern void bio_check_pages_dirty(struct bio *bio); -int bio_iov_iter_bounce(struct bio *bio, struct iov_iter *iter, size_t maxlen, - size_t minsize); -void bio_iov_iter_unbounce(struct bio *bio, bool is_error, bool mark_dirty); +int bio_alloc_bounce_folios(struct bio *bio, size_t total_len, size_t minsize); +void bio_free_folios(struct bio *bio); +int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter, + size_t maxlen, size_t minsize); extern void bio_copy_data(struct bio *dst, struct bio *src); extern void bio_free_pages(struct bio *bio); diff --git a/include/linux/blkdev.h b/include/linux/blkdev.h index 4f7905c3412b..098a65f3e48b 100644 --- a/include/linux/blkdev.h +++ b/include/linux/blkdev.h @@ -1816,9 +1816,11 @@ static inline int bio_split_rw_at(struct bio *bio, */ static inline unsigned int max_integrity_io_size(struct queue_limits *lim) { - return min_t(unsigned int, lim->max_segment_size, - (BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size) << - lim->integrity.interval_exp); + u64 max_intervals; + + max_intervals = BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size; + return min_t(u64, lim->max_segment_size, + max_intervals << lim->integrity.interval_exp); } #define DEFINE_IO_COMP_BATCH(name) struct io_comp_batch name = { } diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index fd2c7115c054..e7a701ce029d 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -59,10 +59,7 @@ struct address_space; struct buffer_head { unsigned long b_state; /* buffer state bitmap (see above) */ struct buffer_head *b_this_page;/* circular list of page's buffers */ - union { - struct page *b_page; /* the page this bh is mapped to */ - struct folio *b_folio; /* the folio this bh is mapped to */ - }; + struct folio *b_folio; /* the folio this bh is mapped to */ sector_t b_blocknr; /* start block number */ size_t b_size; /* size of mapping */ @@ -172,7 +169,36 @@ static __always_inline int buffer_uptodate(const struct buffer_head *bh) static inline unsigned long bh_offset(const struct buffer_head *bh) { - return (unsigned long)(bh)->b_data & (page_size(bh->b_page) - 1); + return (unsigned long)(bh)->b_data & (folio_size(bh->b_folio) - 1); +} + +/** + * kmap_local_bh - Map the data of a buffer. + * @bh: The buffer. + * + * Buffers usually live in the page cache, but a few are built over memory + * which is not. Those carry no folio and b_data is already a kernel address + * which is always mapped, so there is nothing to do for them. Pair with + * kunmap_local_bh(). + * + * Return: A pointer to the buffer's data. + */ +static inline void *kmap_local_bh(const struct buffer_head *bh) +{ + if (!bh->b_folio) + return bh->b_data; + return kmap_local_folio(bh->b_folio, bh_offset(bh)); +} + +/** + * kunmap_local_bh - Unmap the data of a buffer. + * @bh: The buffer. + * @addr: The address returned by kmap_local_bh(). + */ +static inline void kunmap_local_bh(const struct buffer_head *bh, void *addr) +{ + if (bh->b_folio) + kunmap_local(addr); } /* If we *know* page->private refers to buffer_heads */ @@ -338,20 +364,58 @@ static inline void bforget(struct buffer_head *bh) __bforget(bh); } -static inline struct buffer_head * -sb_bread(struct super_block *sb, sector_t block) +/** + * sb_bread - Read a block. + * @sb: The superblock to read from. + * @block: Block number in units of block size. + * + * Read a specified block, and return the buffer head that refers + * to it. The memory is allocated from the movable area so that it can + * be migrated. The returned buffer head has its refcount increased. + * The caller should call brelse() when it has finished with the buffer. + * + * Context: May sleep waiting for I/O. + * Return: NULL if the block was unreadable. + */ +static inline +struct buffer_head *sb_bread(struct super_block *sb, sector_t block) { return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, __GFP_MOVABLE); } -static inline struct buffer_head * -sb_bread_unmovable(struct super_block *sb, sector_t block) +/** + * sb_bread_unmovable - Read a block. + * @sb: The superblock to read from. + * @block: Block number in units of block size. + * + * Read a specified block, and return the buffer head that refers to it. + * The memory is allocated from the unmovable area so that pointers into + * it remain valid after compaction runs. The returned buffer head has + * its refcount increased. The caller should call brelse() when it has + * finished with the buffer. + * + * Context: May sleep waiting for I/O. + * Return: NULL if the block was unreadable. + */ +static inline +struct buffer_head *sb_bread_unmovable(struct super_block *sb, sector_t block) { return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, 0); } -static inline void -sb_breadahead(struct super_block *sb, sector_t block) +/** + * sb_breadahead - Start readahead. + * @sb: Superblock identifying the block device. + * @block: The block to read. + * + * Read this block. The I/O will be flagged as being readahead rather + * than immediate read, but (unlike the page cache), surrounding blocks + * will not be read. + * + * Context: May sleep in order to allocate memory. + */ +static inline +void sb_breadahead(struct super_block *sb, sector_t block) { __breadahead(sb->s_bdev, block, sb->s_blocksize); } diff --git a/include/linux/capability.h b/include/linux/capability.h index f8532d92fcad..622137f66f09 100644 --- a/include/linux/capability.h +++ b/include/linux/capability.h @@ -186,9 +186,9 @@ static inline bool ns_capable_setid(struct user_namespace *ns, int cap) } #endif /* CONFIG_MULTIUSER */ bool privileged_wrt_inode_uidgid(struct user_namespace *ns, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct inode *inode); -bool capable_wrt_inode_uidgid(struct mnt_idmap *idmap, +bool capable_wrt_inode_uidgid(const struct mnt_idmap *idmap, const struct inode *inode, int cap); extern bool file_ns_capable(const struct file *file, struct user_namespace *ns, int cap); extern bool ptracer_capable(struct task_struct *tsk, struct user_namespace *ns); @@ -215,11 +215,11 @@ static inline bool checkpoint_restore_ns_capable_noaudit(struct user_namespace * } /* audit system wants to get cap info from files as well */ -int get_vfs_caps_from_disk(struct mnt_idmap *idmap, +int get_vfs_caps_from_disk(const struct mnt_idmap *idmap, const struct dentry *dentry, struct cpu_vfs_cap_data *cpu_caps); -int cap_convert_nscap(struct mnt_idmap *idmap, struct dentry *dentry, +int cap_convert_nscap(const struct mnt_idmap *idmap, struct dentry *dentry, const void **ivalue, size_t size); #endif /* !_LINUX_CAPABILITY_H */ diff --git a/include/linux/cleanup.h b/include/linux/cleanup.h index b1b5698cbf1b..1fb8058b897d 100644 --- a/include/linux/cleanup.h +++ b/include/linux/cleanup.h @@ -261,10 +261,6 @@ const volatile void * __must_check_fn(const volatile void *val) * CLASS(name, var)(args...): * declare the variable @var as an instance of the named class * - * CLASS_INIT(name, var, init_expr): - * declare the variable @var as an instance of the named class with - * custom initialization expression. - * * Ex. * * DEFINE_CLASS(fdget, struct fd, fdput(_T), fdget(fd), int fd) @@ -302,9 +298,6 @@ static __always_inline class_##_name##_t class_##_name##ext##_constructor(_init_ class_##_name##_t var __cleanup(class_##_name##_destructor) = \ class_##_name##_constructor -#define CLASS_INIT(_name, _var, _init_expr) \ - class_##_name##_t _var __cleanup(class_##_name##_destructor) = (_init_expr) - #define __scoped_class(_name, var, _label, args...) \ for (CLASS(_name, var)(args); ; ({ goto _label; })) \ if (0) { \ diff --git a/include/linux/coredump.h b/include/linux/coredump.h index 7b38ee2e7913..74af57b9406b 100644 --- a/include/linux/coredump.h +++ b/include/linux/coredump.h @@ -6,9 +6,20 @@ #include <linux/mm.h> #include <linux/fs.h> #include <linux/sched/coredump.h> +#include <uapi/linux/coredump.h> #include <asm/siginfo.h> #ifdef CONFIG_COREDUMP +/** + * enum coredump_state - what happened while the coredump was written + * @COREDUMP_STATE_STARTED: the dumper committed to writing a coredump + * @COREDUMP_STATE_TRUNCATED: the dumper stopped before it had written all of it + */ +enum coredump_state { + COREDUMP_STATE_STARTED = (1U << 0), + COREDUMP_STATE_TRUNCATED = (1U << 1), +}; + struct core_vma_metadata { unsigned long start, end; vm_flags_t flags; @@ -21,12 +32,20 @@ struct coredump_params { const kernel_siginfo_t *siginfo; struct file *file; unsigned long limit; - /* MMF_DUMP_FILTER_* bits, snapshot of mm->flags at dump start. */ - unsigned long mm_flags; + /* COREDUMP_MEMORY_* types to dump, the task's or the server's. */ + u64 memory_types; /* Snapshot of dumpable at dump start. */ enum task_dumpable dumpable; int cpu; + /* COREDUMP_* options negotiated with the coredump server. */ + u64 mask; + /* COREDUMP_STATE_* raised while the coredump is written. */ + enum coredump_state state; + /* Record header scratch, NULL unless the coredump is a record stream. */ + struct coredump_record_header *record_hdr; + /* Bytes handed to the file, record headers included. */ loff_t written; + /* Offset in the coredump, record headers excluded. */ loff_t pos; loff_t to_skip; int vma_count; @@ -41,13 +60,13 @@ extern unsigned int core_file_note_size_limit; * These are the only things you should do on a core-file: use only these * functions to write out all the necessary info. */ -extern void dump_skip_to(struct coredump_params *cprm, unsigned long to); -extern void dump_skip(struct coredump_params *cprm, size_t nr); -extern int dump_emit(struct coredump_params *cprm, const void *addr, int nr); -extern int dump_align(struct coredump_params *cprm, int align); -int dump_user_range(struct coredump_params *cprm, unsigned long start, - unsigned long len); -extern void vfs_coredump(const kernel_siginfo_t *siginfo); +void dump_skip_to(struct coredump_params *cprm, unsigned long to); +void dump_skip(struct coredump_params *cprm, size_t nr); +bool dump_emit(struct coredump_params *cprm, const void *addr, int nr); +bool dump_align(struct coredump_params *cprm, int align); +bool dump_user_range(struct coredump_params *cprm, unsigned long start, + unsigned long len); +void vfs_coredump(const kernel_siginfo_t *siginfo); /* * Logging for the coredump code, ratelimited. diff --git a/include/linux/dax.h b/include/linux/dax.h index fe6c3ded1b50..f2d47975d905 100644 --- a/include/linux/dax.h +++ b/include/linux/dax.h @@ -155,8 +155,6 @@ int dax_writeback_mapping_range(struct address_space *mapping, struct dax_device *dax_dev, struct writeback_control *wbc); int dax_folio_reset_order(struct folio *folio); -struct page *dax_layout_busy_page(struct address_space *mapping); -struct page *dax_layout_busy_page_range(struct address_space *mapping, loff_t start, loff_t end); dax_entry_t dax_lock_folio(struct folio *folio); void dax_unlock_folio(struct folio *folio, dax_entry_t cookie); dax_entry_t dax_lock_mapping_entry(struct address_space *mapping, @@ -173,16 +171,6 @@ static inline int fs_dax_get(struct dax_device *dax_dev, void *holder, { return -EOPNOTSUPP; } -static inline struct page *dax_layout_busy_page(struct address_space *mapping) -{ - return NULL; -} - -static inline struct page *dax_layout_busy_page_range(struct address_space *mapping, pgoff_t start, pgoff_t nr_pages) -{ - return NULL; -} - static inline int dax_writeback_mapping_range(struct address_space *mapping, struct dax_device *dax_dev, struct writeback_control *wbc) { diff --git a/include/linux/dcache.h b/include/linux/dcache.h index 4b1ff99608e0..adf239f8205f 100644 --- a/include/linux/dcache.h +++ b/include/linux/dcache.h @@ -116,6 +116,8 @@ struct dentry { * possible! */ + /* lockdep tracking of DCACHE_PAR_LOOKUP locks */ + struct lockdep_map lookup_map; struct list_head d_lru; /* LRU list */ struct hlist_node d_sib; /* child of parent list */ struct hlist_head d_children; /* our children */ @@ -236,7 +238,9 @@ enum dentry_flags { DCACHE_PAR_LOOKUP = BIT(24), /* being looked up (with parent locked shared) */ DCACHE_DENTRY_CURSOR = BIT(25), DCACHE_NORCU = BIT(26), /* No RCU delay for freeing */ - DCACHE_PERSISTENT = BIT(27) + DCACHE_PERSISTENT = BIT(27), +/* 28, 29, 30 free */ + DCACHE_PRIVATE = BIT(31) /* fs-specific flag */ }; #define DCACHE_MANAGED_DENTRY \ @@ -257,7 +261,9 @@ extern void d_delete(struct dentry *); extern struct dentry * d_alloc(struct dentry *, const struct qstr *); extern struct dentry * d_alloc_anon(struct super_block *); extern struct dentry * d_alloc_parallel(struct dentry *, const struct qstr *); +extern struct dentry * d_alloc_trylock(struct dentry *, struct qstr *); extern struct dentry * d_splice_alias(struct inode *, struct dentry *); +struct dentry *d_duplicate(struct dentry *dentry); /* weird procfs mess; *NOT* exported */ extern struct dentry * d_splice_alias_ops(struct inode *, struct dentry *, const struct dentry_operations *); @@ -553,6 +559,36 @@ static inline int simple_positive(const struct dentry *dentry) unsigned long vfs_pressure_ratio(unsigned long val); /** + * d_lookup_release - release ownership of DCACHE_PAR_LOOKUP lock + * @dentry: dentry that is locked + * + * If an in-lookup dentry is to be passed to another thread which + * will drop the in-lookup lock, then d_lookup_release() must be called + * to tell lockdep that this thread no lock holds the lock. The + * thread that receives the lock must call d_lookup_acquire() to + * acquire the lock. + */ +static inline void d_lookup_release(struct dentry *dentry) +{ + if (d_in_lookup(dentry)) + lock_map_release(&dentry->lookup_map); +} + +/** + * d_lookup_acquire - acquire ownership of DCACHE_PAR_LOOKUP lock + * @dentry: dentry that is locked + * + * If an in-lookup dentry was passed to this thread, the + * d_lookup_acquire() must be called to tell lockdep that this + * thread now owns the DCACHE_PAR_LOOKUP lock. + */ +static inline void d_lookup_acquire(struct dentry *dentry) +{ + if (d_in_lookup(dentry)) + lock_map_acquire_try(&dentry->lookup_map); +} + +/** * d_inode - Get the actual inode of this dentry * @dentry: The dentry to query * diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index c45306a9f007..a46781058729 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -25,7 +25,7 @@ struct fdtable { unsigned int max_fds; - struct file __rcu **fd; /* current fd array */ + struct file __rcu **fd __counted_by_ptr(max_fds); /* current fd array */ unsigned long *close_on_exec; unsigned long *open_fds; unsigned long *full_fds_bits; @@ -101,11 +101,22 @@ struct task_struct; void put_files_struct(struct files_struct *fs); int unshare_files(void); +void switch_files_struct(struct task_struct *tsk, struct files_struct *files); +int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp); +enum fd_range_flags { + /* Leave behind all descriptors outside of the specified range. */ + FD_RANGE_EXCEPT = (1U << 0), + + /* Only select descriptors that have close-on-exec set. */ + FD_RANGE_CLOEXEC_ONLY = (1U << 1), +}; + struct fd_range { unsigned int from, to; + enum fd_range_flags flags; }; struct files_struct *dup_fd(struct files_struct *, struct fd_range *) __latent_entropy; -void do_close_on_exec(struct files_struct *); +void close_cloexec_files(struct files_struct *); int iterate_fd(struct files_struct *, unsigned, int (*)(const void *, struct file *, unsigned), const void *); diff --git a/include/linux/file.h b/include/linux/file.h index 27484b444d31..41c3c0be1064 100644 --- a/include/linux/file.h +++ b/include/linux/file.h @@ -12,6 +12,7 @@ #include <linux/errno.h> #include <linux/cleanup.h> #include <linux/err.h> +#include <linux/vfsdebug.h> struct file; @@ -129,117 +130,84 @@ extern unsigned int sysctl_nr_open_min, sysctl_nr_open_max; /* * fd_prepare: Combined fd + file allocation cleanup class. - * @err: Error code to indicate if allocation succeeded. - * @__fd: Allocated fd (may not be accessed directly) - * @__file: Allocated struct file pointer (may not be accessed directly) + * @fd: Allocated fd + * @file: Allocated struct file pointer * * Allocates an fd and a file together. On error paths, automatically cleans * up whichever resource was successfully allocated. Allows flexible file * allocation with different functions per usage. * - * Do not use directly. + * Do not declare directly, use FD_PREPARE(). */ struct fd_prepare { - s32 err; - s32 __fd; /* do not access directly */ - struct file *__file; /* do not access directly */ + int fd; + struct file *file; }; -/* Typedef for fd_prepare cleanup guards. */ -typedef struct fd_prepare class_fd_prepare_t; - -/* - * Accessors for fd_prepare class members. - * _Generic() is used for zero-cost type safety. - */ -#define fd_prepare_fd(_fdf) \ - (_Generic((_fdf), struct fd_prepare: (_fdf).__fd)) - -#define fd_prepare_file(_fdf) \ - (_Generic((_fdf), struct fd_prepare: (_fdf).__file)) - /* Do not use directly. */ -static inline void class_fd_prepare_destructor(const struct fd_prepare *fdf) +static __always_inline void __fd_prepare_cleanup(const struct fd_prepare *fdf) { - if (unlikely(fdf->__fd >= 0)) - put_unused_fd(fdf->__fd); - if (unlikely(!IS_ERR_OR_NULL(fdf->__file))) - fput(fdf->__file); + if (unlikely(fdf->fd >= 0)) { + put_unused_fd(fdf->fd); + fput(fdf->file); + } } /* Do not use directly. */ -static inline int class_fd_prepare_lock_err(const struct fd_prepare *fdf) +static __always_inline struct fd_prepare __fd_prepare(int fd, struct file *file) { - if (unlikely(fdf->err)) - return fdf->err; - if (unlikely(fdf->__fd < 0)) - return fdf->__fd; - if (unlikely(IS_ERR(fdf->__file))) - return PTR_ERR(fdf->__file); - if (unlikely(!fdf->__file)) - return -ENOMEM; - return 0; + if (fd >= 0 && IS_ERR_OR_NULL(file)) { + int err = file ? PTR_ERR(file) : -ENOMEM; + + put_unused_fd(fd); + fd = err; + file = NULL; + } + return (struct fd_prepare){ .fd = fd, .file = file }; } /* - * __FD_PREPARE_INIT - Helper to initialize fd_prepare class. - * @_fd_flags: flags for get_unused_fd_flags() - * @_file_owned: expression that returns struct file * - * - * Returns a struct fd_prepare with fd, file, and err set. - * If fd allocation fails, fd will be negative and err will be set. If - * fd succeeds but file_init_expr fails, file will be ERR_PTR and err - * will be set. The err field is the single source of truth for error - * checking. - */ -#define __FD_PREPARE_INIT(_fd_flags, _file_owned) \ - ({ \ - struct fd_prepare fdf = { \ - .__fd = get_unused_fd_flags((_fd_flags)), \ - }; \ - if (likely(fdf.__fd >= 0)) \ - fdf.__file = (_file_owned); \ - fdf.err = ACQUIRE_ERR(fd_prepare, &fdf); \ - fdf; \ - }) - -/* - * FD_PREPARE - Macro to declare and initialize an fd_prepare variable. + * FD_PREPARE - Declare and initialize an fd_prepare instance. * - * Declares and initializes an fd_prepare variable with automatic - * cleanup. No separate scope required - cleanup happens when variable - * goes out of scope. + * This allocates a new fd and only evaluates @_file_owned if the + * allocation succeeded. Cleanup happens when the variable goes out of + * scope and the guard releases whichever of the descriptor and the file + * was allocated. If fd_publish() was called the fd and file are + * published and cleanup becomes a nop. * - * @_fdf: name of struct fd_prepare variable to define + * @_fdf: name of the const struct fd_prepare pointer to define * @_fd_flags: flags for get_unused_fd_flags() * @_file_owned: struct file to take ownership of (can be expression) */ +#define __FD_PREPARE(_guard, _fdf, _fd_flags, _file_owned) \ + struct fd_prepare _guard __cleanup(__fd_prepare_cleanup) = ({ \ + int __fd = get_unused_fd_flags(_fd_flags); \ + __fd_prepare(__fd, __fd < 0 ? NULL : (_file_owned)); \ + }); \ + const struct fd_prepare *const _fdf = &_guard + #define FD_PREPARE(_fdf, _fd_flags, _file_owned) \ - CLASS_INIT(fd_prepare, _fdf, __FD_PREPARE_INIT(_fd_flags, _file_owned)) + __FD_PREPARE(__UNIQUE_ID(fd_prepare), _fdf, _fd_flags, _file_owned) /* * fd_publish - Publish prepared fd and file to the fd table. - * @_fdf: struct fd_prepare variable + * @fdf: struct fd_prepare pointer defined by FD_PREPARE() */ -#define fd_publish(_fdf) \ - ({ \ - struct fd_prepare *fdp = &(_fdf); \ - VFS_WARN_ON_ONCE(fdp->err); \ - VFS_WARN_ON_ONCE(fdp->__fd < 0); \ - VFS_WARN_ON_ONCE(IS_ERR_OR_NULL(fdp->__file)); \ - fd_install(fdp->__fd, fdp->__file); \ - retain_and_null_ptr(fdp->__file); \ - take_fd(fdp->__fd); \ - }) +static __always_inline int fd_publish(const struct fd_prepare *fdf) +{ + /* Callers only get a const view, the guard itself is writable. */ + struct fd_prepare *guard = (struct fd_prepare *)fdf; + + VFS_WARN_ON_ONCE(guard->fd < 0); + fd_install(guard->fd, guard->file); + return take_fd(guard->fd); +} /* Do not use directly. */ -#define __FD_ADD(_fdf, _fd_flags, _file_owned) \ - ({ \ - FD_PREPARE(_fdf, _fd_flags, _file_owned); \ - s32 ret = _fdf.err; \ - if (likely(!ret)) \ - ret = fd_publish(_fdf); \ - ret; \ +#define __FD_ADD(_fdf, _fd_flags, _file_owned) \ + ({ \ + FD_PREPARE(_fdf, _fd_flags, _file_owned); \ + _fdf->fd < 0 ? _fdf->fd : fd_publish(_fdf); \ }) /* diff --git a/include/linux/fileattr.h b/include/linux/fileattr.h index 58044b598016..09e32b84e02a 100644 --- a/include/linux/fileattr.h +++ b/include/linux/fileattr.h @@ -74,7 +74,7 @@ static inline bool fileattr_has_fsx(const struct file_kattr *fa) } int vfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int vfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int ioctl_getflags(struct file *file, unsigned int __user *argp); int ioctl_setflags(struct file *file, unsigned int __user *argp); diff --git a/include/linux/fs.h b/include/linux/fs.h index f9d1e05e8ae6..3db90996756f 100644 --- a/include/linux/fs.h +++ b/include/linux/fs.h @@ -1436,10 +1436,10 @@ static inline void i_gid_write(struct inode *inode, gid_t gid) * @idmap: idmap of the mount the inode was found from * @inode: inode to map * - * Return: whe inode's i_uid mapped down according to @idmap. + * Return: the inode's i_uid mapped down according to @idmap. * If the inode's i_uid has no mapping INVALID_VFSUID is returned. */ -static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap, +static inline vfsuid_t i_uid_into_vfsuid(const struct mnt_idmap *idmap, const struct inode *inode) { return make_vfsuid(idmap, i_user_ns(inode), inode->i_uid); @@ -1456,7 +1456,7 @@ static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap, * * Return: true if @inode's i_uid field needs to be updated, false if not. */ -static inline bool i_uid_needs_update(struct mnt_idmap *idmap, +static inline bool i_uid_needs_update(const struct mnt_idmap *idmap, const struct iattr *attr, const struct inode *inode) { @@ -1474,7 +1474,7 @@ static inline bool i_uid_needs_update(struct mnt_idmap *idmap, * Safely update @inode's i_uid field translating the vfsuid of any idmapped * mount into the filesystem kuid. */ -static inline void i_uid_update(struct mnt_idmap *idmap, +static inline void i_uid_update(const struct mnt_idmap *idmap, const struct iattr *attr, struct inode *inode) { @@ -1491,7 +1491,7 @@ static inline void i_uid_update(struct mnt_idmap *idmap, * Return: the inode's i_gid mapped down according to @idmap. * If the inode's i_gid has no mapping INVALID_VFSGID is returned. */ -static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap, +static inline vfsgid_t i_gid_into_vfsgid(const struct mnt_idmap *idmap, const struct inode *inode) { return make_vfsgid(idmap, i_user_ns(inode), inode->i_gid); @@ -1508,7 +1508,7 @@ static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap, * * Return: true if @inode's i_gid field needs to be updated, false if not. */ -static inline bool i_gid_needs_update(struct mnt_idmap *idmap, +static inline bool i_gid_needs_update(const struct mnt_idmap *idmap, const struct iattr *attr, const struct inode *inode) { @@ -1526,7 +1526,7 @@ static inline bool i_gid_needs_update(struct mnt_idmap *idmap, * Safely update @inode's i_gid field translating the vfsgid of any idmapped * mount into the filesystem kgid. */ -static inline void i_gid_update(struct mnt_idmap *idmap, +static inline void i_gid_update(const struct mnt_idmap *idmap, const struct iattr *attr, struct inode *inode) { @@ -1544,7 +1544,7 @@ static inline void i_gid_update(struct mnt_idmap *idmap, * an idmapped mount map the caller's fsuid according to @idmap. */ static inline void inode_fsuid_set(struct inode *inode, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { inode->i_uid = mapped_fsuid(idmap, i_user_ns(inode)); } @@ -1558,7 +1558,7 @@ static inline void inode_fsuid_set(struct inode *inode, * an idmapped mount map the caller's fsgid according to @idmap. */ static inline void inode_fsgid_set(struct inode *inode, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { inode->i_gid = mapped_fsgid(idmap, i_user_ns(inode)); } @@ -1575,7 +1575,7 @@ static inline void inode_fsgid_set(struct inode *inode, * Return: true if fsuid and fsgid is mapped, false if not. */ static inline bool fsuidgid_has_mapping(struct super_block *sb, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { struct user_namespace *fs_userns = sb->s_user_ns; kuid_t kuid; @@ -1755,25 +1755,25 @@ static inline bool file_write_not_started(const struct file *file) return sb_write_not_started(file_inode(file)->i_sb); } -bool inode_owner_or_capable(struct mnt_idmap *idmap, +bool inode_owner_or_capable(const struct mnt_idmap *idmap, const struct inode *inode); /* * VFS helper functions.. */ -int vfs_create(struct mnt_idmap *, struct dentry *, umode_t, +int vfs_create(const struct mnt_idmap *, struct dentry *, umode_t, struct delegated_inode *); -struct dentry *vfs_mkdir(struct mnt_idmap *, struct inode *, +struct dentry *vfs_mkdir(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, struct delegated_inode *); -int vfs_mknod(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_mknod(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, dev_t, struct delegated_inode *); -int vfs_symlink(struct mnt_idmap *, struct inode *, +int vfs_symlink(const struct mnt_idmap *, struct inode *, struct dentry *, const char *, struct delegated_inode *); -int vfs_link(struct dentry *, struct mnt_idmap *, struct inode *, +int vfs_link(struct dentry *, const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); -int vfs_rmdir(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_rmdir(const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); -int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_unlink(const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); /** @@ -1787,7 +1787,7 @@ int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *, * @flags: rename flags */ struct renamedata { - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; struct dentry *old_parent; struct dentry *old_dentry; struct dentry *new_parent; @@ -1798,14 +1798,14 @@ struct renamedata { int vfs_rename(struct renamedata *); -static inline int vfs_whiteout(struct mnt_idmap *idmap, +static inline int vfs_whiteout(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry) { return vfs_mknod(idmap, dir, dentry, S_IFCHR | WHITEOUT_MODE, WHITEOUT_DEV, NULL); } -struct file *kernel_tmpfile_open(struct mnt_idmap *idmap, +struct file *kernel_tmpfile_open(const struct mnt_idmap *idmap, const struct path *parentpath, umode_t mode, int open_flag, const struct cred *cred); @@ -1830,12 +1830,12 @@ extern long compat_ptr_ioctl(struct file *file, unsigned int cmd, /* * VFS file helper functions. */ -void inode_init_owner(struct mnt_idmap *idmap, struct inode *inode, +void inode_init_owner(const struct mnt_idmap *idmap, struct inode *inode, const struct inode *dir, umode_t mode); extern bool may_open_dev(const struct path *path); -umode_t mode_strip_sgid(struct mnt_idmap *idmap, +umode_t mode_strip_sgid(const struct mnt_idmap *idmap, const struct inode *dir, umode_t mode); -bool in_group_or_capable(struct mnt_idmap *idmap, +bool in_group_or_capable(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t vfsgid); /* @@ -1994,26 +1994,26 @@ enum fs_update_time { struct inode_operations { struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int); const char * (*get_link) (struct dentry *, struct inode *, struct delayed_call *); - int (*permission) (struct mnt_idmap *, struct inode *, int); + int (*permission) (const struct mnt_idmap *, struct inode *, int); struct posix_acl * (*get_inode_acl)(struct inode *, int, bool); int (*readlink) (struct dentry *, char __user *,int); - int (*create) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*create) (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t); int (*link) (struct dentry *,struct inode *,struct dentry *); int (*unlink) (struct inode *,struct dentry *); - int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*symlink) (const struct mnt_idmap *, struct inode *,struct dentry *, const char *); - struct dentry *(*mkdir) (struct mnt_idmap *, struct inode *, + struct dentry *(*mkdir) (const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); int (*rmdir) (struct inode *,struct dentry *); - int (*mknod) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*mknod) (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t,dev_t); - int (*rename) (struct mnt_idmap *, struct inode *, struct dentry *, + int (*rename) (const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); - int (*setattr) (struct mnt_idmap *, struct dentry *, struct iattr *); - int (*getattr) (struct mnt_idmap *, const struct path *, + int (*setattr) (const struct mnt_idmap *, struct dentry *, struct iattr *); + int (*getattr) (const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); ssize_t (*listxattr) (struct dentry *, char *, size_t); int (*fiemap)(struct inode *, struct fiemap_extent_info *, u64 start, @@ -2024,13 +2024,13 @@ struct inode_operations { int (*atomic_open)(struct inode *, struct dentry *, struct file *, unsigned open_flag, umode_t create_mode); - int (*tmpfile) (struct mnt_idmap *, struct inode *, + int (*tmpfile) (const struct mnt_idmap *, struct inode *, struct file *, umode_t); - struct posix_acl *(*get_acl)(struct mnt_idmap *, struct dentry *, + struct posix_acl *(*get_acl)(const struct mnt_idmap *, struct dentry *, int); - int (*set_acl)(struct mnt_idmap *, struct dentry *, + int (*set_acl)(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); - int (*fileattr_set)(struct mnt_idmap *idmap, + int (*fileattr_set)(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int (*fileattr_get)(struct dentry *dentry, struct file_kattr *fa); struct offset_ctx *(*get_offset_ctx)(struct inode *inode); @@ -2173,7 +2173,7 @@ extern loff_t vfs_dedupe_file_range_one(struct file *src_file, loff_t src_pos, (inode)->i_rdev == WHITEOUT_DEV) #define IS_ANON_FILE(inode) ((inode)->i_flags & S_ANON_INODE) -static inline bool HAS_UNMAPPED_ID(struct mnt_idmap *idmap, +static inline bool HAS_UNMAPPED_ID(const struct mnt_idmap *idmap, struct inode *inode) { return !vfsuid_valid(i_uid_into_vfsuid(idmap, inode)) || @@ -2459,7 +2459,7 @@ struct filename { static_assert(offsetof(struct filename, iname) % sizeof(long) == 0); static_assert(sizeof(struct filename) % 64 == 0); -static inline struct mnt_idmap *file_mnt_idmap(const struct file *file) +static inline const struct mnt_idmap *file_mnt_idmap(const struct file *file) { return mnt_idmap(file->f_path.mnt); } @@ -2483,7 +2483,7 @@ static inline bool is_idmapped_mnt(const struct vfsmount *mnt) } int vfs_truncate(const struct path *, loff_t); -int do_truncate(struct mnt_idmap *, struct dentry *, loff_t start, +int do_truncate(const struct mnt_idmap *, struct dentry *, loff_t start, unsigned int time_attrs, struct file *filp); extern int vfs_fallocate(struct file *file, int mode, loff_t offset, loff_t len); @@ -2707,10 +2707,10 @@ static inline int bmap(struct inode *inode, sector_t *block) } #endif -int notify_change(struct mnt_idmap *, struct dentry *, +int notify_change(const struct mnt_idmap *, struct dentry *, struct iattr *, struct delegated_inode *); -int inode_permission(struct mnt_idmap *, struct inode *, int); -int generic_permission(struct mnt_idmap *, struct inode *, int); +int inode_permission(const struct mnt_idmap *, struct inode *, int); +int generic_permission(const struct mnt_idmap *, struct inode *, int); static inline int file_permission(struct file *file, int mask) { return inode_permission(file_mnt_idmap(file), @@ -2721,12 +2721,12 @@ static inline int path_permission(const struct path *path, int mask) return inode_permission(mnt_idmap(path->mnt), d_inode(path->dentry), mask); } -int __check_sticky(struct mnt_idmap *idmap, struct inode *dir, +int __check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode); -int may_delete_dentry(struct mnt_idmap *idmap, struct inode *dir, +int may_delete_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *victim, bool isdir); -int may_create_dentry(struct mnt_idmap *idmap, +int may_create_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *child); static inline bool execute_ok(struct inode *inode) @@ -3045,9 +3045,9 @@ static inline struct inode *new_inode_pseudo(struct super_block *sb) } extern struct inode *new_inode(struct super_block *sb); extern void free_inode_nonrcu(struct inode *inode); -extern int setattr_should_drop_suidgid(struct mnt_idmap *, struct inode *); +extern int setattr_should_drop_suidgid(const struct mnt_idmap *, struct inode *); extern int file_remove_privs(struct file *); -int setattr_should_drop_sgid(struct mnt_idmap *idmap, +int setattr_should_drop_sgid(const struct mnt_idmap *idmap, const struct inode *inode); /* @@ -3204,7 +3204,7 @@ extern int page_symlink(struct inode *inode, const char *symname, int len); extern const struct inode_operations page_symlink_inode_operations; extern void kfree_link(void *); void fill_mg_cmtime(struct kstat *stat, u32 request_mask, struct inode *inode); -void generic_fillattr(struct mnt_idmap *, u32, struct inode *, struct kstat *); +void generic_fillattr(const struct mnt_idmap *, u32, struct inode *, struct kstat *); void generic_fill_statx_attr(struct inode *inode, struct kstat *stat); void generic_fill_statx_atomic_writes(struct kstat *stat, unsigned int unit_min, @@ -3261,9 +3261,9 @@ extern int dcache_dir_open(struct inode *, struct file *); extern int dcache_dir_close(struct inode *, struct file *); extern loff_t dcache_dir_lseek(struct file *, loff_t, int); extern int dcache_readdir(struct file *, struct dir_context *); -extern int simple_setattr(struct mnt_idmap *, struct dentry *, +extern int simple_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); -extern int simple_getattr(struct mnt_idmap *, const struct path *, +extern int simple_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern int simple_statfs(struct dentry *, struct kstatfs *); extern int simple_open(struct inode *inode, struct file *file); @@ -3276,7 +3276,7 @@ void simple_rename_timestamp(struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry); extern int simple_rename_exchange(struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry); -extern int simple_rename(struct mnt_idmap *, struct inode *, +extern int simple_rename(const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); extern void simple_recursive_removal(struct dentry *, @@ -3397,11 +3397,11 @@ static inline bool generic_ci_validate_strict_name(struct inode *dir, } #endif -int may_setattr(struct mnt_idmap *idmap, struct inode *inode, +int may_setattr(const struct mnt_idmap *idmap, struct inode *inode, unsigned int ia_valid); -int setattr_prepare(struct mnt_idmap *, struct dentry *, struct iattr *); +int setattr_prepare(const struct mnt_idmap *, struct dentry *, struct iattr *); extern int inode_newsize_ok(const struct inode *, loff_t offset); -void setattr_copy(struct mnt_idmap *, struct inode *inode, +void setattr_copy(const struct mnt_idmap *, struct inode *inode, const struct iattr *attr); extern int file_update_time(struct file *file); @@ -3578,7 +3578,7 @@ static inline bool is_sxid(umode_t mode) return mode & (S_ISUID | S_ISGID); } -static inline int check_sticky(struct mnt_idmap *idmap, +static inline int check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode) { if (!(dir->i_mode & S_ISVTX)) @@ -3653,23 +3653,6 @@ extern int vfs_fadvise(struct file *file, loff_t offset, loff_t len, extern int generic_fadvise(struct file *file, loff_t offset, loff_t len, int advice); -static inline bool vfs_empty_path(int dfd, const char __user *path) -{ - char c; - - if (dfd < 0) - return false; - - /* We now allow NULL to be used for empty path. */ - if (!path) - return true; - - if (unlikely(get_user(c, path))) - return false; - - return !c; -} - int generic_atomic_write_valid(struct kiocb *iocb, struct iov_iter *iter); static inline bool extensible_ioctl_valid(unsigned int cmd_a, diff --git a/include/linux/fs_context.h b/include/linux/fs_context.h index 0d6c8a6d7be2..c920aba5177c 100644 --- a/include/linux/fs_context.h +++ b/include/linux/fs_context.h @@ -150,6 +150,10 @@ extern int vfs_parse_fs_param_source(struct fs_context *fc, struct fs_parameter *param); extern void fc_drop_locked(struct fs_context *fc); +extern int get_tree_super(struct fs_context *fc, + int (*test)(struct super_block *, struct fs_context *), + int (*fill_super)(struct super_block *sb, + struct fs_context *fc)); extern int get_tree_nodev(struct fs_context *fc, int (*fill_super)(struct super_block *sb, struct fs_context *fc)); diff --git a/include/linux/fscache-cache.h b/include/linux/fscache-cache.h index 4c91a019972b..ee524c863fa9 100644 --- a/include/linux/fscache-cache.h +++ b/include/linux/fscache-cache.h @@ -67,7 +67,7 @@ struct fscache_cache_ops { /* Change the size of a data object */ void (*resize_cookie)(struct netfs_cache_resources *cres, - loff_t new_size); + uoff_t new_size); /* Invalidate an object */ bool (*invalidate_cookie)(struct fscache_cookie *cookie); diff --git a/include/linux/fscache.h b/include/linux/fscache.h index 58fdb9605425..f2d958bd1f48 100644 --- a/include/linux/fscache.h +++ b/include/linux/fscache.h @@ -112,7 +112,7 @@ struct fscache_cookie { struct list_head proc_link; /* Link in proc list */ struct list_head commit_link; /* Link in commit queue */ struct work_struct work; /* Commit/relinq/withdraw work */ - loff_t object_size; /* Size of the netfs object */ + uoff_t object_size; /* Size of the netfs object */ unsigned long unused_at; /* Time at which unused (jiffies) */ unsigned long flags; #define FSCACHE_COOKIE_RELINQUISHED 0 /* T if cookie has been relinquished */ @@ -147,6 +147,23 @@ struct fscache_cookie { }; }; +enum fscache_extent_type { + FSCACHE_EXTENT_DATA, + FSCACHE_EXTENT_ZERO, +} __mode(byte); + +/* + * Cache occupancy information. + */ +struct fscache_occupancy { + unsigned long long query_from; /* Point to query from */ + unsigned long long query_to; /* Point to query to */ + unsigned long long cached_from[2]; /* Point at which cache extents start */ + unsigned long long cached_to[2]; /* Point at which cache extents end */ + unsigned int granularity; /* Granularity desired */ + enum fscache_extent_type cached_type[2]; /* Type of cache extent */ +}; + /* * slow-path functions for when there is actually caching available, and the * netfs does actually have a valid token @@ -163,22 +180,22 @@ extern struct fscache_cookie *__fscache_acquire_cookie( u8, const void *, size_t, const void *, size_t, - loff_t); + uoff_t); extern void __fscache_use_cookie(struct fscache_cookie *, bool); -extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const loff_t *); +extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const uoff_t *); extern void __fscache_relinquish_cookie(struct fscache_cookie *, bool); -extern void __fscache_resize_cookie(struct fscache_cookie *, loff_t); -extern void __fscache_invalidate(struct fscache_cookie *, const void *, loff_t, unsigned int); +extern void __fscache_resize_cookie(struct fscache_cookie *, uoff_t); +extern void __fscache_invalidate(struct fscache_cookie *, const void *, uoff_t, unsigned int); extern int __fscache_begin_read_operation(struct netfs_cache_resources *, struct fscache_cookie *); extern int __fscache_begin_write_operation(struct netfs_cache_resources *, struct fscache_cookie *); void __fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool cond); -extern void __fscache_clear_page_bits(struct address_space *, loff_t, size_t); +extern void __fscache_clear_page_bits(struct address_space *, uoff_t, size_t); /** * fscache_acquire_volume - Register a volume as desiring caching services @@ -249,7 +266,7 @@ struct fscache_cookie *fscache_acquire_cookie(struct fscache_volume *volume, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { if (!fscache_volume_valid(volume)) return NULL; @@ -286,7 +303,7 @@ static inline void fscache_use_cookie(struct fscache_cookie *cookie, */ static inline void fscache_unuse_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_valid(cookie)) __fscache_unuse_cookie(cookie, aux_data, object_size); @@ -327,7 +344,7 @@ static inline void *fscache_get_aux(struct fscache_cookie *cookie) */ static inline void fscache_update_aux(struct fscache_cookie *cookie, - const void *aux_data, const loff_t *object_size) + const void *aux_data, const uoff_t *object_size) { void *p = fscache_get_aux(cookie); @@ -343,7 +360,7 @@ extern atomic_t fscache_n_updates; static inline void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { #ifdef CONFIG_FSCACHE_STATS atomic_inc(&fscache_n_updates); @@ -369,7 +386,7 @@ void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data */ static inline void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_enabled(cookie)) __fscache_update_cookie(cookie, aux_data, object_size); @@ -386,7 +403,7 @@ void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, * description. */ static inline -void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) +void fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size) { if (fscache_cookie_enabled(cookie)) __fscache_resize_cookie(cookie, new_size); @@ -413,7 +430,7 @@ void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) */ static inline void fscache_invalidate(struct fscache_cookie *cookie, - const void *aux_data, loff_t size, unsigned int flags) + const void *aux_data, uoff_t size, unsigned int flags) { if (fscache_cookie_enabled(cookie)) __fscache_invalidate(cookie, aux_data, size, flags); @@ -502,7 +519,7 @@ static inline void fscache_end_operation(struct netfs_cache_resources *cres) */ static inline int fscache_read(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -561,7 +578,7 @@ int fscache_begin_write_operation(struct netfs_cache_resources *cres, */ static inline int fscache_write(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -581,7 +598,7 @@ int fscache_write(struct netfs_cache_resources *cres, * waiting. */ static inline void fscache_clear_page_bits(struct address_space *mapping, - loff_t start, size_t len, + uoff_t start, size_t len, bool caching) { if (caching) @@ -615,7 +632,7 @@ static inline void fscache_clear_page_bits(struct address_space *mapping, */ static inline void fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool caching) diff --git a/include/linux/iomap.h b/include/linux/iomap.h index bc7ae6327dbf..59718f73c15a 100644 --- a/include/linux/iomap.h +++ b/include/linux/iomap.h @@ -483,13 +483,35 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno, #define IOMAP_IOEND_BOUNDARY (1U << 2) /* is direct I/O */ #define IOMAP_IOEND_DIRECT (1U << 3) +/* generate integrity (PI) information */ +#ifdef CONFIG_BLK_DEV_INTEGRITY +#define IOMAP_IOEND_INTEGRITY (1U << 4) +#else +#define IOMAP_IOEND_INTEGRITY 0 +#endif /* CONFIG_BLK_DEV_INTEGRITY */ /* * Flags that if set on either ioend prevent the merge of two ioends. * (IOMAP_IOEND_BOUNDARY also prevents merges, but only one-way) */ #define IOMAP_IOEND_NOMERGE_FLAGS \ - (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT) + (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT | \ + IOMAP_IOEND_INTEGRITY) + +/* ioend flags directly implied by iomap flags */ +static inline u16 iomap_ioend_flags(const struct iomap *iomap) +{ + unsigned int flags = 0; + + if (iomap->type == IOMAP_UNWRITTEN) + flags |= IOMAP_IOEND_UNWRITTEN; + if (iomap->flags & IOMAP_F_SHARED) + flags |= IOMAP_IOEND_SHARED; + if (iomap->flags & IOMAP_F_INTEGRITY) + flags |= IOMAP_IOEND_INTEGRITY; + + return flags; +} /* * Structure for writeback I/O completions. @@ -500,6 +522,7 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno, struct iomap_ioend { struct list_head io_list; /* next ioend in chain */ u16 io_flags; /* IOMAP_IOEND_* */ + u32 io_bvec_offset; /* offset into first bvec */ struct inode *io_inode; /* file being written to */ size_t io_size; /* size of the extent */ atomic_t io_remaining; /* completetion defer count */ @@ -517,6 +540,13 @@ static inline struct iomap_ioend *iomap_ioend_from_bio(struct bio *bio) return container_of(bio, struct iomap_ioend, io_bio); } +#define BVEC_ITER_IOEND(_ioend) \ +{ \ + .bi_sector = (_ioend)->io_sector, \ + .bi_size = (_ioend)->io_size, \ + .bi_offset = (_ioend)->io_bvec_offset, \ +} + struct iomap_writeback_ops { /* * Performs writeback on the passed in range @@ -565,6 +595,7 @@ void iomap_finish_ioends(struct iomap_ioend *ioend, int error); void iomap_ioend_try_merge(struct iomap_ioend *ioend, struct list_head *more_ioends); void iomap_sort_ioends(struct list_head *ioend_list); +int iomap_ioend_integrity_verify(struct iomap_ioend *ioend); ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio, loff_t pos, loff_t end_pos, unsigned int dirty_len); int iomap_ioend_writeback_submit(struct iomap_writepage_ctx *wpc, int error); @@ -577,6 +608,11 @@ void iomap_finish_folio_write(struct inode *inode, struct folio *folio, int iomap_writeback_folio(struct iomap_writepage_ctx *wpc, struct folio *folio); int iomap_writepages(struct iomap_writepage_ctx *wpc); +void iomap_bounce_read(struct iomap_ioend *orig_ioend, unsigned int minsize, + void (*submit_ioend)(struct iomap_ioend *ioend)); +void iomap_bounce_read_end_io(struct iomap_ioend *ioend, struct bio *orig_bio, + int error); + struct iomap_read_folio_ctx { const struct iomap_read_ops *ops; struct folio *cur_folio; diff --git a/include/linux/lsm_hook_defs.h b/include/linux/lsm_hook_defs.h index 65c9609ec207..c9561564585e 100644 --- a/include/linux/lsm_hook_defs.h +++ b/include/linux/lsm_hook_defs.h @@ -36,6 +36,7 @@ LSM_HOOK(int, 0, binder_transfer_file, const struct cred *from, LSM_HOOK(int, 0, ptrace_access_check, struct task_struct *child, unsigned int mode) LSM_HOOK(int, 0, ptrace_traceme, struct task_struct *parent) +LSM_HOOK(int, 0, mem_foll_force, const struct cred *subject, bool opened_by_owner) LSM_HOOK(int, 0, capget, const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, kernel_cap_t *permitted) LSM_HOOK(int, 0, capset, struct cred *new, const struct cred *old, @@ -94,7 +95,7 @@ LSM_HOOK(int, 0, path_mkdir, const struct path *dir, struct dentry *dentry, LSM_HOOK(int, 0, path_rmdir, const struct path *dir, struct dentry *dentry) LSM_HOOK(int, 0, path_mknod, const struct path *dir, struct dentry *dentry, umode_t mode, unsigned int dev) -LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, const struct mnt_idmap *idmap, struct dentry *dentry) LSM_HOOK(int, 0, path_truncate, const struct path *path) LSM_HOOK(int, 0, path_symlink, const struct path *dir, struct dentry *dentry, @@ -122,7 +123,7 @@ LSM_HOOK(int, 0, inode_init_security_anon, struct inode *inode, const struct qstr *name, const struct inode *context_inode) LSM_HOOK(int, 0, inode_create, struct inode *dir, struct dentry *dentry, umode_t mode) -LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, const struct mnt_idmap *idmap, struct inode *inode) LSM_HOOK(int, 0, inode_link, struct dentry *old_dentry, struct inode *dir, struct dentry *new_dentry) @@ -140,39 +141,39 @@ LSM_HOOK(int, 0, inode_readlink, struct dentry *dentry) LSM_HOOK(int, 0, inode_follow_link, struct dentry *dentry, struct inode *inode, bool rcu) LSM_HOOK(int, 0, inode_permission, struct inode *inode, int mask) -LSM_HOOK(int, 0, inode_setattr, struct mnt_idmap *idmap, struct dentry *dentry, +LSM_HOOK(int, 0, inode_setattr, const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) -LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) LSM_HOOK(int, 0, inode_getattr, const struct path *path) LSM_HOOK(int, 0, inode_xattr_skipcap, const char *name) -LSM_HOOK(int, 0, inode_setxattr, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_setxattr, const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) LSM_HOOK(void, LSM_RET_VOID, inode_post_setxattr, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) LSM_HOOK(int, 0, inode_getxattr, struct dentry *dentry, const char *name) LSM_HOOK(int, 0, inode_listxattr, struct dentry *dentry) -LSM_HOOK(int, 0, inode_removexattr, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_removexattr, const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) LSM_HOOK(void, LSM_RET_VOID, inode_post_removexattr, struct dentry *dentry, const char *name) LSM_HOOK(int, 0, inode_file_setattr, struct dentry *dentry, struct file_kattr *fa) LSM_HOOK(int, 0, inode_file_getattr, struct dentry *dentry, struct file_kattr *fa) -LSM_HOOK(int, 0, inode_set_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_set_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) LSM_HOOK(void, LSM_RET_VOID, inode_post_set_acl, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) -LSM_HOOK(int, 0, inode_get_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_get_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) -LSM_HOOK(int, 0, inode_remove_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_remove_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) -LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) LSM_HOOK(int, 0, inode_need_killpriv, struct dentry *dentry) -LSM_HOOK(int, 0, inode_killpriv, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_killpriv, const struct mnt_idmap *idmap, struct dentry *dentry) -LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, struct mnt_idmap *idmap, +LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) LSM_HOOK(int, -EOPNOTSUPP, inode_setsecurity, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/include/linux/mnt_idmapping.h b/include/linux/mnt_idmapping.h index e71a6070a8f8..78eeef4c2996 100644 --- a/include/linux/mnt_idmapping.h +++ b/include/linux/mnt_idmapping.h @@ -8,8 +8,8 @@ struct mnt_idmap; struct user_namespace; -extern struct mnt_idmap nop_mnt_idmap; -extern struct mnt_idmap invalid_mnt_idmap; +extern const struct mnt_idmap nop_mnt_idmap; +extern const struct mnt_idmap invalid_mnt_idmap; extern struct user_namespace init_user_ns; typedef struct { @@ -121,19 +121,19 @@ static inline bool vfsgid_eq_kgid(vfsgid_t vfsgid, kgid_t kgid) int vfsgid_in_group_p(vfsgid_t vfsgid); -struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap); -void mnt_idmap_put(struct mnt_idmap *idmap); +const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap); +void mnt_idmap_put(const struct mnt_idmap *idmap); -vfsuid_t make_vfsuid(struct mnt_idmap *idmap, +vfsuid_t make_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kuid_t kuid); -vfsgid_t make_vfsgid(struct mnt_idmap *idmap, +vfsgid_t make_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kgid_t kgid); -kuid_t from_vfsuid(struct mnt_idmap *idmap, +kuid_t from_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid); -kgid_t from_vfsgid(struct mnt_idmap *idmap, +kgid_t from_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid); /** @@ -148,7 +148,7 @@ kgid_t from_vfsgid(struct mnt_idmap *idmap, * * Return: true if @vfsuid has a mapping in the filesystem, false if not. */ -static inline bool vfsuid_has_fsmapping(struct mnt_idmap *idmap, +static inline bool vfsuid_has_fsmapping(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid) { @@ -186,7 +186,7 @@ static inline kuid_t vfsuid_into_kuid(vfsuid_t vfsuid) * * Return: true if @vfsgid has a mapping in the filesystem, false if not. */ -static inline bool vfsgid_has_fsmapping(struct mnt_idmap *idmap, +static inline bool vfsgid_has_fsmapping(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid) { @@ -225,7 +225,7 @@ static inline kgid_t vfsgid_into_kgid(vfsgid_t vfsgid) * * Return: the caller's current fsuid mapped up according to @idmap. */ -static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap, +static inline kuid_t mapped_fsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns) { return from_vfsuid(idmap, fs_userns, VFSUIDT_INIT(current_fsuid())); @@ -244,7 +244,7 @@ static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap, * * Return: the caller's current fsgid mapped up according to @idmap. */ -static inline kgid_t mapped_fsgid(struct mnt_idmap *idmap, +static inline kgid_t mapped_fsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns) { return from_vfsgid(idmap, fs_userns, VFSGIDT_INIT(current_fsgid())); diff --git a/include/linux/mount.h b/include/linux/mount.h index acfe7ef86a1b..e90ccafef281 100644 --- a/include/linux/mount.h +++ b/include/linux/mount.h @@ -59,10 +59,10 @@ struct vfsmount { struct dentry *mnt_root; /* root of the mounted tree */ struct super_block *mnt_sb; /* pointer to superblock */ int mnt_flags; - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; } __randomize_layout; -static inline struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt) +static inline const struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt) { /* Pairs with smp_store_release() in do_idmap_mount(). */ return READ_ONCE(mnt->mnt_idmap); diff --git a/include/linux/namei.h b/include/linux/namei.h index 86d657b24fc6..c4436e5c2ba6 100644 --- a/include/linux/namei.h +++ b/include/linux/namei.h @@ -32,8 +32,9 @@ enum { MAX_NESTED_LINKS = 8 }; #define LOOKUP_CREATE BIT(17) /* ... in object creation */ #define LOOKUP_EXCL BIT(18) /* ... in target must not exist */ #define LOOKUP_RENAME_TARGET BIT(19) /* ... in destination of rename() */ +#define LOOKUP_SHARED BIT(20) /* Parent lock is held shared */ -/* 4 spare bits for intent */ +/* 3 spare bits for intent */ /* Scoping flags for lookup. */ #define LOOKUP_NO_SYMLINKS BIT(24) /* No symlink crossing. */ @@ -70,24 +71,24 @@ extern struct dentry *try_lookup_noperm(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm_unlocked(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm_positive_unlocked(struct qstr *, struct dentry *); -struct dentry *lookup_one(struct mnt_idmap *, struct qstr *, struct dentry *); -struct dentry *lookup_one_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one(const struct mnt_idmap *, struct qstr *, struct dentry *); +struct dentry *lookup_one_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *lookup_one_positive_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *lookup_one_positive_killable(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_killable(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *start_creating(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_creating(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_removing(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_removing(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_creating_killable(struct mnt_idmap *idmap, +struct dentry *start_creating_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_removing_killable(struct mnt_idmap *idmap, +struct dentry *start_removing_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); struct dentry *start_creating_noperm(struct dentry *parent, struct qstr *name); diff --git a/include/linux/netfs.h b/include/linux/netfs.h index b4dd32863dd4..67e010b6994b 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -22,6 +22,7 @@ enum netfs_sreq_ref_trace; typedef struct mempool mempool_t; +struct fscache_occupancy; struct folio_queue; /** @@ -62,8 +63,8 @@ struct netfs_inode { struct fscache_cookie *cache; #endif struct list_head wb_queue; /* Queue of processes wanting to do writeback */ - loff_t _remote_i_size; /* Size of the remote file */ - loff_t _zero_point; /* Size after which we assume there's no data + uoff_t _remote_i_size; /* Size of the remote file */ + uoff_t _zero_point; /* Size after which we assume there's no data * on the server */ spinlock_t lock; /* Lock covering wb_queue */ atomic_t io_count; /* Number of outstanding reqs */ @@ -125,6 +126,12 @@ static inline struct netfs_group *netfs_folio_group(struct folio *folio) return priv; } +enum netfs_cache_collect { + NETFS_CACHE_COLLECT_WRITE_GAP, /* Gap in collection, no state either way */ + NETFS_CACHE_COLLECT_WRITE_DATA, /* Currently collecting good writes */ + NETFS_CACHE_COLLECT_WRITE_CANCEL, /* Currently collecting cancelled writes */ +}; + /* * Stream of I/O subrequests going to a particular destination, such as the * server or the local cache. This is mainly intended for writing where we may @@ -142,7 +149,7 @@ struct netfs_io_stream { void (*issue_write)(struct netfs_io_subrequest *subreq); /* Collection tracking */ struct list_head subrequests; /* Contributory I/O operations */ - unsigned long long collected_to; /* Position we've collected results to */ + uoff_t collected_to; /* Position we've collected results to */ size_t transferred; /* The amount transferred from this stream */ unsigned short error; /* Aggregate error for the stream */ enum netfs_io_source source; /* Where to read from/write to */ @@ -152,6 +159,7 @@ struct netfs_io_stream { bool need_retry; /* T if this stream needs retrying */ bool failed; /* T if this stream failed */ bool transferred_valid; /* T is ->transferred is valid */ + enum netfs_cache_collect cache_collect; /* Current writeback cache collect state */ }; /* @@ -161,8 +169,11 @@ struct netfs_cache_resources { const struct netfs_cache_ops *ops; void *cache_priv; void *cache_priv2; - unsigned int debug_id; /* Cookie debug ID */ + uoff_t cache_i_size; /* Initial size of cache file */ + unsigned int cookie_id; /* Cache cookie debug ID */ + unsigned int object_id; /* Cache object debug ID */ unsigned int inval_counter; /* object->inval_counter at begin_op */ + unsigned int dio_size; /* DIO block size */ }; /* @@ -177,7 +188,7 @@ struct netfs_io_subrequest { struct work_struct work; struct list_head rreq_link; /* Link in rreq->subrequests */ struct iov_iter io_iter; /* Iterator for this subrequest */ - unsigned long long start; /* Where to start the I/O */ + uoff_t start; /* Where to start the I/O */ size_t len; /* Size of the I/O */ size_t transferred; /* Amount of data transferred */ refcount_t ref; @@ -196,6 +207,7 @@ struct netfs_io_subrequest { #define NETFS_SREQ_IN_PROGRESS 8 /* Unlocked when the subrequest completes */ #define NETFS_SREQ_NEED_RETRY 9 /* Set if the filesystem requests a retry */ #define NETFS_SREQ_FAILED 10 /* Set if the subreq failed unretryably */ +#define NETFS_SREQ_CANCELLED 11 /* Set if the subreq was cancelled by netfslib */ }; enum netfs_io_origin { @@ -208,7 +220,6 @@ enum netfs_io_origin { NETFS_DIO_READ, /* This is a direct I/O read */ NETFS_WRITEBACK, /* This write was triggered by writepages */ NETFS_WRITEBACK_SINGLE, /* This monolithic write was triggered by writepages */ - NETFS_WRITETHROUGH, /* This write was made by netfs_perform_write() */ NETFS_UNBUFFERED_WRITE, /* This is an unbuffered write */ NETFS_DIO_WRITE, /* This is a direct I/O write */ NETFS_PGPRIV2_COPY_TO_CACHE, /* [DEPRECATED] This is writing read data to the cache */ @@ -243,17 +254,18 @@ struct netfs_io_request { void *netfs_priv; /* Private data for the netfs */ void *netfs_priv2; /* Private data for the netfs */ struct bio_vec *direct_bv; /* DIO buffer list (when handling iovec-iter) */ - unsigned long long submitted; /* Amount submitted for I/O so far */ - unsigned long long len; /* Length of the request */ + uoff_t submitted; /* Amount submitted for I/O so far */ + uoff_t len; /* Length of the request */ size_t transferred; /* Amount to be indicated as transferred */ size_t progress_at; /* Report read progress when hit this much read */ long error; /* 0 or error that occurred */ - unsigned long long i_size; /* Size of the file */ - unsigned long long start; /* Start position */ + uoff_t i_size; /* Size of the file */ + uoff_t start; /* Start position */ atomic64_t issued_to; /* Write issuer folio cursor */ - unsigned long long collected_to; /* Point we've collected to */ - unsigned long long cleaned_to; /* Position we've cleaned folios to */ - unsigned long long abandon_to; /* Position to abandon folios to */ + uoff_t collected_to; /* Point we've collected to */ + uoff_t cache_coll_to; /* Point the cache has collected to */ + uoff_t cleaned_to; /* Position we've cleaned folios to */ + uoff_t abandon_to; /* Position to abandon folios to */ const struct folio *no_unlock_folio; /* Don't unlock this folio after read */ gfp_t gfp; /* GFP flags to use */ unsigned int direct_bv_count; /* Number of elements in direct_bv[] */ @@ -273,14 +285,18 @@ struct netfs_io_request { #define NETFS_RREQ_FAILED 3 /* The request failed */ #define NETFS_RREQ_RETRYING 4 /* Set if we're in the retry path */ #define NETFS_RREQ_SHORT_TRANSFER 5 /* Set if we have a short transfer */ -#define NETFS_RREQ_OFFLOAD_COLLECTION 8 /* Offload collection to workqueue */ -#define NETFS_RREQ_NO_UNLOCK_FOLIO 9 /* Don't unlock no_unlock_folio on completion */ +#define NETFS_RREQ_CACHE_STOP 8 /* Set to stop caching (ENOBUFS or error) */ +#define NETFS_RREQ_CACHE_ERROR 9 /* Set if we got an error from the cache */ #define NETFS_RREQ_CANCEL_CACHING 10 /* Set to cancel caching */ -#define NETFS_RREQ_UPLOAD_TO_SERVER 11 /* Need to write to the server */ -#define NETFS_RREQ_USE_IO_ITER 12 /* Use ->io_iter rather than ->i_pages */ +#define NETFS_RREQ_OFFLOAD_COLLECTION 12 /* Offload collection to workqueue */ +#define NETFS_RREQ_NO_UNLOCK_FOLIO 13 /* Don't unlock no_unlock_folio on completion */ +#define NETFS_RREQ_UPLOAD_TO_SERVER 14 /* Need to write to the server */ +#define NETFS_RREQ_USE_IO_ITER 15 /* Use ->io_iter rather than ->i_pages */ #define NETFS_RREQ_NEED_PUT_RA_REFS 17 /* Need to put the folio refs RA gave us */ +#ifdef CONFIG_NETFS_PGPRIV2 #define NETFS_RREQ_USE_PGPRIV2 31 /* [DEPRECATED] Use PG_private_2 to mark * write to cache on read */ +#endif const struct netfs_request_ops *netfs_ops; }; @@ -299,12 +315,12 @@ struct netfs_request_ops { int (*prepare_read)(struct netfs_io_subrequest *subreq); void (*issue_read)(struct netfs_io_subrequest *subreq); bool (*is_still_valid)(struct netfs_io_request *rreq); - int (*check_write_begin)(struct file *file, loff_t pos, unsigned len, + int (*check_write_begin)(struct file *file, uoff_t pos, unsigned len, struct folio **foliop, void **_fsdata); void (*done)(struct netfs_io_request *rreq); /* Modification handling */ - void (*update_i_size)(struct inode *inode, loff_t i_size); + void (*update_i_size)(struct inode *inode, uoff_t i_size); void (*post_modify)(struct inode *inode); /* Write request handling */ @@ -332,7 +348,7 @@ struct netfs_cache_ops { /* Read data from the cache */ int (*read)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -340,7 +356,7 @@ struct netfs_cache_ops { /* Write data to the cache */ int (*write)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv); @@ -350,15 +366,14 @@ struct netfs_cache_ops { /* Expand readahead request */ void (*expand_readahead)(struct netfs_cache_resources *cres, - unsigned long long *_start, - unsigned long long *_len, - unsigned long long i_size); + uoff_t *_start, + uoff_t *_len, + uoff_t i_size); /* Prepare a read operation, shortening it to a cached/uncached * boundary as appropriate. */ - enum netfs_io_source (*prepare_read)(struct netfs_io_subrequest *subreq, - unsigned long long i_size); + int (*prepare_read)(struct netfs_io_subrequest *subreq); /* Prepare a write subrequest, working out if we're allowed to do it * and finding out the maximum amount of data to gather before @@ -371,15 +386,24 @@ struct netfs_cache_ops { * actually do. */ int (*prepare_write)(struct netfs_cache_resources *cres, - loff_t *_start, size_t *_len, size_t upper_len, - loff_t i_size, bool no_space_allocated_yet); + uoff_t *_start, size_t *_len, size_t upper_len, + uoff_t i_size, bool no_space_allocated_yet); /* Query the occupancy of the cache in a region, returning where the * next chunk of data starts and how long it is. */ int (*query_occupancy)(struct netfs_cache_resources *cres, - loff_t start, size_t len, size_t granularity, - loff_t *_data_start, size_t *_data_len); + struct fscache_occupancy *occ); + + /* Collect the result of buffered writeback to the cache. This + * includes copying a read to the cache. block_type is one of: + * - NETFS_CACHE_COLLECT_WRITE_DATA for a block of data + * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped + * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a cancellation gap + */ + void (*collect_write)(struct netfs_io_request *wreq, + uoff_t start, size_t len, + enum netfs_cache_collect block_type); }; /* High-level read API. */ @@ -410,7 +434,7 @@ struct readahead_control; void netfs_readahead(struct readahead_control *); int netfs_read_folio(struct file *, struct folio *); int netfs_write_begin(struct netfs_inode *, struct file *, - struct address_space *, loff_t pos, unsigned int len, + struct address_space *, uoff_t pos, unsigned int len, struct folio **, void **fsdata); int netfs_writepages(struct address_space *mapping, struct writeback_control *wbc); @@ -488,10 +512,10 @@ static inline struct netfs_inode *netfs_inode(struct inode *inode) * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_remote_i_size(const struct inode *inode) +static inline uoff_t netfs_read_remote_i_size(const struct inode *inode) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long remote_i_size; + uoff_t remote_i_size; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -526,7 +550,7 @@ static inline unsigned long long netfs_read_remote_i_size(const struct inode *in * spinning forever. */ static inline void netfs_write_remote_i_size(struct inode *inode, - unsigned long long remote_i_size) + uoff_t remote_i_size) { struct netfs_inode *ictx = netfs_inode(inode); @@ -563,10 +587,10 @@ static inline void netfs_write_remote_i_size(struct inode *inode, * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_zero_point(const struct inode *inode) +static inline uoff_t netfs_read_zero_point(const struct inode *inode) { struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long zero_point; + uoff_t zero_point; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -601,7 +625,7 @@ static inline unsigned long long netfs_read_zero_point(const struct inode *inode * forever. */ static inline void netfs_write_zero_point(struct inode *inode, - unsigned long long zero_point) + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -642,9 +666,9 @@ static inline void netfs_write_zero_point(struct inode *inode, * archs it makes no difference if preempt is enabled or not. */ static inline void netfs_read_sizes(const struct inode *inode, - unsigned long long *i_size, - unsigned long long *remote_i_size, - unsigned long long *zero_point) + uoff_t *i_size, + uoff_t *remote_i_size, + uoff_t *zero_point) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); #if BITS_PER_LONG==32 && defined(CONFIG_SMP) @@ -690,9 +714,9 @@ static inline void netfs_read_sizes(const struct inode *inode, * forever. */ static inline void netfs_write_sizes(struct inode *inode, - unsigned long long i_size, - unsigned long long remote_i_size, - unsigned long long zero_point) + uoff_t i_size, + uoff_t remote_i_size, + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -760,7 +784,7 @@ static inline void netfs_inode_init(struct netfs_inode *ctx, * Inform the netfs lib that a file got resized so that it can adjust its state. */ static inline void netfs_resize_file(struct netfs_inode *ictx, - unsigned long long new_i_size, + uoff_t new_i_size, bool changed_on_server) { #if BITS_PER_LONG==32 && defined(CONFIG_SMP) diff --git a/include/linux/nfs_fs.h b/include/linux/nfs_fs.h index b85a73ae7919..d2c716322c6f 100644 --- a/include/linux/nfs_fs.h +++ b/include/linux/nfs_fs.h @@ -437,11 +437,11 @@ extern int nfs_refresh_inode(struct inode *, struct nfs_fattr *); extern int nfs_post_op_update_inode(struct inode *inode, struct nfs_fattr *fattr); extern int nfs_post_op_update_inode_force_wcc(struct inode *inode, struct nfs_fattr *fattr); extern int nfs_post_op_update_inode_force_wcc_locked(struct inode *inode, struct nfs_fattr *fattr); -extern int nfs_getattr(struct mnt_idmap *, const struct path *, +extern int nfs_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void nfs_access_add_cache(struct inode *, struct nfs_access_entry *, const struct cred *); extern void nfs_access_set_mask(struct nfs_access_entry *, u32); -extern int nfs_permission(struct mnt_idmap *, struct inode *, int); +extern int nfs_permission(const struct mnt_idmap *, struct inode *, int); extern int nfs_open(struct inode *, struct file *); extern int nfs_attribute_cache_expired(struct inode *inode); extern int nfs_revalidate_inode(struct inode *inode, unsigned long flags); @@ -450,7 +450,7 @@ extern int nfs_clear_invalid_mapping(struct address_space *mapping); extern bool nfs_mapping_need_revalidate_inode(struct inode *inode); extern int nfs_revalidate_mapping(struct inode *inode, struct address_space *mapping); extern int nfs_revalidate_mapping_rcu(struct inode *inode); -extern int nfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +extern int nfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern void nfs_setattr_update_inode(struct inode *inode, struct iattr *attr, struct nfs_fattr *); extern void nfs_setsecurity(struct inode *inode, struct nfs_fattr *fattr); extern struct nfs_open_context *get_nfs_open_context(struct nfs_open_context *ctx); diff --git a/include/linux/posix_acl.h b/include/linux/posix_acl.h index 62d497763e25..caf500bed993 100644 --- a/include/linux/posix_acl.h +++ b/include/linux/posix_acl.h @@ -74,20 +74,20 @@ extern int __posix_acl_create(struct posix_acl **, gfp_t, umode_t *); extern int __posix_acl_chmod(struct posix_acl **, gfp_t, umode_t); extern struct posix_acl *get_posix_acl(struct inode *, int); -int set_posix_acl(struct mnt_idmap *, struct dentry *, int, +int set_posix_acl(const struct mnt_idmap *, struct dentry *, int, struct posix_acl *); struct posix_acl *get_cached_acl_rcu(struct inode *inode, int type); struct posix_acl *posix_acl_clone(const struct posix_acl *acl, gfp_t flags); #ifdef CONFIG_FS_POSIX_ACL -int posix_acl_chmod(struct mnt_idmap *, struct dentry *, umode_t); +int posix_acl_chmod(const struct mnt_idmap *, struct dentry *, umode_t); extern int posix_acl_create(struct inode *, umode_t *, struct posix_acl **, struct posix_acl **); -int posix_acl_update_mode(struct mnt_idmap *, struct inode *, umode_t *, +int posix_acl_update_mode(const struct mnt_idmap *, struct inode *, umode_t *, struct posix_acl **); -int simple_set_acl(struct mnt_idmap *, struct dentry *, +int simple_set_acl(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); extern int simple_acl_create(struct inode *, struct inode *); @@ -96,7 +96,7 @@ void set_cached_acl(struct inode *inode, int type, struct posix_acl *acl); void forget_cached_acl(struct inode *inode, int type); void forget_all_cached_acls(struct inode *inode); int posix_acl_valid(struct user_namespace *, const struct posix_acl *); -int posix_acl_permission(struct mnt_idmap *, struct inode *, +int posix_acl_permission(const struct mnt_idmap *, struct inode *, const struct posix_acl *, int); static inline void cache_no_acl(struct inode *inode) @@ -105,16 +105,16 @@ static inline void cache_no_acl(struct inode *inode) inode->i_default_acl = NULL; } -int vfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); -struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -int vfs_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); int posix_acl_listxattr(struct inode *inode, char **buffer, ssize_t *remaining_size); #else -static inline int posix_acl_chmod(struct mnt_idmap *idmap, +static inline int posix_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode) { return 0; @@ -141,21 +141,21 @@ static inline void forget_all_cached_acls(struct inode *inode) { } -static inline int vfs_set_acl(struct mnt_idmap *idmap, +static inline int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, struct posix_acl *acl) { return -EOPNOTSUPP; } -static inline struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +static inline struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return ERR_PTR(-EOPNOTSUPP); } -static inline int vfs_remove_acl(struct mnt_idmap *idmap, +static inline int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return -EOPNOTSUPP; diff --git a/include/linux/quotaops.h b/include/linux/quotaops.h index f9c0f9d7c9d9..0c64ca674e77 100644 --- a/include/linux/quotaops.h +++ b/include/linux/quotaops.h @@ -20,7 +20,7 @@ static inline struct quota_info *sb_dqopt(struct super_block *sb) } /* i_rwsem must being held */ -static inline bool is_quota_modification(struct mnt_idmap *idmap, +static inline bool is_quota_modification(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *ia) { return ((ia->ia_valid & ATTR_SIZE) || @@ -109,7 +109,7 @@ int dquot_set_dqblk(struct super_block *sb, struct kqid id, struct qc_dqblk *di); int __dquot_transfer(struct inode *inode, struct dquot **transfer_to); -int dquot_transfer(struct mnt_idmap *idmap, struct inode *inode, +int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr); static inline struct mem_dqinfo *sb_dqinfo(struct super_block *sb, int type) @@ -229,7 +229,7 @@ static inline void dquot_free_inode(struct inode *inode) { } -static inline int dquot_transfer(struct mnt_idmap *idmap, +static inline int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr) { return 0; diff --git a/include/linux/sched.h b/include/linux/sched.h index d35ae49a991f..78bfc0e32e56 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1817,7 +1817,7 @@ extern struct pid __rcu *cad_pid; * I am cleaning dirty pages from some other bdi. */ #define PF_KTHREAD 0x00200000 /* I am a kernel thread */ #define PF_RANDOMIZE 0x00400000 /* Randomize virtual address space */ -#define PF__HOLE__00800000 0x00800000 +#define PF_NO_NOTIFY_SIGNAL 0x00800000 /* see no_notify_signal_save() */ #define PF__HOLE__01000000 0x01000000 #define PF__HOLE__02000000 0x02000000 #define PF_NO_SETAFFINITY 0x04000000 /* Userland is not allowed to meddle with cpus_mask */ diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h index d45a5476b97d..d9419dc902f6 100644 --- a/include/linux/sched/signal.h +++ b/include/linux/sched/signal.h @@ -2,6 +2,7 @@ #ifndef _LINUX_SCHED_SIGNAL_H #define _LINUX_SCHED_SIGNAL_H +#include <linux/cleanup.h> #include <linux/rculist.h> #include <linux/signal.h> #include <linux/sched.h> @@ -79,9 +80,9 @@ struct core_thread { }; struct core_state { - atomic_t nr_threads; - struct core_thread dumper; - struct completion startup; + /* Threads the dumper still waits for. */ + atomic_t threads_remaining; + struct core_thread *tasks; }; /* @@ -384,14 +385,36 @@ static inline int task_sigpending(struct task_struct *p) return unlikely(test_tsk_thread_flag(p,TIF_SIGPENDING)); } +/* Prevent TIF_NOTIFY_SIGNAL from interrupting this task. */ +static inline unsigned int no_notify_signal_save(void) +{ + unsigned int flags = current->flags; + + current->flags |= PF_NO_NOTIFY_SIGNAL; + return flags; +} + +/* Restore the previous PF_NO_NOTIFY_SIGNAL state. */ +static inline void no_notify_signal_restore(unsigned int flags) +{ + current_restore_flags(flags, PF_NO_NOTIFY_SIGNAL); +} + +DEFINE_LOCK_GUARD_0(no_notify_signal, + _T->flags = no_notify_signal_save(), + no_notify_signal_restore(_T->flags), + unsigned int flags) + static inline int signal_pending(struct task_struct *p) { /* * TIF_NOTIFY_SIGNAL isn't really a signal, but it requires the same * behavior in terms of ensuring that we break out of wait loops - * so that notify signal callbacks can be processed. + * so that notify signal callbacks can be processed. Not for a task + * that asked not to be interrupted by it, see no_notify_signal_save(). */ - if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL))) + if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL)) && + likely(!(READ_ONCE(p->flags) & PF_NO_NOTIFY_SIGNAL))) return 1; return task_sigpending(p); } diff --git a/include/linux/security.h b/include/linux/security.h index 153e9043058f..f7ff72ff956b 100644 --- a/include/linux/security.h +++ b/include/linux/security.h @@ -185,11 +185,11 @@ extern int cap_capset(struct cred *new, const struct cred *old, extern int cap_bprm_creds_from_file(struct linux_binprm *bprm, const struct file *file); int cap_inode_setxattr(struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int cap_inode_removexattr(struct mnt_idmap *idmap, +int cap_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name); int cap_inode_need_killpriv(struct dentry *dentry); -int cap_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry); -int cap_inode_getsecurity(struct mnt_idmap *idmap, +int cap_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry); +int cap_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc); extern int cap_mmap_addr(unsigned long addr); @@ -338,6 +338,7 @@ int security_binder_transfer_file(const struct cred *from, const struct cred *to, const struct file *file); int security_ptrace_access_check(struct task_struct *child, unsigned int mode); int security_ptrace_traceme(struct task_struct *parent); +int security_mem_foll_force(const struct cred *subject, bool opened_by_owner); int security_capget(const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, @@ -405,7 +406,7 @@ int security_inode_init_security_anon(struct inode *inode, const struct qstr *name, const struct inode *context_inode); int security_inode_create(struct inode *dir, struct dentry *dentry, umode_t mode); -void security_inode_post_create_tmpfile(struct mnt_idmap *idmap, +void security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode); int security_inode_link(struct dentry *old_dentry, struct inode *dir, struct dentry *new_dentry); @@ -422,31 +423,31 @@ int security_inode_readlink(struct dentry *dentry); int security_inode_follow_link(struct dentry *dentry, struct inode *inode, bool rcu); int security_inode_permission(struct inode *inode, int mask); -int security_inode_setattr(struct mnt_idmap *idmap, +int security_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -void security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +void security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid); int security_inode_getattr(const struct path *path); -int security_inode_setxattr(struct mnt_idmap *idmap, +int security_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int security_inode_set_acl(struct mnt_idmap *idmap, +int security_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); void security_inode_post_set_acl(struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); -int security_inode_get_acl(struct mnt_idmap *idmap, +int security_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -int security_inode_remove_acl(struct mnt_idmap *idmap, +int security_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -void security_inode_post_remove_acl(struct mnt_idmap *idmap, +void security_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); void security_inode_post_setxattr(struct dentry *dentry, const char *name, const void *value, size_t size, int flags); int security_inode_getxattr(struct dentry *dentry, const char *name); int security_inode_listxattr(struct dentry *dentry); -int security_inode_removexattr(struct mnt_idmap *idmap, +int security_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name); void security_inode_post_removexattr(struct dentry *dentry, const char *name); int security_inode_file_setattr(struct dentry *dentry, @@ -454,8 +455,8 @@ int security_inode_file_setattr(struct dentry *dentry, int security_inode_file_getattr(struct dentry *dentry, struct file_kattr *fa); int security_inode_need_killpriv(struct dentry *dentry); -int security_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry); -int security_inode_getsecurity(struct mnt_idmap *idmap, +int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry); +int security_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc); int security_inode_setsecurity(struct inode *inode, const char *name, const void *value, size_t size, int flags); @@ -676,6 +677,12 @@ static inline int security_ptrace_traceme(struct task_struct *parent) return cap_ptrace_traceme(parent); } +static inline int security_mem_foll_force(const struct cred *subject, + bool opened_by_owner) +{ + return 0; +} + static inline int security_capget(const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, @@ -910,7 +917,7 @@ static inline int security_inode_create(struct inode *dir, } static inline void -security_inode_post_create_tmpfile(struct mnt_idmap *idmap, struct inode *inode) +security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode) { } static inline int security_inode_link(struct dentry *old_dentry, @@ -979,7 +986,7 @@ static inline int security_inode_permission(struct inode *inode, int mask) return 0; } -static inline int security_inode_setattr(struct mnt_idmap *idmap, +static inline int security_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { @@ -987,7 +994,7 @@ static inline int security_inode_setattr(struct mnt_idmap *idmap, } static inline void -security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) { } @@ -996,14 +1003,14 @@ static inline int security_inode_getattr(const struct path *path) return 0; } -static inline int security_inode_setxattr(struct mnt_idmap *idmap, +static inline int security_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { return cap_inode_setxattr(dentry, name, value, size, flags); } -static inline int security_inode_set_acl(struct mnt_idmap *idmap, +static inline int security_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) @@ -1016,21 +1023,21 @@ static inline void security_inode_post_set_acl(struct dentry *dentry, struct posix_acl *kacl) { } -static inline int security_inode_get_acl(struct mnt_idmap *idmap, +static inline int security_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return 0; } -static inline int security_inode_remove_acl(struct mnt_idmap *idmap, +static inline int security_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return 0; } -static inline void security_inode_post_remove_acl(struct mnt_idmap *idmap, +static inline void security_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { } @@ -1050,7 +1057,7 @@ static inline int security_inode_listxattr(struct dentry *dentry) return 0; } -static inline int security_inode_removexattr(struct mnt_idmap *idmap, +static inline int security_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { @@ -1078,13 +1085,13 @@ static inline int security_inode_need_killpriv(struct dentry *dentry) return cap_inode_need_killpriv(dentry); } -static inline int security_inode_killpriv(struct mnt_idmap *idmap, +static inline int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry) { return cap_inode_killpriv(idmap, dentry); } -static inline int security_inode_getsecurity(struct mnt_idmap *idmap, +static inline int security_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) @@ -2085,7 +2092,7 @@ int security_path_mkdir(const struct path *dir, struct dentry *dentry, umode_t m int security_path_rmdir(const struct path *dir, struct dentry *dentry); int security_path_mknod(const struct path *dir, struct dentry *dentry, umode_t mode, unsigned int dev); -void security_path_post_mknod(struct mnt_idmap *idmap, struct dentry *dentry); +void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry); int security_path_truncate(const struct path *path); int security_path_symlink(const struct path *dir, struct dentry *dentry, const char *old_name); @@ -2120,7 +2127,7 @@ static inline int security_path_mknod(const struct path *dir, struct dentry *den return 0; } -static inline void security_path_post_mknod(struct mnt_idmap *idmap, +static inline void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry) { } diff --git a/include/linux/splice.h b/include/linux/splice.h index 9dec4861d09f..0e6c955dc6ff 100644 --- a/include/linux/splice.h +++ b/include/linux/splice.h @@ -79,8 +79,8 @@ ssize_t add_to_pipe(struct pipe_inode_info *pipe, struct pipe_buffer *buf); ssize_t vfs_splice_read(struct file *in, loff_t *ppos, struct pipe_inode_info *pipe, size_t len, unsigned int flags); -ssize_t splice_direct_to_actor(struct file *file, struct splice_desc *sd, - splice_direct_actor *actor); +ssize_t vfs_splice_to_actor(struct file *file, loff_t pos, size_t count, + splice_direct_actor *actor, void *private); ssize_t do_splice(struct file *in, loff_t *off_in, struct file *out, loff_t *off_out, size_t len, unsigned int flags); ssize_t do_splice_direct(struct file *in, loff_t *ppos, struct file *out, diff --git a/include/linux/uidgid.h b/include/linux/uidgid.h index 2dc767e08f54..02403629b49f 100644 --- a/include/linux/uidgid.h +++ b/include/linux/uidgid.h @@ -130,9 +130,9 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid) return from_kgid(ns, gid) != (gid_t) -1; } -u32 map_id_down(struct uid_gid_map *map, u32 id); -u32 map_id_up(struct uid_gid_map *map, u32 id); -u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count); +u32 map_id_down(const struct uid_gid_map *map, u32 id); +u32 map_id_up(const struct uid_gid_map *map, u32 id); +u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count); #else @@ -182,17 +182,17 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid) return gid_valid(gid); } -static inline u32 map_id_down(struct uid_gid_map *map, u32 id) +static inline u32 map_id_down(const struct uid_gid_map *map, u32 id) { return id; } -static inline u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count) +static inline u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count) { return id; } -static inline u32 map_id_up(struct uid_gid_map *map, u32 id) +static inline u32 map_id_up(const struct uid_gid_map *map, u32 id) { return id; } diff --git a/include/linux/user_namespace.h b/include/linux/user_namespace.h index e38d9e60569f..91232053775d 100644 --- a/include/linux/user_namespace.h +++ b/include/linux/user_namespace.h @@ -29,8 +29,8 @@ struct uid_gid_map { /* 64 bytes -- 1 cache line */ u32 nr_extents; }; struct { - struct uid_gid_extent *forward; - struct uid_gid_extent *reverse; + struct uid_gid_extent *forward __counted_by_ptr(nr_extents); + struct uid_gid_extent *reverse __counted_by_ptr(nr_extents); }; }; }; @@ -207,6 +207,13 @@ extern bool in_userns(const struct user_namespace *ancestor, const struct user_namespace *child); extern bool current_in_userns(const struct user_namespace *target_ns); struct ns_common *ns_get_owner(struct ns_common *ns); + +#if IS_ENABLED(CONFIG_KUNIT) +extern int uid_gid_map_insert_extent(struct uid_gid_map *map, + struct uid_gid_extent *extent); +extern int uid_gid_map_sort(struct uid_gid_map *map); +#endif /* CONFIG_KUNIT */ + #else static inline struct user_namespace *get_user_ns(struct user_namespace *ns) diff --git a/include/linux/wait_bit.h b/include/linux/wait_bit.h index 553d7b23e3ad..af077ed4caf6 100644 --- a/include/linux/wait_bit.h +++ b/include/linux/wait_bit.h @@ -433,6 +433,32 @@ do { \ }) /** + * wait_var_event_state - wait for a variable to be updated and notified + * @var: the address of variable being waited on + * @condition: the condition to wait for + * @state: the task state to sleep in, %TASK_UNINTERRUPTIBLE etc. + * + * Wait for a @condition to be true, only re-checking when a wake up is + * received for the given @var (an arbitrary kernel address which need + * not be directly related to the given condition, but usually is). + * + * Returns 0 if the condition became true, or %-ERESTARTSYS if a signal + * arrived which @state allows to interrupt. + * + * The condition should normally use smp_load_acquire() or a similarly + * ordered access to ensure that any changes to memory made before the + * condition became true will be visible after the wait completes. + */ +#define wait_var_event_state(var, condition, state) \ +({ \ + int __ret = 0; \ + might_sleep(); \ + if (!(condition)) \ + __ret = ___wait_var_event(var, condition, (state), 0, 0, schedule()); \ + __ret; \ +}) + +/** * wait_var_event_any_lock - wait for a variable to be updated under a lock * @var: the address of the variable being waited on * @condition: condition to wait for diff --git a/include/linux/xattr.h b/include/linux/xattr.h index 54ac3cbc133f..4cc4257de084 100644 --- a/include/linux/xattr.h +++ b/include/linux/xattr.h @@ -47,7 +47,7 @@ struct xattr_handler { struct inode *inode, const char *name, void *buffer, size_t size); int (*set)(const struct xattr_handler *, - struct mnt_idmap *idmap, struct dentry *dentry, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *buffer, size_t size, int flags); }; @@ -77,25 +77,25 @@ struct xattr { }; ssize_t __vfs_getxattr(struct dentry *, struct inode *, const char *, void *, size_t); -ssize_t vfs_getxattr(struct mnt_idmap *, struct dentry *, const char *, +ssize_t vfs_getxattr(const struct mnt_idmap *, struct dentry *, const char *, void *, size_t); ssize_t vfs_listxattr(struct dentry *d, char *list, size_t size); -int __vfs_setxattr(struct mnt_idmap *, struct dentry *, struct inode *, +int __vfs_setxattr(const struct mnt_idmap *, struct dentry *, struct inode *, const char *, const void *, size_t, int); -int __vfs_setxattr_noperm(struct mnt_idmap *, struct dentry *, +int __vfs_setxattr_noperm(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int); -int __vfs_setxattr_locked(struct mnt_idmap *, struct dentry *, +int __vfs_setxattr_locked(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int, struct delegated_inode *); -int vfs_setxattr(struct mnt_idmap *, struct dentry *, const char *, +int vfs_setxattr(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int); -int __vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *); -int __vfs_removexattr_locked(struct mnt_idmap *, struct dentry *, +int __vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *); +int __vfs_removexattr_locked(const struct mnt_idmap *, struct dentry *, const char *, struct delegated_inode *); -int vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *); +int vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *); ssize_t generic_listxattr(struct dentry *dentry, char *buffer, size_t buffer_size); -int vfs_getxattr_alloc(struct mnt_idmap *idmap, +int vfs_getxattr_alloc(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, char **xattr_value, size_t size, gfp_t flags); diff --git a/include/trace/events/cachefiles.h b/include/trace/events/cachefiles.h index e3101410e8b2..a19233e8ae78 100644 --- a/include/trace/events/cachefiles.h +++ b/include/trace/events/cachefiles.h @@ -52,6 +52,8 @@ enum cachefiles_coherency_trace { cachefiles_coherency_check_ok, cachefiles_coherency_check_type, cachefiles_coherency_check_xattr, + cachefiles_coherency_discontiguous, + cachefiles_coherency_remove, cachefiles_coherency_set_fail, cachefiles_coherency_set_ok, cachefiles_coherency_vol_check_cmp, @@ -63,9 +65,11 @@ enum cachefiles_coherency_trace { }; enum cachefiles_trunc_trace { + cachefiles_trunc_clear_padding, cachefiles_trunc_dio_adjust, cachefiles_trunc_expand_tmpfile, cachefiles_trunc_shrink, + cachefiles_trunc_zap, }; enum cachefiles_prepare_read_trace { @@ -80,11 +84,14 @@ enum cachefiles_prepare_read_trace { }; enum cachefiles_error_trace { + cachefiles_trace_alignment_error, + cachefiles_trace_create_nospace, cachefiles_trace_fallocate_error, cachefiles_trace_getxattr_error, cachefiles_trace_link_error, cachefiles_trace_lookup_error, cachefiles_trace_mkdir_error, + cachefiles_trace_mkdir_nospace, cachefiles_trace_notify_change_error, cachefiles_trace_open_error, cachefiles_trace_read_error, @@ -97,6 +104,8 @@ enum cachefiles_error_trace { cachefiles_trace_trunc_error, cachefiles_trace_unlink_error, cachefiles_trace_write_error, + cachefiles_trace_write_nospace, + cachefiles_trace_write_nospace_2, }; #endif @@ -136,6 +145,8 @@ enum cachefiles_error_trace { EM(cachefiles_coherency_check_ok, "OK ") \ EM(cachefiles_coherency_check_type, "BAD type") \ EM(cachefiles_coherency_check_xattr, "BAD xatt") \ + EM(cachefiles_coherency_discontiguous, "--- gap ") \ + EM(cachefiles_coherency_remove, "REMOVE ") \ EM(cachefiles_coherency_set_fail, "SET fail") \ EM(cachefiles_coherency_set_ok, "SET ok ") \ EM(cachefiles_coherency_vol_check_cmp, "VOL BAD cmp ") \ @@ -146,9 +157,11 @@ enum cachefiles_error_trace { E_(cachefiles_coherency_vol_set_ok, "VOL SET ok ") #define cachefiles_trunc_traces \ + EM(cachefiles_trunc_clear_padding, "CLRPAD") \ EM(cachefiles_trunc_dio_adjust, "DIOADJ") \ EM(cachefiles_trunc_expand_tmpfile, "EXPTMP") \ - E_(cachefiles_trunc_shrink, "SHRINK") + EM(cachefiles_trunc_shrink, "SHRINK") \ + E_(cachefiles_trunc_zap, "ZAP ") #define cachefiles_prepare_read_traces \ EM(cachefiles_trace_read_after_eof, "after-eof ") \ @@ -161,11 +174,14 @@ enum cachefiles_error_trace { E_(cachefiles_trace_read_seek_nxio, "seek-enxio") #define cachefiles_error_traces \ + EM(cachefiles_trace_alignment_error, "align") \ + EM(cachefiles_trace_create_nospace, "create-nospace") \ EM(cachefiles_trace_fallocate_error, "fallocate") \ EM(cachefiles_trace_getxattr_error, "getxattr") \ EM(cachefiles_trace_link_error, "link") \ EM(cachefiles_trace_lookup_error, "lookup") \ EM(cachefiles_trace_mkdir_error, "mkdir") \ + EM(cachefiles_trace_mkdir_nospace, "mkdir-nospace") \ EM(cachefiles_trace_notify_change_error, "notify_change") \ EM(cachefiles_trace_open_error, "open") \ EM(cachefiles_trace_read_error, "read") \ @@ -177,7 +193,9 @@ enum cachefiles_error_trace { EM(cachefiles_trace_tmpfile_error, "tmpfile") \ EM(cachefiles_trace_trunc_error, "trunc") \ EM(cachefiles_trace_unlink_error, "unlink") \ - E_(cachefiles_trace_write_error, "write") + EM(cachefiles_trace_write_error, "write") \ + EM(cachefiles_trace_write_nospace, "write-nospace") \ + E_(cachefiles_trace_write_nospace_2, "write-nospace-2") /* @@ -371,12 +389,12 @@ TRACE_EVENT(cachefiles_rename, TRACE_EVENT(cachefiles_coherency, TP_PROTO(struct cachefiles_object *obj, - ino_t ino, + ino_t ino, uoff_t obj_size, const void *disk_aux, enum cachefiles_content content, enum cachefiles_coherency_trace why), - TP_ARGS(obj, ino, disk_aux, content, why), + TP_ARGS(obj, ino, obj_size, disk_aux, content, why), /* Note that obj may be NULL */ TP_STRUCT__entry( @@ -384,6 +402,7 @@ TRACE_EVENT(cachefiles_coherency, __field(enum cachefiles_coherency_trace, why) __field(enum cachefiles_content, content) __field(u64, ino) + __field(u64, obj_size) __field(u64, aux) __field(u64, disk_aux) ), @@ -398,6 +417,7 @@ TRACE_EVENT(cachefiles_coherency, __entry->why = why; __entry->content = content; __entry->ino = ino; + __entry->obj_size = obj_size; __entry->aux = be64_to_cpup((__be64 *)obj->cookie->inline_aux); /* cachefiles_xattr::data is 2-byte aligned but not 8-byte aligned. */ @@ -412,10 +432,11 @@ TRACE_EVENT(cachefiles_coherency, } ), - TP_printk("o=%08x %s B=%llx c=%u aux=%llx dsk=%llx", + TP_printk("o=%08x %s B=%llx oz=%llx c=%u aux=%llx dsk=%llx", __entry->obj, __print_symbolic(__entry->why, cachefiles_coherency_traces), __entry->ino, + __entry->obj_size, __entry->content, __entry->aux, __entry->disk_aux) @@ -449,7 +470,7 @@ TRACE_EVENT(cachefiles_vol_coherency, TRACE_EVENT(cachefiles_prep_read, TP_PROTO(struct cachefiles_object *obj, - loff_t start, + uoff_t start, size_t len, unsigned short flags, enum netfs_io_source source, @@ -464,7 +485,7 @@ TRACE_EVENT(cachefiles_prep_read, __field(enum netfs_io_source, source) __field(enum cachefiles_prepare_read_trace, why) __field(size_t, len) - __field(loff_t, start) + __field(uoff_t, start) __field(unsigned int, netfs_inode) __field(unsigned int, cache_inode) ), @@ -492,16 +513,16 @@ TRACE_EVENT(cachefiles_prep_read, TRACE_EVENT(cachefiles_read, TP_PROTO(struct cachefiles_object *obj, struct inode *backer, - loff_t start, + uoff_t start, size_t len), TP_ARGS(obj, backer, start, len), TP_STRUCT__entry( - __field(unsigned int, obj) - __field(unsigned int, backer) - __field(size_t, len) - __field(loff_t, start) + __field(unsigned int, obj) + __field(unsigned int, backer) + __field(size_t, len) + __field(uoff_t, start) ), TP_fast_assign( @@ -521,16 +542,16 @@ TRACE_EVENT(cachefiles_read, TRACE_EVENT(cachefiles_write, TP_PROTO(struct cachefiles_object *obj, struct inode *backer, - loff_t start, + uoff_t start, size_t len), TP_ARGS(obj, backer, start, len), TP_STRUCT__entry( - __field(unsigned int, obj) - __field(unsigned int, backer) - __field(size_t, len) - __field(loff_t, start) + __field(unsigned int, obj) + __field(unsigned int, backer) + __field(size_t, len) + __field(uoff_t, start) ), TP_fast_assign( @@ -549,7 +570,7 @@ TRACE_EVENT(cachefiles_write, TRACE_EVENT(cachefiles_trunc, TP_PROTO(struct cachefiles_object *obj, struct inode *backer, - loff_t from, loff_t to, enum cachefiles_trunc_trace why), + uoff_t from, uoff_t to, enum cachefiles_trunc_trace why), TP_ARGS(obj, backer, from, to, why), @@ -557,8 +578,8 @@ TRACE_EVENT(cachefiles_trunc, __field(unsigned int, obj) __field(unsigned int, backer) __field(enum cachefiles_trunc_trace, why) - __field(loff_t, from) - __field(loff_t, to) + __field(uoff_t, from) + __field(uoff_t, to) ), TP_fast_assign( @@ -694,6 +715,26 @@ TRACE_EVENT(cachefiles_io_error, __entry->error) ); +TRACE_EVENT(cachefiles_no_space, + TP_PROTO(struct cachefiles_object *obj, enum cachefiles_error_trace trace), + + TP_ARGS(obj, trace), + + TP_STRUCT__entry( + __field(unsigned int, obj) + __field(enum cachefiles_error_trace, trace) + ), + + TP_fast_assign( + __entry->obj = obj ? obj->debug_id : 0; + __entry->trace = trace; + ), + + TP_printk("o=%08x %s", + __entry->obj, + __print_symbolic(__entry->trace, cachefiles_error_traces)) + ); + #endif /* _TRACE_CACHEFILES_H */ /* This part must be outside protection */ diff --git a/include/trace/events/fscache.h b/include/trace/events/fscache.h index f1a73aa83fbb..8735d428ebd9 100644 --- a/include/trace/events/fscache.h +++ b/include/trace/events/fscache.h @@ -460,13 +460,13 @@ TRACE_EVENT(fscache_relinquish, ); TRACE_EVENT(fscache_invalidate, - TP_PROTO(struct fscache_cookie *cookie, loff_t new_size), + TP_PROTO(struct fscache_cookie *cookie, uoff_t new_size), TP_ARGS(cookie, new_size), TP_STRUCT__entry( __field(unsigned int, cookie ) - __field(loff_t, new_size ) + __field(uoff_t, new_size ) ), TP_fast_assign( @@ -479,14 +479,14 @@ TRACE_EVENT(fscache_invalidate, ); TRACE_EVENT(fscache_resize, - TP_PROTO(struct fscache_cookie *cookie, loff_t new_size), + TP_PROTO(struct fscache_cookie *cookie, uoff_t new_size), TP_ARGS(cookie, new_size), TP_STRUCT__entry( __field(unsigned int, cookie ) - __field(loff_t, old_size ) - __field(loff_t, new_size ) + __field(uoff_t, old_size ) + __field(uoff_t, new_size ) ), TP_fast_assign( diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index 3fec3e8f91c8..bf1e1f185b05 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -30,8 +30,7 @@ EM(netfs_write_trace_dio_write, "DIO-WRITE") \ EM(netfs_write_trace_unbuffered_write, "UNB-WRITE") \ EM(netfs_write_trace_writeback, "WRITEBACK") \ - EM(netfs_write_trace_writeback_single, "WB-SINGLE") \ - E_(netfs_write_trace_writethrough, "WRITETHRU") + E_(netfs_write_trace_writeback_single, "WB-SINGLE") #define netfs_rreq_origins \ EM(NETFS_READAHEAD, "RA") \ @@ -43,13 +42,17 @@ EM(NETFS_DIO_READ, "DR") \ EM(NETFS_WRITEBACK, "WB") \ EM(NETFS_WRITEBACK_SINGLE, "W1") \ - EM(NETFS_WRITETHROUGH, "WT") \ EM(NETFS_UNBUFFERED_WRITE, "UW") \ EM(NETFS_DIO_WRITE, "DW") \ E_(NETFS_PGPRIV2_COPY_TO_CACHE, "2C") #define netfs_rreq_traces \ + EM(netfs_rreq_trace_all_queued, "ALL-Q ") \ EM(netfs_rreq_trace_assess, "ASSESS ") \ + EM(netfs_rreq_trace_cache_cancelled, "CA-CNCL") \ + EM(netfs_rreq_trace_cache_failed, "CA-FAIL") \ + EM(netfs_rreq_trace_cache_fail_collect, "CA-F-CO") \ + EM(netfs_rreq_trace_cache_no_space, "CA-NOSP") \ EM(netfs_rreq_trace_collect, "COLLECT") \ EM(netfs_rreq_trace_complete, "COMPLET") \ EM(netfs_rreq_trace_copy, "COPY ") \ @@ -58,11 +61,14 @@ EM(netfs_rreq_trace_end_copy_to_cache, "END-C2C") \ EM(netfs_rreq_trace_free, "FREE ") \ EM(netfs_rreq_trace_intr, "INTR ") \ + EM(netfs_rreq_trace_inval_cache, "INVL-CA") \ EM(netfs_rreq_trace_ki_complete, "KI-CMPL") \ EM(netfs_rreq_trace_ra_put_ref, "RA-PUT ") \ EM(netfs_rreq_trace_recollect, "RECLLCT") \ EM(netfs_rreq_trace_redirty, "REDIRTY") \ EM(netfs_rreq_trace_resubmit, "RESUBMT") \ + EM(netfs_rreq_trace_retry_begin, "RETRY-BEGIN") \ + EM(netfs_rreq_trace_retry_end, "RETRY-END") \ EM(netfs_rreq_trace_set_abandon, "S-ABNDN") \ EM(netfs_rreq_trace_set_pause, "PAUSE ") \ EM(netfs_rreq_trace_unlock, "UNLOCK ") \ @@ -94,8 +100,10 @@ EM(netfs_sreq_trace_abandoned, "ABNDN") \ EM(netfs_sreq_trace_add_donations, "+DON ") \ EM(netfs_sreq_trace_added, "ADD ") \ + EM(netfs_sreq_trace_cache_nofile, "CA-!F") \ EM(netfs_sreq_trace_cache_nowrite, "CA-NW") \ EM(netfs_sreq_trace_cache_prepare, "CA-PR") \ + EM(netfs_sreq_trace_cache_waitfail, "CA-!W") \ EM(netfs_sreq_trace_cache_write, "CA-WR") \ EM(netfs_sreq_trace_cancel, "CANCL") \ EM(netfs_sreq_trace_clear, "CLEAR") \ @@ -134,12 +142,12 @@ #define netfs_failures \ EM(netfs_fail_check_write_begin, "check-write-begin") \ - EM(netfs_fail_copy_to_cache, "copy-to-cache") \ EM(netfs_fail_dio_read_short, "dio-read-short") \ EM(netfs_fail_dio_read_zero, "dio-read-zero") \ EM(netfs_fail_read, "read") \ EM(netfs_fail_short_read, "short-read") \ EM(netfs_fail_prepare_write, "prep-write") \ + EM(netfs_fail_upload, "upload") \ E_(netfs_fail_write, "write") #define netfs_rreq_ref_traces \ @@ -194,11 +202,11 @@ EM(netfs_folio_trace_alloc_buffer, "alloc-buf") \ EM(netfs_folio_trace_cancel_copy, "cancel-copy") \ EM(netfs_folio_trace_cancel_store, "cancel-store") \ - EM(netfs_folio_trace_clear, "clear") \ - EM(netfs_folio_trace_clear_cc, "clear-cc") \ - EM(netfs_folio_trace_clear_g, "clear-g") \ - EM(netfs_folio_trace_clear_s, "clear-s") \ EM(netfs_folio_trace_end_copy, "end-copy") \ + EM(netfs_folio_trace_endwb, "endwb") \ + EM(netfs_folio_trace_endwb_cc, "endwb-cc") \ + EM(netfs_folio_trace_endwb_g, "endwb-g") \ + EM(netfs_folio_trace_endwb_s, "endwb-s") \ EM(netfs_folio_trace_filled_gaps, "filled-gaps") \ EM(netfs_folio_trace_invalidate_all, "inval-all") \ EM(netfs_folio_trace_invalidate_front, "inval-front") \ @@ -223,9 +231,7 @@ EM(netfs_folio_trace_sched_copy, "sched-copy") \ EM(netfs_folio_trace_store, "store") \ EM(netfs_folio_trace_store_copy, "store-copy") \ - EM(netfs_folio_trace_store_plus, "store+") \ - EM(netfs_folio_trace_wthru, "wthru") \ - E_(netfs_folio_trace_wthru_plus, "wthru+") + E_(netfs_folio_trace_store_plus, "store+") #define netfs_collect_contig_traces \ EM(netfs_contig_trace_collect, "Collect") \ @@ -301,7 +307,7 @@ netfs_folioq_traces; TRACE_EVENT(netfs_read, TP_PROTO(struct netfs_io_request *rreq, - loff_t start, size_t len, + uoff_t start, size_t len, enum netfs_read_trace what), TP_ARGS(rreq, start, len, what), @@ -309,8 +315,9 @@ TRACE_EVENT(netfs_read, TP_STRUCT__entry( __field(unsigned int, rreq) __field(unsigned int, cookie) - __field(loff_t, i_size) - __field(loff_t, start) + __field(unsigned int, object) + __field(uoff_t, i_size) + __field(uoff_t, start) __field(size_t, len) __field(enum netfs_read_trace, what) __field(u64, netfs_inode) @@ -318,7 +325,8 @@ TRACE_EVENT(netfs_read, TP_fast_assign( __entry->rreq = rreq->debug_id; - __entry->cookie = rreq->cache_resources.debug_id; + __entry->cookie = rreq->cache_resources.cookie_id; + __entry->object = rreq->cache_resources.object_id; __entry->i_size = rreq->i_size; __entry->start = start; __entry->len = len; @@ -326,10 +334,10 @@ TRACE_EVENT(netfs_read, __entry->netfs_inode = rreq->inode->i_ino; ), - TP_printk("R=%08x %s c=%08x ni=%llx s=%llx l=%zx sz=%llx", + TP_printk("R=%08x %s c=%08x o=%08x ni=%llx s=%llx l=%zx sz=%llx", __entry->rreq, __print_symbolic(__entry->what, netfs_read_traces), - __entry->cookie, + __entry->cookie, __entry->object, __entry->netfs_inode, __entry->start, __entry->len, __entry->i_size) ); @@ -377,7 +385,7 @@ TRACE_EVENT(netfs_sreq, __field(u8, slot) __field(size_t, len) __field(size_t, transferred) - __field(loff_t, start) + __field(uoff_t, start) ), TP_fast_assign( @@ -418,7 +426,7 @@ TRACE_EVENT(netfs_failure, __field(enum netfs_failure, what) __field(size_t, len) __field(size_t, transferred) - __field(loff_t, start) + __field(uoff_t, start) ), TP_fast_assign( @@ -501,6 +509,7 @@ TRACE_EVENT(netfs_folio, TP_STRUCT__entry( __field(u64, ino) __field(pgoff_t, index) + __field(unsigned long, pfn) __field(unsigned int, nr) __field(enum netfs_folio_trace, why) ), @@ -511,9 +520,11 @@ TRACE_EVENT(netfs_folio, __entry->why = why; __entry->index = folio->index; __entry->nr = folio_nr_pages(folio); + __entry->pfn = folio_pfn(folio); ), - TP_printk("i=%05llx ix=%05lx-%05lx %s", + TP_printk("p=%lx i=%05llx ix=%05lx-%05lx %s", + __entry->pfn, __entry->ino, __entry->index, __entry->index + __entry->nr - 1, __print_symbolic(__entry->why, netfs_folio_traces)) ); @@ -524,10 +535,10 @@ TRACE_EVENT(netfs_write_iter, TP_ARGS(iocb, from), TP_STRUCT__entry( - __field(unsigned long long, start) - __field(size_t, len) - __field(unsigned int, flags) - __field(unsigned int, ino) + __field(uoff_t, start) + __field(size_t, len) + __field(unsigned int, flags) + __field(unsigned int, ino) ), TP_fast_assign( @@ -550,27 +561,27 @@ TRACE_EVENT(netfs_write, TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned int, cookie) + __field(unsigned int, object) __field(unsigned int, ino) __field(enum netfs_write_trace, what) - __field(unsigned long long, start) - __field(unsigned long long, len) + __field(uoff_t, start) + __field(uoff_t, len) ), TP_fast_assign( - struct netfs_inode *__ctx = netfs_inode(wreq->inode); - struct fscache_cookie *__cookie = netfs_i_cookie(__ctx); __entry->wreq = wreq->debug_id; - __entry->cookie = __cookie ? __cookie->debug_id : 0; + __entry->cookie = wreq->cache_resources.cookie_id; + __entry->object = wreq->cache_resources.object_id; __entry->ino = wreq->inode->i_ino; __entry->what = what; __entry->start = wreq->start; __entry->len = wreq->len; ), - TP_printk("R=%08x %s c=%08x i=%x by=%llx-%llx", + TP_printk("R=%08x %s c=%08x o=%08x i=%x by=%llx-%llx", __entry->wreq, __print_symbolic(__entry->what, netfs_write_traces), - __entry->cookie, + __entry->cookie, __entry->object, __entry->ino, __entry->start, __entry->start + __entry->len - 1) ); @@ -582,25 +593,26 @@ TRACE_EVENT(netfs_copy2cache, TP_ARGS(rreq, creq), TP_STRUCT__entry( - __field(unsigned int, rreq) - __field(unsigned int, creq) - __field(unsigned int, cookie) - __field(unsigned int, ino) + __field(unsigned int, rreq) + __field(unsigned int, creq) + __field(unsigned int, cookie) + __field(unsigned int, object) + __field(unsigned int, ino) ), TP_fast_assign( - struct netfs_inode *__ctx = netfs_inode(rreq->inode); - struct fscache_cookie *__cookie = netfs_i_cookie(__ctx); __entry->rreq = rreq->debug_id; __entry->creq = creq->debug_id; - __entry->cookie = __cookie ? __cookie->debug_id : 0; + __entry->cookie = rreq->cache_resources.cookie_id; + __entry->object = rreq->cache_resources.object_id; __entry->ino = rreq->inode->i_ino; ), - TP_printk("R=%08x CR=%08x c=%08x i=%x ", + TP_printk("R=%08x CR=%08x c=%08x o=%08x i=%x ", __entry->rreq, __entry->creq, __entry->cookie, + __entry->object, __entry->ino) ); @@ -610,10 +622,10 @@ TRACE_EVENT(netfs_collect, TP_ARGS(wreq), TP_STRUCT__entry( - __field(unsigned int, wreq) - __field(unsigned int, len) - __field(unsigned long long, transferred) - __field(unsigned long long, start) + __field(unsigned int, wreq) + __field(unsigned int, len) + __field(uoff_t, transferred) + __field(uoff_t, start) ), TP_fast_assign( @@ -636,12 +648,12 @@ TRACE_EVENT(netfs_collect_sreq, TP_ARGS(wreq, subreq), TP_STRUCT__entry( - __field(unsigned int, wreq) - __field(unsigned int, subreq) - __field(unsigned int, stream) - __field(unsigned int, len) - __field(unsigned int, transferred) - __field(unsigned long long, start) + __field(unsigned int, wreq) + __field(unsigned int, subreq) + __field(unsigned int, stream) + __field(unsigned int, len) + __field(unsigned int, transferred) + __field(uoff_t, start) ), TP_fast_assign( @@ -660,37 +672,30 @@ TRACE_EVENT(netfs_collect_sreq, TRACE_EVENT(netfs_collect_folio, TP_PROTO(const struct netfs_io_request *wreq, - const struct folio *folio, - unsigned long long fend, - unsigned long long collected_to), + const struct folio *folio), - TP_ARGS(wreq, folio, fend, collected_to), + TP_ARGS(wreq, folio), TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned long, index) - __field(unsigned long long, fend) - __field(unsigned long long, cleaned_to) - __field(unsigned long long, collected_to) + __field(unsigned int, nr) ), TP_fast_assign( __entry->wreq = wreq->debug_id; __entry->index = folio->index; - __entry->fend = fend; - __entry->cleaned_to = wreq->cleaned_to; - __entry->collected_to = collected_to; + __entry->nr = folio_nr_pages(folio); ), - TP_printk("R=%08x ix=%05lx r=%llx-%llx t=%llx/%llx", + TP_printk("R=%08x ix=%05lx-%05lx", __entry->wreq, __entry->index, - (unsigned long long)__entry->index * PAGE_SIZE, __entry->fend, - __entry->cleaned_to, __entry->collected_to) + __entry->index + __entry->nr - 1) ); TRACE_EVENT(netfs_collect_state, TP_PROTO(const struct netfs_io_request *wreq, - unsigned long long collected_to, + uoff_t collected_to, unsigned int notes), TP_ARGS(wreq, collected_to, notes), @@ -698,8 +703,8 @@ TRACE_EVENT(netfs_collect_state, TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned int, notes) - __field(unsigned long long, collected_to) - __field(unsigned long long, cleaned_to) + __field(uoff_t, collected_to) + __field(uoff_t, cleaned_to) ), TP_fast_assign( @@ -718,7 +723,7 @@ TRACE_EVENT(netfs_collect_state, TRACE_EVENT(netfs_collect_gap, TP_PROTO(const struct netfs_io_request *wreq, const struct netfs_io_stream *stream, - unsigned long long jump_to, char type), + uoff_t jump_to, char type), TP_ARGS(wreq, stream, jump_to, type), @@ -726,8 +731,8 @@ TRACE_EVENT(netfs_collect_gap, __field(unsigned int, wreq) __field(unsigned char, stream) __field(unsigned char, type) - __field(unsigned long long, from) - __field(unsigned long long, to) + __field(uoff_t, from) + __field(uoff_t, to) ), TP_fast_assign( @@ -752,8 +757,8 @@ TRACE_EVENT(netfs_collect_stream, TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned char, stream) - __field(unsigned long long, collected_to) - __field(unsigned long long, issued_to) + __field(uoff_t, collected_to) + __field(uoff_t, issued_to) ), TP_fast_assign( diff --git a/include/uapi/linux/close_range.h b/include/uapi/linux/close_range.h index 2d804281554c..7da9ed95258a 100644 --- a/include/uapi/linux/close_range.h +++ b/include/uapi/linux/close_range.h @@ -2,11 +2,34 @@ #ifndef _UAPI_LINUX_CLOSE_RANGE_H #define _UAPI_LINUX_CLOSE_RANGE_H -/* Unshare the file descriptor table before closing file descriptors. */ -#define CLOSE_RANGE_UNSHARE (1U << 1) +/* + * A macro of one of these names defined before this header is parsed, by + * a libc or by a program's own fallback, would replace the enumerator. + */ +#undef CLOSE_RANGE_UNSHARE +#undef CLOSE_RANGE_CLOEXEC +#undef CLOSE_RANGE_EXCEPT +#undef CLOSE_RANGE_CLOEXEC_ONLY -/* Set the FD_CLOEXEC bit instead of closing the file descriptor. */ -#define CLOSE_RANGE_CLOEXEC (1U << 2) +enum close_range_flags { + /* Unshare the file descriptor table before closing file descriptors. */ + CLOSE_RANGE_UNSHARE = (1U << 1), + + /* Set the FD_CLOEXEC bit instead of closing the file descriptor. */ + CLOSE_RANGE_CLOEXEC = (1U << 2), + + /* Act on every file descriptor outside of the given range instead. */ + CLOSE_RANGE_EXCEPT = (1U << 3), + + /* Only close file descriptors that have the FD_CLOEXEC bit set. */ + CLOSE_RANGE_CLOEXEC_ONLY = (1U << 4), +}; + +/* Keep #ifdef working and let glibc skip its own definitions. */ +#define CLOSE_RANGE_UNSHARE CLOSE_RANGE_UNSHARE +#define CLOSE_RANGE_CLOEXEC CLOSE_RANGE_CLOEXEC +#define CLOSE_RANGE_EXCEPT CLOSE_RANGE_EXCEPT +#define CLOSE_RANGE_CLOEXEC_ONLY CLOSE_RANGE_CLOEXEC_ONLY #endif /* _UAPI_LINUX_CLOSE_RANGE_H */ diff --git a/include/uapi/linux/coredump.h b/include/uapi/linux/coredump.h index dc3789b78af0..6d0c53b534ea 100644 --- a/include/uapi/linux/coredump.h +++ b/include/uapi/linux/coredump.h @@ -11,12 +11,53 @@ * @COREDUMP_USERSPACE: userspace writes coredump * @COREDUMP_REJECT: don't generate coredump * @COREDUMP_WAIT: wait for coredump server + * @COREDUMP_RECORDS: send the coredump as a sequence of records instead of + * as a plain byte stream, see struct coredump_record_header; + * requires COREDUMP_KERNEL + * @COREDUMP_SPARSE: describe the holes in the coredump as zero records + * instead of transferring them; requires COREDUMP_RECORDS + * @COREDUMP_MEMORY_TYPES: dump the memory types in + * coredump_ack->memory_types instead of the ones + * the task selected; requires COREDUMP_KERNEL */ enum { COREDUMP_KERNEL = (1ULL << 0), COREDUMP_USERSPACE = (1ULL << 1), COREDUMP_REJECT = (1ULL << 2), COREDUMP_WAIT = (1ULL << 3), + COREDUMP_RECORDS = (1ULL << 4), + COREDUMP_SPARSE = (1ULL << 5), + COREDUMP_MEMORY_TYPES = (1ULL << 6), +}; + +/** + * coredump memory types + * @COREDUMP_MEMORY_ANON_PRIVATE: anonymous private memory + * @COREDUMP_MEMORY_ANON_SHARED: anonymous shared memory + * @COREDUMP_MEMORY_FILE_PRIVATE: file-backed private memory + * @COREDUMP_MEMORY_FILE_SHARED: file-backed shared memory + * @COREDUMP_MEMORY_ELF_HEADERS: the first page of a file-backed private + * mapping that starts an ELF file + * @COREDUMP_MEMORY_HUGETLB_PRIVATE: hugetlb private memory + * @COREDUMP_MEMORY_HUGETLB_SHARED: hugetlb shared memory + * @COREDUMP_MEMORY_DAX_PRIVATE: DAX private memory + * @COREDUMP_MEMORY_DAX_SHARED: DAX shared memory + * + * A bitmask of memory types a coredump may request to be included. New + * memory type bits must ensure that they do not steal memory from an + * existing one so a coredump server will continue to get the same + * coredumps even if a new bit is introduced. + */ +enum { + COREDUMP_MEMORY_ANON_PRIVATE = (1ULL << 0), + COREDUMP_MEMORY_ANON_SHARED = (1ULL << 1), + COREDUMP_MEMORY_FILE_PRIVATE = (1ULL << 2), + COREDUMP_MEMORY_FILE_SHARED = (1ULL << 3), + COREDUMP_MEMORY_ELF_HEADERS = (1ULL << 4), + COREDUMP_MEMORY_HUGETLB_PRIVATE = (1ULL << 5), + COREDUMP_MEMORY_HUGETLB_SHARED = (1ULL << 6), + COREDUMP_MEMORY_DAX_PRIVATE = (1ULL << 7), + COREDUMP_MEMORY_DAX_SHARED = (1ULL << 8), }; /** @@ -24,17 +65,19 @@ enum { * @size: size of struct coredump_req * @size_ack: known size of struct coredump_ack on this kernel * @mask: supported features + * @memory_types: the memory types the task selected + * @memory_types_mask: the memory types this kernel knows * * When a coredump happens the kernel will connect to the coredump * socket and send a coredump request to the coredump server. The @size * member is set to the size of struct coredump_req and provides a hint * to userspace how much data can be read. Userspace may use MSG_PEEK to * peek the size of struct coredump_req and then choose to consume it in - * one go. Userspace may also simply read a COREDUMP_ACK_SIZE_VER0 + * one go. Userspace may also simply read a COREDUMP_REQ_SIZE_VER0 * request. If the size the kernel sends is larger userspace simply * discards any remaining data. * - * The coredump_req->mask member is set to the currently know features. + * The coredump_req->mask member is set to the currently known features. * Userspace may only set coredump_ack->mask to the bits raised by the * kernel in coredump_req->mask. * @@ -42,15 +85,27 @@ enum { * struct coredump_ack the kernel knows. Userspace may only send up to * coredump_req->size_ack bytes to the kernel and must set * coredump_ack->size accordingly. + * + * @memory_types is set to the default memory types that are included in + * the coredump. This can be overridden by raising bits in + * coredump_ack->memory_types. + * + * @memory_types_mask contains a bitmask of all memory types the kernel + * knows about. A coredump server may only raise bits in + * coredump_ack->memory_types that are raised in + * coredump_req->memory_types_mask. */ struct coredump_req { __u32 size; __u32 size_ack; __u64 mask; + __u64 memory_types; + __u64 memory_types_mask; }; enum { COREDUMP_REQ_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_REQ_SIZE_VER1 = 32U, /* memory_types and memory_types_mask added */ }; /** @@ -58,6 +113,8 @@ enum { * @size: size of the struct * @spare: unused * @mask: features kernel is supposed to use + * @memory_types: memory types to dump, only with COREDUMP_MEMORY_TYPES + * in @mask * * The @size member must be set to the size of struct coredump_ack. It * may never exceed what the kernel returned in coredump_req->size_ack @@ -67,15 +124,30 @@ enum { * The @mask member must be set to the features the coredump server * wants the kernel to use. Only bits the kernel returned in * coredump_req->mask may be set. + * + * If COREDUMP_MEMORY_TYPES is raised in @mask the kernel dumps the + * memory types set in the @memory_types mask. Zero is valid and dumps + * no memory apart from the mappings that are always dumped. + * + * Note that memory a task excluded via MADV_DONTDUMP is always left + * out. A coredump server wanting to add or drop memory types instead of + * outright replacing it should simply copy coredump_req->memory_types + * and then mask off or raise types as needed. + * + * Note that @memory_types must be zero if COREDUMP_MEMORY_TYPES isn't + * raised. COREDUMP_MEMORY_TYPES requires COREDUMP_KERNEL and an ack of + * at least COREDUMP_ACK_SIZE_VER1 bytes. */ struct coredump_ack { __u32 size; __u32 spare; __u64 mask; + __u64 memory_types; }; enum { COREDUMP_ACK_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_ACK_SIZE_VER1 = 24U, /* memory_types added */ }; /** @@ -83,11 +155,12 @@ enum { * * The kernel will place a single byte on the coredump socket. The * markers notify userspace whether the coredump ack succeeded or - * failed. + * failed. After any marker other than COREDUMP_MARK_REQACK the kernel + * closes the connection and no coredump is generated. * * @COREDUMP_MARK_MINSIZE: the provided coredump_ack size was too small * @COREDUMP_MARK_MAXSIZE: the provided coredump_ack size was too big - * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask was invalid + * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask or memory types were invalid * @COREDUMP_MARK_CONFLICTING: the provided coredump_ack mask has conflicting options * @COREDUMP_MARK_REQACK: the coredump request and ack was successful * @__COREDUMP_MARK_MAX: the maximum coredump mark value @@ -101,4 +174,72 @@ enum coredump_mark { __COREDUMP_MARK_MAX = (1U << 31), }; +/** + * enum coredump_record_type - Type of a coredump record + * + * @COREDUMP_RECORD_DATA: the header is followed by ->len bytes of data + * @COREDUMP_RECORD_END: the coredump ends here, the header is not followed + * by any data and no further record is sent + * @COREDUMP_RECORD_ZERO: the header stands for ->len zero bytes and is not + * followed by any data + * @__COREDUMP_RECORD_TYPE_MAX: the maximum coredump record type value + */ +enum coredump_record_type { + COREDUMP_RECORD_DATA = 0U, + COREDUMP_RECORD_END = 1U, + COREDUMP_RECORD_ZERO = 2U, + __COREDUMP_RECORD_TYPE_MAX = (1U << 31), +}; + +/** + * struct coredump_record_header - header of a coredump record + * @size: size of struct coredump_record_header + * @type: one of enum coredump_record_type + * @flags: modifiers for this record + * @offset: offset in the coredump this record starts at + * @len: number of coredump bytes this record accounts for + * + * If the coredump server raises COREDUMP_RECORDS in coredump_ack->mask + * the kernel doesn't send the coredump as a plain byte stream. It sends + * a sequence of records instead. A COREDUMP_RECORD_DATA record is + * followed by @len bytes of actual coredump data. A + * COREDUMP_RECORD_ZERO record is followed by nothing and stands for + * @len zero bytes. A server that didn't raise COREDUMP_SPARSE never + * sees a zero record. Records arrive in order and leave no gaps. So + * @offset is the sum of the @len of all records before it. + * + * The last record is a COREDUMP_RECORD_END record. It is followed by + * nothing. Its @len is zero. Its @offset is the size of the coredump. + * The kernel only sends it once it has written the whole coredump. A + * server that hits end-of-file without having seen an end record must + * treat the coredump as incomplete. + * + * The @size member is set to the size of struct coredump_record_header + * the kernel knows and lets the header grow later. It comes first so it + * can be peeked. Userspace must consume @size bytes and discard + * anything beyond what it knows. It must refuse a @size smaller than + * COREDUMP_RECORD_HEADER_SIZE_VER0. @size covers the header alone. + * @offset and @len count coredump bytes. + * + * The @flags member carries modifiers that change how the record is to + * be interpreted. No flag is defined yet. Userspace must refuse a + * record carrying a flag or a type it doesn't know. Every new record + * type is raised in coredump_req->mask as a feature of its own. A + * server only ever sees the types it asked for. + * + * COREDUMP_RECORDS must be combined with COREDUMP_KERNEL, and + * COREDUMP_SPARSE with COREDUMP_RECORDS. + */ +struct coredump_record_header { + __u32 size; + __u32 type; + __u64 flags; + __u64 offset; + __u64 len; +}; + +enum { + COREDUMP_RECORD_HEADER_SIZE_VER0 = 32U, /* size of first published struct */ +}; + #endif /* _UAPI_LINUX_COREDUMP_H */ diff --git a/include/uapi/linux/fs.h b/include/uapi/linux/fs.h index 34c6f219462a..a46c33692aa2 100644 --- a/include/uapi/linux/fs.h +++ b/include/uapi/linux/fs.h @@ -88,7 +88,7 @@ struct fstrim_range { * We include a length field because some filesystems (vfat) have an identifier * that we do want to expose as a UUID, but doesn't have the standard length. * - * We use a fixed size buffer beacuse this interface will, by fiat, never + * We use a fixed size buffer because this interface will, by fiat, never * support "UUIDs" longer than 16 bytes; we don't want to force all downstream * users to have to deal with that. */ |
