From d0a6c3fd4b4b823857f8e5cebb208b33f39c0d87 Mon Sep 17 00:00:00 2001 From: Krzysztof Kozlowski Date: Wed, 20 Nov 2019 21:43:32 +0800 Subject: [PATCH 001/148] UPSTREAM: fuse: fix Kconfig indentation Adjust indentation from spaces to tab (+optional two spaces) as in coding style with command like: $ sed -e 's/^ /\t/' -i */Kconfig Change-Id: Id5c81e7146e0b5cb8fdfb06fb7c2355c818169c9 Signed-off-by: Krzysztof Kozlowski Signed-off-by: Miklos Szeredi --- fs/fuse/Kconfig | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/fuse/Kconfig b/fs/fuse/Kconfig index 0635cba19971..eb2a585572dc 100644 --- a/fs/fuse/Kconfig +++ b/fs/fuse/Kconfig @@ -34,7 +34,7 @@ config VIRTIO_FS select VIRTIO help The Virtio Filesystem allows guests to mount file systems from the - host. + host. If you want to share files between guests or with the host, answer Y - or M. + or M. From 24455287513836b245c42fde100547edb7cb2741 Mon Sep 17 00:00:00 2001 From: YueHaibing Date: Mon, 11 Nov 2019 20:23:59 +0800 Subject: [PATCH 002/148] UPSTREAM: virtiofs: Fix old-style declaration There expect the 'static' keyword to come first in a declaration, and we get warnings like this with "make W=1": fs/fuse/virtio_fs.c:687:1: warning: 'static' is not at beginning of declaration [-Wold-style-declaration] fs/fuse/virtio_fs.c:692:1: warning: 'static' is not at beginning of declaration [-Wold-style-declaration] fs/fuse/virtio_fs.c:1029:1: warning: 'static' is not at beginning of declaration [-Wold-style-declaration] Change-Id: I225793f08d2189f6854bcbfe10d76435a6d729e6 Signed-off-by: YueHaibing Reviewed-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi --- fs/fuse/virtio_fs.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 97259c6afac8..b15e1380c93b 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -731,12 +731,12 @@ static int virtio_fs_restore(struct virtio_device *vdev) } #endif /* CONFIG_PM_SLEEP */ -const static struct virtio_device_id id_table[] = { +static const struct virtio_device_id id_table[] = { { VIRTIO_ID_FS, VIRTIO_DEV_ANY_ID }, {}, }; -const static unsigned int feature_table[] = {}; +static const unsigned int feature_table[] = {}; static struct virtio_driver virtio_fs_driver = { .driver.name = KBUILD_MODNAME, @@ -1073,7 +1073,7 @@ __releases(fiq->lock) } } -const static struct fuse_iqueue_ops virtio_fs_fiq_ops = { +static const struct fuse_iqueue_ops virtio_fs_fiq_ops = { .wake_forget_and_unlock = virtio_fs_wake_forget_and_unlock, .wake_interrupt_and_unlock = virtio_fs_wake_interrupt_and_unlock, .wake_pending_and_unlock = virtio_fs_wake_pending_and_unlock, From 981abb84565de1f2dadafbf97285f38118b534c8 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 30 Oct 2019 11:07:17 -0400 Subject: [PATCH 003/148] UPSTREAM: virtiofs: Use a common function to send forget Currently we are duplicating logic to send forgets at two places. Consolidate the code by calling one helper function. This also uses virtqueue_add_outbuf() instead of virtqueue_add_sgs(). Former is simpler to call. Change-Id: I8f4fcba8c445487cdcdfb5e8e63c9945cdd364c4 Signed-off-by: Vivek Goyal Reviewed-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi --- fs/fuse/virtio_fs.c | 150 +++++++++++++++++++------------------------- 1 file changed, 63 insertions(+), 87 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index b15e1380c93b..704c1709e350 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -329,17 +329,71 @@ static void virtio_fs_request_dispatch_work(struct work_struct *work) } } +/* + * Returns 1 if queue is full and sender should wait a bit before sending + * next request, 0 otherwise. + */ +static int send_forget_request(struct virtio_fs_vq *fsvq, + struct virtio_fs_forget *forget, + bool in_flight) +{ + struct scatterlist sg; + struct virtqueue *vq; + int ret = 0; + bool notify; + + spin_lock(&fsvq->lock); + if (!fsvq->connected) { + if (in_flight) + dec_in_flight_req(fsvq); + kfree(forget); + goto out; + } + + sg_init_one(&sg, forget, sizeof(*forget)); + vq = fsvq->vq; + dev_dbg(&vq->vdev->dev, "%s\n", __func__); + + ret = virtqueue_add_outbuf(vq, &sg, 1, forget, GFP_ATOMIC); + if (ret < 0) { + if (ret == -ENOMEM || ret == -ENOSPC) { + pr_debug("virtio-fs: Could not queue FORGET: err=%d. Will try later\n", + ret); + list_add_tail(&forget->list, &fsvq->queued_reqs); + schedule_delayed_work(&fsvq->dispatch_work, + msecs_to_jiffies(1)); + if (!in_flight) + inc_in_flight_req(fsvq); + /* Queue is full */ + ret = 1; + } else { + pr_debug("virtio-fs: Could not queue FORGET: err=%d. Dropping it.\n", + ret); + kfree(forget); + if (in_flight) + dec_in_flight_req(fsvq); + } + goto out; + } + + if (!in_flight) + inc_in_flight_req(fsvq); + notify = virtqueue_kick_prepare(vq); + spin_unlock(&fsvq->lock); + + if (notify) + virtqueue_notify(vq); + return ret; +out: + spin_unlock(&fsvq->lock); + return ret; +} + static void virtio_fs_hiprio_dispatch_work(struct work_struct *work) { struct virtio_fs_forget *forget; struct virtio_fs_vq *fsvq = container_of(work, struct virtio_fs_vq, dispatch_work.work); - struct virtqueue *vq = fsvq->vq; - struct scatterlist sg; - struct scatterlist *sgs[] = {&sg}; - bool notify; - int ret; - pr_debug("virtio-fs: worker %s called.\n", __func__); while (1) { spin_lock(&fsvq->lock); @@ -351,43 +405,9 @@ static void virtio_fs_hiprio_dispatch_work(struct work_struct *work) } list_del(&forget->list); - if (!fsvq->connected) { - dec_in_flight_req(fsvq); - spin_unlock(&fsvq->lock); - kfree(forget); - continue; - } - - sg_init_one(&sg, forget, sizeof(*forget)); - - /* Enqueue the request */ - dev_dbg(&vq->vdev->dev, "%s\n", __func__); - ret = virtqueue_add_sgs(vq, sgs, 1, 0, forget, GFP_ATOMIC); - if (ret < 0) { - if (ret == -ENOMEM || ret == -ENOSPC) { - pr_debug("virtio-fs: Could not queue FORGET: err=%d. Will try later\n", - ret); - list_add_tail(&forget->list, - &fsvq->queued_reqs); - schedule_delayed_work(&fsvq->dispatch_work, - msecs_to_jiffies(1)); - } else { - pr_debug("virtio-fs: Could not queue FORGET: err=%d. Dropping it.\n", - ret); - dec_in_flight_req(fsvq); - kfree(forget); - } - spin_unlock(&fsvq->lock); - return; - } - - notify = virtqueue_kick_prepare(vq); spin_unlock(&fsvq->lock); - - if (notify) - virtqueue_notify(vq); - pr_debug("virtio-fs: worker %s dispatched one forget request.\n", - __func__); + if (send_forget_request(fsvq, forget, true)) + return; } } @@ -757,14 +777,9 @@ __releases(fiq->lock) { struct fuse_forget_link *link; struct virtio_fs_forget *forget; - struct scatterlist sg; - struct scatterlist *sgs[] = {&sg}; struct virtio_fs *fs; - struct virtqueue *vq; struct virtio_fs_vq *fsvq; - bool notify; u64 unique; - int ret; link = fuse_dequeue_forget(fiq, 1, NULL); unique = fuse_get_unique(fiq); @@ -786,46 +801,7 @@ __releases(fiq->lock) .nlookup = link->forget_one.nlookup, }; - sg_init_one(&sg, forget, sizeof(*forget)); - - /* Enqueue the request */ - spin_lock(&fsvq->lock); - - if (!fsvq->connected) { - kfree(forget); - spin_unlock(&fsvq->lock); - goto out; - } - - vq = fsvq->vq; - dev_dbg(&vq->vdev->dev, "%s\n", __func__); - - ret = virtqueue_add_sgs(vq, sgs, 1, 0, forget, GFP_ATOMIC); - if (ret < 0) { - if (ret == -ENOMEM || ret == -ENOSPC) { - pr_debug("virtio-fs: Could not queue FORGET: err=%d. Will try later.\n", - ret); - list_add_tail(&forget->list, &fsvq->queued_reqs); - schedule_delayed_work(&fsvq->dispatch_work, - msecs_to_jiffies(1)); - inc_in_flight_req(fsvq); - } else { - pr_debug("virtio-fs: Could not queue FORGET: err=%d. Dropping it.\n", - ret); - kfree(forget); - } - spin_unlock(&fsvq->lock); - goto out; - } - - inc_in_flight_req(fsvq); - notify = virtqueue_kick_prepare(vq); - - spin_unlock(&fsvq->lock); - - if (notify) - virtqueue_notify(vq); -out: + send_forget_request(fsvq, forget, false); kfree(link); } From eb673e0f24a3d54d864ad124bc89922336034106 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 30 Oct 2019 11:07:18 -0400 Subject: [PATCH 004/148] UPSTREAM: virtiofs: Do not send forget request "struct list_head" element We are sending whole of virtio_fs_forget struct to the other end over virtqueue. Other end does not need to see elements like "struct list". That's internal detail of guest kernel. Fix it. Change-Id: If479023788488482d81947f6a29ccaa3bf5fd952 Signed-off-by: Vivek Goyal Reviewed-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi --- fs/fuse/virtio_fs.c | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 704c1709e350..1ff721997ca5 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -48,11 +48,15 @@ struct virtio_fs { unsigned int num_request_queues; /* number of request queues */ }; -struct virtio_fs_forget { +struct virtio_fs_forget_req { struct fuse_in_header ih; struct fuse_forget_in arg; +}; + +struct virtio_fs_forget { /* This request can be temporarily queued on virt queue */ struct list_head list; + struct virtio_fs_forget_req req; }; struct virtio_fs_req_work { @@ -341,6 +345,7 @@ static int send_forget_request(struct virtio_fs_vq *fsvq, struct virtqueue *vq; int ret = 0; bool notify; + struct virtio_fs_forget_req *req = &forget->req; spin_lock(&fsvq->lock); if (!fsvq->connected) { @@ -350,7 +355,7 @@ static int send_forget_request(struct virtio_fs_vq *fsvq, goto out; } - sg_init_one(&sg, forget, sizeof(*forget)); + sg_init_one(&sg, req, sizeof(*req)); vq = fsvq->vq; dev_dbg(&vq->vdev->dev, "%s\n", __func__); @@ -777,6 +782,7 @@ __releases(fiq->lock) { struct fuse_forget_link *link; struct virtio_fs_forget *forget; + struct virtio_fs_forget_req *req; struct virtio_fs *fs; struct virtio_fs_vq *fsvq; u64 unique; @@ -790,14 +796,15 @@ __releases(fiq->lock) /* Allocate a buffer for the request */ forget = kmalloc(sizeof(*forget), GFP_NOFS | __GFP_NOFAIL); + req = &forget->req; - forget->ih = (struct fuse_in_header){ + req->ih = (struct fuse_in_header){ .opcode = FUSE_FORGET, .nodeid = link->forget_one.nodeid, .unique = unique, - .len = sizeof(*forget), + .len = sizeof(*req), }; - forget->arg = (struct fuse_forget_in){ + req->arg = (struct fuse_forget_in){ .nlookup = link->forget_one.nlookup, }; From df5988d130e6d91a23ce161a8a92aefad670cb44 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 30 Oct 2019 11:07:19 -0400 Subject: [PATCH 005/148] UPSTREAM: virtiofs: Use completions while waiting for queue to be drained While we wait for queue to finish draining, use completions instead of usleep_range(). This is better way of waiting for event. Change-Id: I53d083cafbbec2060c3fa16f91eeb3322bb55a59 Signed-off-by: Vivek Goyal Reviewed-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi --- fs/fuse/virtio_fs.c | 39 +++++++++++++++++++++++++++++---------- 1 file changed, 29 insertions(+), 10 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 1ff721997ca5..071764a1c0a0 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -35,6 +35,7 @@ struct virtio_fs_vq { struct fuse_dev *fud; bool connected; long in_flight; + struct completion in_flight_zero; /* No inflight requests */ char name[24]; } ____cacheline_aligned_in_smp; @@ -91,6 +92,8 @@ static inline void dec_in_flight_req(struct virtio_fs_vq *fsvq) { WARN_ON(fsvq->in_flight <= 0); fsvq->in_flight--; + if (!fsvq->in_flight) + complete(&fsvq->in_flight_zero); } static void release_virtio_fs_obj(struct kref *ref) @@ -121,22 +124,23 @@ static void virtio_fs_drain_queue(struct virtio_fs_vq *fsvq) WARN_ON(fsvq->in_flight < 0); /* Wait for in flight requests to finish.*/ - while (1) { - spin_lock(&fsvq->lock); - if (!fsvq->in_flight) { - spin_unlock(&fsvq->lock); - break; - } + spin_lock(&fsvq->lock); + if (fsvq->in_flight) { + /* We are holding virtio_fs_mutex. There should not be any + * waiters waiting for completion. + */ + reinit_completion(&fsvq->in_flight_zero); + spin_unlock(&fsvq->lock); + wait_for_completion(&fsvq->in_flight_zero); + } else { spin_unlock(&fsvq->lock); - /* TODO use completion instead of timeout */ - usleep_range(1000, 2000); } flush_work(&fsvq->done_work); flush_delayed_work(&fsvq->dispatch_work); } -static void virtio_fs_drain_all_queues(struct virtio_fs *fs) +static void virtio_fs_drain_all_queues_locked(struct virtio_fs *fs) { struct virtio_fs_vq *fsvq; int i; @@ -147,6 +151,19 @@ static void virtio_fs_drain_all_queues(struct virtio_fs *fs) } } +static void virtio_fs_drain_all_queues(struct virtio_fs *fs) +{ + /* Provides mutual exclusion between ->remove and ->kill_sb + * paths. We don't want both of these draining queue at the + * same time. Current completion logic reinits completion + * and that means there should not be any other thread + * doing reinit or waiting for completion already. + */ + mutex_lock(&virtio_fs_mutex); + virtio_fs_drain_all_queues_locked(fs); + mutex_unlock(&virtio_fs_mutex); +} + static void virtio_fs_start_all_queues(struct virtio_fs *fs) { struct virtio_fs_vq *fsvq; @@ -627,6 +644,7 @@ static int virtio_fs_setup_vqs(struct virtio_device *vdev, INIT_LIST_HEAD(&fs->vqs[VQ_HIPRIO].end_reqs); INIT_DELAYED_WORK(&fs->vqs[VQ_HIPRIO].dispatch_work, virtio_fs_hiprio_dispatch_work); + init_completion(&fs->vqs[VQ_HIPRIO].in_flight_zero); spin_lock_init(&fs->vqs[VQ_HIPRIO].lock); /* Initialize the requests virtqueues */ @@ -637,6 +655,7 @@ static int virtio_fs_setup_vqs(struct virtio_device *vdev, virtio_fs_request_dispatch_work); INIT_LIST_HEAD(&fs->vqs[i].queued_reqs); INIT_LIST_HEAD(&fs->vqs[i].end_reqs); + init_completion(&fs->vqs[i].in_flight_zero); snprintf(fs->vqs[i].name, sizeof(fs->vqs[i].name), "requests.%u", i - VQ_REQUEST); callbacks[i] = virtio_fs_vq_done; @@ -731,7 +750,7 @@ static void virtio_fs_remove(struct virtio_device *vdev) /* This device is going away. No one should get new reference */ list_del_init(&fs->list); virtio_fs_stop_all_queues(fs); - virtio_fs_drain_all_queues(fs); + virtio_fs_drain_all_queues_locked(fs); vdev->config->reset(vdev); virtio_fs_cleanup_vqs(vdev, fs); From 7c84a9e7c70cb3b17e95d3235ba401f59a5fe5c2 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Mon, 4 May 2020 14:33:15 -0400 Subject: [PATCH 006/148] UPSTREAM: virtiofs: do not use fuse_fill_super_common() for device installation fuse_fill_super_common() allocates and installs one fuse_device. Hence virtiofs allocates and install all fuse devices by itself except one. This makes logic little twisted. There does not seem to be any real need that why virtiofs can't allocate and install all fuse devices itself. So opt out of fuse device allocation and installation while calling fuse_fill_super_common(). Regular fuse still wants fuse_fill_super_common() to install fuse_device. It needs to prevent against races where two mounters are trying to mount fuse using same fd. In that case one will succeed while other will get -EINVAL. virtiofs does not have this issue because sget_fc() resolves the race w.r.t multiple mounters and only one instance of virtio_fs_fill_super() should be in progress for same filesystem. Change-Id: I4c2e65bdbc10e0a22adbc648cdf3fcb34903b880 Signed-off-by: Vivek Goyal Signed-off-by: Miklos Szeredi --- fs/fuse/inode.c | 19 ++++++++++++------- fs/fuse/virtio_fs.c | 9 +++------ 2 files changed, 15 insertions(+), 13 deletions(-) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index b32c5c84004c..ca1dc0a06137 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1163,7 +1163,7 @@ EXPORT_SYMBOL_GPL(fuse_dev_free); int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) { - struct fuse_dev *fud; + struct fuse_dev *fud = NULL; struct fuse_conn *fc = get_fuse_conn_super(sb); struct inode *root; struct dentry *root_dentry; @@ -1205,9 +1205,12 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) if (sb->s_user_ns != &init_user_ns) sb->s_xattr = fuse_no_acl_xattr_handlers; - fud = fuse_dev_alloc_install(fc); - if (!fud) - goto err; + if (ctx->fudptr) { + err = -ENOMEM; + fud = fuse_dev_alloc_install(fc); + if (!fud) + goto err; + } fc->dev = sb->s_dev; fc->sb = sb; @@ -1241,7 +1244,7 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) mutex_lock(&fuse_mutex); err = -EINVAL; - if (*ctx->fudptr) + if (ctx->fudptr && *ctx->fudptr) goto err_unlock; err = fuse_ctl_add_conn(fc); @@ -1250,7 +1253,8 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) list_add_tail(&fc->entry, &fuse_conn_list); sb->s_root = root_dentry; - *ctx->fudptr = fud; + if (ctx->fudptr) + *ctx->fudptr = fud; mutex_unlock(&fuse_mutex); return 0; @@ -1258,7 +1262,8 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) mutex_unlock(&fuse_mutex); dput(root_dentry); err_dev_free: - fuse_dev_free(fud); + if (fud) + fuse_dev_free(fud); err: return err; } diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 071764a1c0a0..73abe6193df4 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1114,7 +1114,7 @@ static int virtio_fs_fill_super(struct super_block *sb) err = -ENOMEM; /* Allocate fuse_dev for hiprio and notification queues */ - for (i = 0; i < VQ_REQUEST; i++) { + for (i = 0; i < fs->nvqs; i++) { struct virtio_fs_vq *fsvq = &fs->vqs[i]; fsvq->fud = fuse_dev_alloc(); @@ -1122,18 +1122,15 @@ static int virtio_fs_fill_super(struct super_block *sb) goto err_free_fuse_devs; } - ctx.fudptr = (void **)&fs->vqs[VQ_REQUEST].fud; + /* virtiofs allocates and installs its own fuse devices */ + ctx.fudptr = NULL; err = fuse_fill_super_common(sb, &ctx); if (err < 0) goto err_free_fuse_devs; - fc = fs->vqs[VQ_REQUEST].fud->fc; - for (i = 0; i < fs->nvqs; i++) { struct virtio_fs_vq *fsvq = &fs->vqs[i]; - if (i == VQ_REQUEST) - continue; /* already initialized */ fuse_dev_install(fsvq->fud, fc); } From 6ac3068f303438d14974f741b502824f46b7e1b3 Mon Sep 17 00:00:00 2001 From: "Michael S. Tsirkin" Date: Wed, 5 Aug 2020 05:39:36 -0400 Subject: [PATCH 007/148] UPSTREAM: virtio_fs: convert to LE accessors Virtio fs is modern-only. Use LE accessors for config space. Change-Id: I24aede8e6e82a175779782ae7f9077020946ae32 Signed-off-by: Michael S. Tsirkin --- fs/fuse/virtio_fs.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 73abe6193df4..389705609f02 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -616,8 +616,8 @@ static int virtio_fs_setup_vqs(struct virtio_device *vdev, unsigned int i; int ret = 0; - virtio_cread(vdev, struct virtio_fs_config, num_request_queues, - &fs->num_request_queues); + virtio_cread_le(vdev, struct virtio_fs_config, num_request_queues, + &fs->num_request_queues); if (fs->num_request_queues == 0) return -EINVAL; From 8350f125f2ca0230379c3453e15073aedc3957d4 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:44 -0400 Subject: [PATCH 008/148] UPSTREAM: virtiofs: provide a helper function for virtqueue initialization This reduces code duplication and make it little easier to read code. Change-Id: I3e24120b0a985658ffa6f09d705cf4925cc5f11d Signed-off-by: Vivek Goyal Signed-off-by: Miklos Szeredi --- fs/fuse/virtio_fs.c | 50 +++++++++++++++++++++++++++------------------ 1 file changed, 30 insertions(+), 20 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 389705609f02..25b132e675fd 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -24,6 +24,8 @@ enum { VQ_REQUEST }; +#define VQ_NAME_LEN 24 + /* Per-virtqueue state */ struct virtio_fs_vq { spinlock_t lock; @@ -36,7 +38,7 @@ struct virtio_fs_vq { bool connected; long in_flight; struct completion in_flight_zero; /* No inflight requests */ - char name[24]; + char name[VQ_NAME_LEN]; } ____cacheline_aligned_in_smp; /* A virtio-fs device instance */ @@ -606,6 +608,26 @@ static void virtio_fs_vq_done(struct virtqueue *vq) schedule_work(&fsvq->done_work); } +static void virtio_fs_init_vq(struct virtio_fs_vq *fsvq, char *name, + int vq_type) +{ + strncpy(fsvq->name, name, VQ_NAME_LEN); + spin_lock_init(&fsvq->lock); + INIT_LIST_HEAD(&fsvq->queued_reqs); + INIT_LIST_HEAD(&fsvq->end_reqs); + init_completion(&fsvq->in_flight_zero); + + if (vq_type == VQ_REQUEST) { + INIT_WORK(&fsvq->done_work, virtio_fs_requests_done_work); + INIT_DELAYED_WORK(&fsvq->dispatch_work, + virtio_fs_request_dispatch_work); + } else { + INIT_WORK(&fsvq->done_work, virtio_fs_hiprio_done_work); + INIT_DELAYED_WORK(&fsvq->dispatch_work, + virtio_fs_hiprio_dispatch_work); + } +} + /* Initialize virtqueues */ static int virtio_fs_setup_vqs(struct virtio_device *vdev, struct virtio_fs *fs) @@ -621,7 +643,7 @@ static int virtio_fs_setup_vqs(struct virtio_device *vdev, if (fs->num_request_queues == 0) return -EINVAL; - fs->nvqs = 1 + fs->num_request_queues; + fs->nvqs = VQ_REQUEST + fs->num_request_queues; fs->vqs = kcalloc(fs->nvqs, sizeof(fs->vqs[VQ_HIPRIO]), GFP_KERNEL); if (!fs->vqs) return -ENOMEM; @@ -635,29 +657,17 @@ static int virtio_fs_setup_vqs(struct virtio_device *vdev, goto out; } + /* Initialize the hiprio/forget request virtqueue */ callbacks[VQ_HIPRIO] = virtio_fs_vq_done; - snprintf(fs->vqs[VQ_HIPRIO].name, sizeof(fs->vqs[VQ_HIPRIO].name), - "hiprio"); + virtio_fs_init_vq(&fs->vqs[VQ_HIPRIO], "hiprio", VQ_HIPRIO); names[VQ_HIPRIO] = fs->vqs[VQ_HIPRIO].name; - INIT_WORK(&fs->vqs[VQ_HIPRIO].done_work, virtio_fs_hiprio_done_work); - INIT_LIST_HEAD(&fs->vqs[VQ_HIPRIO].queued_reqs); - INIT_LIST_HEAD(&fs->vqs[VQ_HIPRIO].end_reqs); - INIT_DELAYED_WORK(&fs->vqs[VQ_HIPRIO].dispatch_work, - virtio_fs_hiprio_dispatch_work); - init_completion(&fs->vqs[VQ_HIPRIO].in_flight_zero); - spin_lock_init(&fs->vqs[VQ_HIPRIO].lock); /* Initialize the requests virtqueues */ for (i = VQ_REQUEST; i < fs->nvqs; i++) { - spin_lock_init(&fs->vqs[i].lock); - INIT_WORK(&fs->vqs[i].done_work, virtio_fs_requests_done_work); - INIT_DELAYED_WORK(&fs->vqs[i].dispatch_work, - virtio_fs_request_dispatch_work); - INIT_LIST_HEAD(&fs->vqs[i].queued_reqs); - INIT_LIST_HEAD(&fs->vqs[i].end_reqs); - init_completion(&fs->vqs[i].in_flight_zero); - snprintf(fs->vqs[i].name, sizeof(fs->vqs[i].name), - "requests.%u", i - VQ_REQUEST); + char vq_name[VQ_NAME_LEN]; + + snprintf(vq_name, VQ_NAME_LEN, "requests.%u", i - VQ_REQUEST); + virtio_fs_init_vq(&fs->vqs[i], vq_name, VQ_REQUEST); callbacks[i] = virtio_fs_vq_done; names[i] = fs->vqs[i].name; } From 623454131d82a599041f5cc8df59cbca065954f2 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:45 -0400 Subject: [PATCH 009/148] UPSTREAM: virtiofs: get rid of no_mount_options This option was introduced so that for virtio_fs we don't show any mounts options fuse_show_options(). Because we don't offer any of these options to be controlled by mounter. Very soon we are planning to introduce option "dax" which mounter should be able to specify. And no_mount_options does not work anymore. Change-Id: Ie1e206608dc662526e2316a5bcbc6312384cc7e5 Signed-off-by: Vivek Goyal Signed-off-by: Miklos Szeredi --- fs/fuse/fuse_i.h | 8 ++++---- fs/fuse/inode.c | 30 ++++++++++++++++-------------- fs/fuse/virtio_fs.c | 1 - 3 files changed, 20 insertions(+), 19 deletions(-) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 2eec57bb3707..4d58c3b84c11 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -502,7 +502,7 @@ struct fuse_fs_context { bool destroy:1; bool no_control:1; bool no_force_umount:1; - bool no_mount_options:1; + bool legacy_opts_show:1; unsigned int max_read; unsigned int blksize; const char *subtype; @@ -630,6 +630,9 @@ struct fuse_conn { /** cache READLINK responses in page cache */ unsigned cache_symlinks:1; + /* show legacy mount options */ + unsigned int legacy_opts_show:1; + /* * The following bitfields are only for optimization purposes * and hence races in setting them will not cause malfunction @@ -737,9 +740,6 @@ struct fuse_conn { /** Do not allow MNT_FORCE umount */ unsigned int no_force_umount:1; - /* Do not show mount options */ - unsigned int no_mount_options:1; - /** Passthrough mode for read/write IO */ unsigned int passthrough:1; diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index ca1dc0a06137..a629c30d4ea6 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -583,19 +583,20 @@ static int fuse_show_options(struct seq_file *m, struct dentry *root) struct super_block *sb = root->d_sb; struct fuse_conn *fc = get_fuse_conn_super(sb); - if (fc->no_mount_options) - return 0; - - seq_printf(m, ",user_id=%u", from_kuid_munged(fc->user_ns, fc->user_id)); - seq_printf(m, ",group_id=%u", from_kgid_munged(fc->user_ns, fc->group_id)); - if (fc->default_permissions) - seq_puts(m, ",default_permissions"); - if (fc->allow_other) - seq_puts(m, ",allow_other"); - if (fc->max_read != ~0) - seq_printf(m, ",max_read=%u", fc->max_read); - if (sb->s_bdev && sb->s_blocksize != FUSE_DEFAULT_BLKSIZE) - seq_printf(m, ",blksize=%lu", sb->s_blocksize); + if (fc->legacy_opts_show) { + seq_printf(m, ",user_id=%u", + from_kuid_munged(fc->user_ns, fc->user_id)); + seq_printf(m, ",group_id=%u", + from_kgid_munged(fc->user_ns, fc->group_id)); + if (fc->default_permissions) + seq_puts(m, ",default_permissions"); + if (fc->allow_other) + seq_puts(m, ",allow_other"); + if (fc->max_read != ~0) + seq_printf(m, ",max_read=%u", fc->max_read); + if (sb->s_bdev && sb->s_blocksize != FUSE_DEFAULT_BLKSIZE) + seq_printf(m, ",blksize=%lu", sb->s_blocksize); + } return 0; } @@ -1227,11 +1228,11 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) fc->allow_other = ctx->allow_other; fc->user_id = ctx->user_id; fc->group_id = ctx->group_id; + fc->legacy_opts_show = ctx->legacy_opts_show; fc->max_read = max_t(unsigned, 4096, ctx->max_read); fc->destroy = ctx->destroy; fc->no_control = ctx->no_control; fc->no_force_umount = ctx->no_force_umount; - fc->no_mount_options = ctx->no_mount_options; err = -ENOMEM; root = fuse_get_root_inode(sb, ctx->rootmode); @@ -1356,6 +1357,7 @@ static int fuse_init_fs_context(struct fs_context *fc) ctx->max_read = ~0; ctx->blksize = FUSE_DEFAULT_BLKSIZE; + ctx->legacy_opts_show = true; #ifdef CONFIG_BLOCK if (fc->fs_type == &fuseblk_fs_type) { diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 25b132e675fd..e4f14fa0f5d9 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1107,7 +1107,6 @@ static int virtio_fs_fill_super(struct super_block *sb) .destroy = true, .no_control = true, .no_force_umount = true, - .no_mount_options = true, }; mutex_lock(&virtio_fs_mutex); From 30fe78a969073ff30e27b60c3011766b87d7889b Mon Sep 17 00:00:00 2001 From: Stefan Hajnoczi Date: Wed, 19 Aug 2020 18:19:46 -0400 Subject: [PATCH 010/148] UPSTREAM: virtiofs: set up virtio_fs dax_device Setup a dax device. Use the shm capability to find the cache entry and map it. The DAX window is accessed by the fs/dax.c infrastructure and must have struct pages (at least on x86). Use devm_memremap_pages() to map the DAX window PCI BAR and allocate struct page. Change-Id: I46c5a56c7add5d3afb9799db8ab0f64d3ea3ee0f Signed-off-by: Stefan Hajnoczi Signed-off-by: Dr. David Alan Gilbert Signed-off-by: Vivek Goyal Signed-off-by: Sebastien Boeuf Signed-off-by: Liu Bo Signed-off-by: Miklos Szeredi --- fs/fuse/virtio_fs.c | 138 +++++++++++++++++++++++++++++++++ include/uapi/linux/virtio_fs.h | 3 + 2 files changed, 141 insertions(+) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index e4f14fa0f5d9..1e533e320779 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -5,12 +5,16 @@ */ #include +#include +#include +#include #include #include #include #include #include #include +#include #include "fuse_i.h" /* List of virtio-fs device instances and a lock for the list. Also provides @@ -49,6 +53,12 @@ struct virtio_fs { struct virtio_fs_vq *vqs; unsigned int nvqs; /* number of virtqueues */ unsigned int num_request_queues; /* number of request queues */ + struct dax_device *dax_dev; + + /* DAX memory window where file contents are mapped */ + void *window_kaddr; + phys_addr_t window_phys_addr; + size_t window_len; }; struct virtio_fs_forget_req { @@ -696,6 +706,130 @@ static void virtio_fs_cleanup_vqs(struct virtio_device *vdev, vdev->config->del_vqs(vdev); } +/* Map a window offset to a page frame number. The window offset will have + * been produced by .iomap_begin(), which maps a file offset to a window + * offset. + */ +static long virtio_fs_direct_access(struct dax_device *dax_dev, pgoff_t pgoff, + long nr_pages, void **kaddr, pfn_t *pfn) +{ + struct virtio_fs *fs = dax_get_private(dax_dev); + phys_addr_t offset = PFN_PHYS(pgoff); + size_t max_nr_pages = fs->window_len/PAGE_SIZE - pgoff; + + if (kaddr) + *kaddr = fs->window_kaddr + offset; + if (pfn) + *pfn = phys_to_pfn_t(fs->window_phys_addr + offset, + PFN_DEV | PFN_MAP); + return nr_pages > max_nr_pages ? max_nr_pages : nr_pages; +} + +static size_t virtio_fs_copy_from_iter(struct dax_device *dax_dev, + pgoff_t pgoff, void *addr, + size_t bytes, struct iov_iter *i) +{ + return copy_from_iter(addr, bytes, i); +} + +static size_t virtio_fs_copy_to_iter(struct dax_device *dax_dev, + pgoff_t pgoff, void *addr, + size_t bytes, struct iov_iter *i) +{ + return copy_to_iter(addr, bytes, i); +} + +static int virtio_fs_zero_page_range(struct dax_device *dax_dev, + pgoff_t pgoff, size_t nr_pages) +{ + long rc; + void *kaddr; + + rc = dax_direct_access(dax_dev, pgoff, nr_pages, &kaddr, NULL); + if (rc < 0) + return rc; + memset(kaddr, 0, nr_pages << PAGE_SHIFT); + dax_flush(dax_dev, kaddr, nr_pages << PAGE_SHIFT); + return 0; +} + +static const struct dax_operations virtio_fs_dax_ops = { + .direct_access = virtio_fs_direct_access, + .copy_from_iter = virtio_fs_copy_from_iter, + .copy_to_iter = virtio_fs_copy_to_iter, + .zero_page_range = virtio_fs_zero_page_range, +}; + +static void virtio_fs_cleanup_dax(void *data) +{ + struct dax_device *dax_dev = data; + + kill_dax(dax_dev); + put_dax(dax_dev); +} + +static int virtio_fs_setup_dax(struct virtio_device *vdev, struct virtio_fs *fs) +{ + struct virtio_shm_region cache_reg; + struct dev_pagemap *pgmap; + bool have_cache; + + if (!IS_ENABLED(CONFIG_FUSE_DAX)) + return 0; + + /* Get cache region */ + have_cache = virtio_get_shm_region(vdev, &cache_reg, + (u8)VIRTIO_FS_SHMCAP_ID_CACHE); + if (!have_cache) { + dev_notice(&vdev->dev, "%s: No cache capability\n", __func__); + return 0; + } + + if (!devm_request_mem_region(&vdev->dev, cache_reg.addr, cache_reg.len, + dev_name(&vdev->dev))) { + dev_warn(&vdev->dev, "could not reserve region addr=0x%llx len=0x%llx\n", + cache_reg.addr, cache_reg.len); + return -EBUSY; + } + + dev_notice(&vdev->dev, "Cache len: 0x%llx @ 0x%llx\n", cache_reg.len, + cache_reg.addr); + + pgmap = devm_kzalloc(&vdev->dev, sizeof(*pgmap), GFP_KERNEL); + if (!pgmap) + return -ENOMEM; + + pgmap->type = MEMORY_DEVICE_FS_DAX; + + /* Ideally we would directly use the PCI BAR resource but + * devm_memremap_pages() wants its own copy in pgmap. So + * initialize a struct resource from scratch (only the start + * and end fields will be used). + */ + pgmap->res = (struct resource){ + .name = "virtio-fs dax window", + .start = (phys_addr_t) cache_reg.addr, + .end = (phys_addr_t) cache_reg.addr + cache_reg.len - 1, + }; + + fs->window_kaddr = devm_memremap_pages(&vdev->dev, pgmap); + if (IS_ERR(fs->window_kaddr)) + return PTR_ERR(fs->window_kaddr); + + fs->window_phys_addr = (phys_addr_t) cache_reg.addr; + fs->window_len = (phys_addr_t) cache_reg.len; + + dev_dbg(&vdev->dev, "%s: window kaddr 0x%px phys_addr 0x%llx len 0x%llx\n", + __func__, fs->window_kaddr, cache_reg.addr, cache_reg.len); + + fs->dax_dev = alloc_dax(fs, NULL, &virtio_fs_dax_ops, 0); + if (IS_ERR(fs->dax_dev)) + return PTR_ERR(fs->dax_dev); + + return devm_add_action_or_reset(&vdev->dev, virtio_fs_cleanup_dax, + fs->dax_dev); +} + static int virtio_fs_probe(struct virtio_device *vdev) { struct virtio_fs *fs; @@ -717,6 +851,10 @@ static int virtio_fs_probe(struct virtio_device *vdev) /* TODO vq affinity */ + ret = virtio_fs_setup_dax(vdev, fs); + if (ret < 0) + goto out_vqs; + /* Bring the device online in case the filesystem is mounted and * requests need to be sent before we return. */ diff --git a/include/uapi/linux/virtio_fs.h b/include/uapi/linux/virtio_fs.h index b02eb2ac3d99..2f64abce781f 100644 --- a/include/uapi/linux/virtio_fs.h +++ b/include/uapi/linux/virtio_fs.h @@ -16,4 +16,7 @@ struct virtio_fs_config { __u32 num_request_queues; } __attribute__((packed)); +/* For the id field in virtio_pci_shm_cap */ +#define VIRTIO_FS_SHMCAP_ID_CACHE 0 + #endif /* _UAPI_LINUX_VIRTIO_FS_H */ From 1b8443d29a5957142a7485c1a2cd4dc70e6484df Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:47 -0400 Subject: [PATCH 011/148] UPSTREAM: virtiofs: add a mount option to enable dax Add a mount option to allow using dax with virtio_fs. Change-Id: Ib5c6f9ff7006ad80803338e8139c65143bff43ac Signed-off-by: Vivek Goyal Signed-off-by: Miklos Szeredi --- fs/fuse/Kconfig | 13 ++++++++ fs/fuse/Makefile | 7 ++-- fs/fuse/dax.c | 36 ++++++++++++++++++++ fs/fuse/fuse_i.h | 14 ++++++++ fs/fuse/inode.c | 18 +++++++++- fs/fuse/virtio_fs.c | 81 +++++++++++++++++++++++++++++++++++++-------- 6 files changed, 151 insertions(+), 18 deletions(-) create mode 100644 fs/fuse/dax.c diff --git a/fs/fuse/Kconfig b/fs/fuse/Kconfig index eb2a585572dc..e3a2a61c5ae9 100644 --- a/fs/fuse/Kconfig +++ b/fs/fuse/Kconfig @@ -38,3 +38,16 @@ config VIRTIO_FS If you want to share files between guests or with the host, answer Y or M. + +config FUSE_DAX + bool "Virtio Filesystem Direct Host Memory Access support" + default y + depends on VIRTIO_FS + depends on FS_DAX + depends on DAX_DRIVER + help + This allows bypassing guest page cache and allows mapping host page + cache directly in guest address space. + + If you want to allow mounting a Virtio Filesystem with the "dax" + option, answer Y. diff --git a/fs/fuse/Makefile b/fs/fuse/Makefile index 6971454a2bdf..8729e1e2033b 100644 --- a/fs/fuse/Makefile +++ b/fs/fuse/Makefile @@ -7,6 +7,7 @@ obj-$(CONFIG_FUSE_FS) += fuse.o obj-$(CONFIG_CUSE) += cuse.o obj-$(CONFIG_VIRTIO_FS) += virtiofs.o -fuse-objs := dev.o dir.o file.o inode.o control.o xattr.o acl.o readdir.o -fuse-objs += passthrough.o -virtiofs-y += virtio_fs.o +fuse-y := dev.o dir.o file.o inode.o control.o xattr.o acl.o readdir.o passthrough.o +fuse-$(CONFIG_FUSE_DAX) += dax.o + +virtiofs-y := virtio_fs.o diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c new file mode 100644 index 000000000000..9660d01f49a5 --- /dev/null +++ b/fs/fuse/dax.c @@ -0,0 +1,36 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * dax: direct host memory access + * Copyright (C) 2020 Red Hat, Inc. + */ + +#include "fuse_i.h" + +#include + +struct fuse_conn_dax { + /* DAX device */ + struct dax_device *dev; +}; + +void fuse_dax_conn_free(struct fuse_conn *fc) +{ + kfree(fc->dax); +} + +int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev) +{ + struct fuse_conn_dax *fcd; + + if (!dax_dev) + return 0; + + fcd = kzalloc(sizeof(*fcd), GFP_KERNEL); + if (!fcd) + return -ENOMEM; + + fcd->dev = dax_dev; + + fc->dax = fcd; + return 0; +} diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 4d58c3b84c11..cc554464a218 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -503,10 +503,14 @@ struct fuse_fs_context { bool no_control:1; bool no_force_umount:1; bool legacy_opts_show:1; + bool dax:1; unsigned int max_read; unsigned int blksize; const char *subtype; + /* DAX device, may be NULL */ + struct dax_device *dax_dev; + /* fuse_dev pointer to fill in, should contain NULL on entry */ void **fudptr; }; @@ -779,6 +783,11 @@ struct fuse_conn { /** List of device instances belonging to this connection */ struct list_head devices; +#ifdef CONFIG_FUSE_DAX + /* Dax specific conn data, non-NULL if DAX is enabled */ + struct fuse_conn_dax *dax; +#endif + /** IDR for passthrough requests */ struct idr passthrough_req; @@ -1133,6 +1142,11 @@ unsigned int fuse_len_args(unsigned int numargs, struct fuse_arg *args); u64 fuse_get_unique(struct fuse_iqueue *fiq); void fuse_free_conn(struct fuse_conn *fc); +/* dax.c */ + +int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev); +void fuse_dax_conn_free(struct fuse_conn *fc); + /* passthrough.c */ int fuse_passthrough_open(struct fuse_dev *fud, u32 lower_fd); int fuse_passthrough_setup(struct fuse_conn *fc, struct fuse_file *ff, diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index a629c30d4ea6..f1979432523c 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -597,6 +597,11 @@ static int fuse_show_options(struct seq_file *m, struct dentry *root) if (sb->s_bdev && sb->s_blocksize != FUSE_DEFAULT_BLKSIZE) seq_printf(m, ",blksize=%lu", sb->s_blocksize); } +#ifdef CONFIG_FUSE_DAX + if (fc->dax) + seq_puts(m, ",dax"); +#endif + return 0; } @@ -663,6 +668,8 @@ void fuse_conn_put(struct fuse_conn *fc) if (refcount_dec_and_test(&fc->count)) { struct fuse_iqueue *fiq = &fc->iq; + if (IS_ENABLED(CONFIG_FUSE_DAX)) + fuse_dax_conn_free(fc); if (fiq->ops->release) fiq->ops->release(fiq); put_pid_ns(fc->pid_ns); @@ -1206,11 +1213,17 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) if (sb->s_user_ns != &init_user_ns) sb->s_xattr = fuse_no_acl_xattr_handlers; + if (IS_ENABLED(CONFIG_FUSE_DAX)) { + err = fuse_dax_conn_alloc(fc, ctx->dax_dev); + if (err) + goto err; + } + if (ctx->fudptr) { err = -ENOMEM; fud = fuse_dev_alloc_install(fc); if (!fud) - goto err; + goto err_free_dax; } fc->dev = sb->s_dev; @@ -1265,6 +1278,9 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) err_dev_free: if (fud) fuse_dev_free(fud); + err_free_dax: + if (IS_ENABLED(CONFIG_FUSE_DAX)) + fuse_dax_conn_free(fc); err: return err; } diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 1e533e320779..88299b00cd2e 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -13,6 +13,7 @@ #include #include #include +#include #include #include #include "fuse_i.h" @@ -81,6 +82,44 @@ struct virtio_fs_req_work { static int virtio_fs_enqueue_req(struct virtio_fs_vq *fsvq, struct fuse_req *req, bool in_flight); +enum { + OPT_DAX, +}; + +static const struct fs_parameter_spec virtio_fs_parameters[] = { + fsparam_flag("dax", OPT_DAX), + {} +}; + +static int virtio_fs_parse_param(struct fs_context *fc, + struct fs_parameter *param) +{ + struct fs_parse_result result; + struct fuse_fs_context *ctx = fc->fs_private; + int opt; + + opt = fs_parse(fc, virtio_fs_parameters, param, &result); + if (opt < 0) + return opt; + + switch (opt) { + case OPT_DAX: + ctx->dax = 1; + break; + default: + return -EINVAL; + } + + return 0; +} + +static void virtio_fs_free_fc(struct fs_context *fc) +{ + struct fuse_fs_context *ctx = fc->fs_private; + + kfree(ctx); +} + static inline struct virtio_fs_vq *vq_to_fsvq(struct virtqueue *vq) { struct virtio_fs *fs = vq->vdev->priv; @@ -1230,23 +1269,27 @@ static const struct fuse_iqueue_ops virtio_fs_fiq_ops = { .release = virtio_fs_fiq_release, }; -static int virtio_fs_fill_super(struct super_block *sb) +static inline void virtio_fs_ctx_set_defaults(struct fuse_fs_context *ctx) +{ + ctx->rootmode = S_IFDIR; + ctx->default_permissions = 1; + ctx->allow_other = 1; + ctx->max_read = UINT_MAX; + ctx->blksize = 512; + ctx->destroy = true; + ctx->no_control = true; + ctx->no_force_umount = true; +} + +static int virtio_fs_fill_super(struct super_block *sb, struct fs_context *fsc) { struct fuse_conn *fc = get_fuse_conn_super(sb); struct virtio_fs *fs = fc->iq.priv; + struct fuse_fs_context *ctx = fsc->fs_private; unsigned int i; int err; - struct fuse_fs_context ctx = { - .rootmode = S_IFDIR, - .default_permissions = 1, - .allow_other = 1, - .max_read = UINT_MAX, - .blksize = 512, - .destroy = true, - .no_control = true, - .no_force_umount = true, - }; + virtio_fs_ctx_set_defaults(ctx); mutex_lock(&virtio_fs_mutex); /* After holding mutex, make sure virtiofs device is still there. @@ -1270,8 +1313,10 @@ static int virtio_fs_fill_super(struct super_block *sb) } /* virtiofs allocates and installs its own fuse devices */ - ctx.fudptr = NULL; - err = fuse_fill_super_common(sb, &ctx); + ctx->fudptr = NULL; + if (ctx->dax) + ctx->dax_dev = fs->dax_dev; + err = fuse_fill_super_common(sb, ctx); if (err < 0) goto err_free_fuse_devs; @@ -1385,7 +1430,7 @@ static int virtio_fs_get_tree(struct fs_context *fsc) return PTR_ERR(sb); if (!sb->s_root) { - err = virtio_fs_fill_super(sb); + err = virtio_fs_fill_super(sb, fsc); if (err) { deactivate_locked_super(sb); return err; @@ -1400,11 +1445,19 @@ static int virtio_fs_get_tree(struct fs_context *fsc) } static const struct fs_context_operations virtio_fs_context_ops = { + .free = virtio_fs_free_fc, + .parse_param = virtio_fs_parse_param, .get_tree = virtio_fs_get_tree, }; static int virtio_fs_init_fs_context(struct fs_context *fsc) { + struct fuse_fs_context *ctx; + + ctx = kzalloc(sizeof(struct fuse_fs_context), GFP_KERNEL); + if (!ctx) + return -ENOMEM; + fsc->fs_private = ctx; fsc->ops = &virtio_fs_context_ops; return 0; } From 65a6f1e52a2a5f7ca403134174409115ec674e79 Mon Sep 17 00:00:00 2001 From: Thomas Turner Date: Tue, 7 Oct 2025 20:53:29 +0100 Subject: [PATCH 012/148] fuse: backport virtiofs to use fs_parameter_spec We are missing this commit, which is difficult to apply right now. Author: Al Viro Date: Sat Sep 7 07:23:15 2019 -0400 fs_parse: fold fs_parameter_desc/fs_parameter_spec The former contains nothing but a pointer to an array of the latter... Signed-off-by: Al Viro Change-Id: Ie7b01f06eaf8211bdcf00f3d3a0c9202552d740d --- fs/fuse/virtio_fs.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 88299b00cd2e..f5830d07342f 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -86,11 +86,15 @@ enum { OPT_DAX, }; -static const struct fs_parameter_spec virtio_fs_parameters[] = { +static const struct fs_parameter_spec virtio_param_specs[] = { fsparam_flag("dax", OPT_DAX), {} }; +static const struct fs_parameter_description virtio_fs_parameters = { + .specs = virtio_param_specs, +}; + static int virtio_fs_parse_param(struct fs_context *fc, struct fs_parameter *param) { @@ -98,7 +102,7 @@ static int virtio_fs_parse_param(struct fs_context *fc, struct fuse_fs_context *ctx = fc->fs_private; int opt; - opt = fs_parse(fc, virtio_fs_parameters, param, &result); + opt = fs_parse(fc, &virtio_fs_parameters, param, &result); if (opt < 0) return opt; From 14b3f15e0a567b21e972d46edcda56b0fd60edd7 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 5 Feb 2020 08:15:46 -0500 Subject: [PATCH 013/148] UPSTREAM: fuse: Support RENAME_WHITEOUT flag Allow fuse to pass RENAME_WHITEOUT to fuse server. Overlayfs on top of virtiofs uses RENAME_WHITEOUT. Without this patch renaming a directory in overlayfs (dir is on lower) fails with -EINVAL. With this patch it works. Change-Id: I660cdd90b15c0d7e5d776a0b7a8081408310c00f Signed-off-by: Vivek Goyal Signed-off-by: Miklos Szeredi --- fs/fuse/dir.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 7cf55a7635fa..97fb62426a18 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -874,7 +874,7 @@ static int fuse_rename2(struct inode *olddir, struct dentry *oldent, if (fuse_is_bad(olddir)) return -EIO; - if (flags & ~(RENAME_NOREPLACE | RENAME_EXCHANGE)) + if (flags & ~(RENAME_NOREPLACE | RENAME_EXCHANGE | RENAME_WHITEOUT)) return -EINVAL; if (flags) { From df3932afeed29720840ba7099959dda1240caa25 Mon Sep 17 00:00:00 2001 From: zhengbin Date: Tue, 14 Jan 2020 20:39:45 +0800 Subject: [PATCH 014/148] UPSTREAM: fuse: use true,false for bool variable Fixes coccicheck warning: fs/fuse/readdir.c:335:1-19: WARNING: Assignment of 0/1 to bool variable fs/fuse/file.c:1398:2-19: WARNING: Assignment of 0/1 to bool variable fs/fuse/file.c:1400:2-20: WARNING: Assignment of 0/1 to bool variable fs/fuse/cuse.c:454:1-20: WARNING: Assignment of 0/1 to bool variable fs/fuse/cuse.c:455:1-19: WARNING: Assignment of 0/1 to bool variable fs/fuse/inode.c:497:2-17: WARNING: Assignment of 0/1 to bool variable fs/fuse/inode.c:504:2-23: WARNING: Assignment of 0/1 to bool variable fs/fuse/inode.c:511:2-22: WARNING: Assignment of 0/1 to bool variable fs/fuse/inode.c:518:2-23: WARNING: Assignment of 0/1 to bool variable fs/fuse/inode.c:522:2-26: WARNING: Assignment of 0/1 to bool variable fs/fuse/inode.c:526:2-18: WARNING: Assignment of 0/1 to bool variable fs/fuse/inode.c:1000:1-20: WARNING: Assignment of 0/1 to bool variable Reported-by: Hulk Robot Change-Id: I4e418bd3e007b02344a2bfca5f4dfda3735c5356 Signed-off-by: zhengbin Signed-off-by: Miklos Szeredi --- fs/fuse/cuse.c | 4 ++-- fs/fuse/file.c | 4 ++-- fs/fuse/inode.c | 14 +++++++------- fs/fuse/readdir.c | 2 +- 4 files changed, 12 insertions(+), 12 deletions(-) diff --git a/fs/fuse/cuse.c b/fs/fuse/cuse.c index 072b68fbf8f4..0e25c31d55a6 100644 --- a/fs/fuse/cuse.c +++ b/fs/fuse/cuse.c @@ -451,8 +451,8 @@ static int cuse_send_init(struct cuse_conn *cc) ap->args.out_args[0].size = sizeof(ia->out); ap->args.out_args[0].value = &ia->out; ap->args.out_args[1].size = CUSE_INIT_INFO_MAX; - ap->args.out_argvar = 1; - ap->args.out_pages = 1; + ap->args.out_argvar = true; + ap->args.out_pages = true; ap->num_pages = 1; ap->pages = &ia->page; ap->descs = &ia->desc; diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 8a715e80c486..0875cf9d85b3 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1437,9 +1437,9 @@ static int fuse_get_user_pages(struct fuse_args_pages *ap, struct iov_iter *ii, ap->args.user_pages = true; if (write) - ap->args.in_pages = 1; + ap->args.in_pages = true; else - ap->args.out_pages = 1; + ap->args.out_pages = true; *nbytesp = nbytes; diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index f1979432523c..858b9ce472be 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -519,36 +519,36 @@ static int fuse_parse_param(struct fs_context *fc, struct fs_parameter *param) case OPT_FD: ctx->fd = result.uint_32; - ctx->fd_present = 1; + ctx->fd_present = true; break; case OPT_ROOTMODE: if (!fuse_valid_type(result.uint_32)) return invalf(fc, "fuse: Invalid rootmode"); ctx->rootmode = result.uint_32; - ctx->rootmode_present = 1; + ctx->rootmode_present = true; break; case OPT_USER_ID: ctx->user_id = make_kuid(fc->user_ns, result.uint_32); if (!uid_valid(ctx->user_id)) return invalf(fc, "fuse: Invalid user_id"); - ctx->user_id_present = 1; + ctx->user_id_present = true; break; case OPT_GROUP_ID: ctx->group_id = make_kgid(fc->user_ns, result.uint_32); if (!gid_valid(ctx->group_id)) return invalf(fc, "fuse: Invalid group_id"); - ctx->group_id_present = 1; + ctx->group_id_present = true; break; case OPT_DEFAULT_PERMISSIONS: - ctx->default_permissions = 1; + ctx->default_permissions = true; break; case OPT_ALLOW_OTHER: - ctx->allow_other = 1; + ctx->allow_other = true; break; case OPT_MAX_READ: @@ -1038,7 +1038,7 @@ void fuse_send_init(struct fuse_conn *fc) /* Variable length argument used for backward compatibility with interface version < 7.5. Rest of init_out is zeroed by do_get_request(), so a short reply is not a problem */ - ia->args.out_argvar = 1; + ia->args.out_argvar = true; ia->args.out_args[0].size = sizeof(ia->out); ia->args.out_args[0].value = &ia->out; ia->args.force = true; diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index 512609da8590..c5203b5a0057 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -340,7 +340,7 @@ static int fuse_readdir_uncached(struct file *file, struct dir_context *ctx) return -ENOMEM; plus = fuse_use_readdirplus(inode, ctx); - ap->args.out_pages = 1; + ap->args.out_pages = true; ap->num_pages = 1; ap->pages = &page; ap->descs = &desc; From 1766a0200b218f03359b3946f7869caedcf9c2a1 Mon Sep 17 00:00:00 2001 From: Kirill Tkhai Date: Fri, 28 Feb 2020 15:15:24 +0300 Subject: [PATCH 015/148] UPSTREAM: fuse: Update stale comment in queue_interrupt() Fixes: 04ec5af0776e "fuse: export fuse_end_request()" Change-Id: I0ac5b54cd95774f27f38a6f99804184fe3799d3d Signed-off-by: Kirill Tkhai Signed-off-by: Miklos Szeredi --- fs/fuse/dev.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index 90afaa7eeeeb..59ecb5f3f955 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -347,7 +347,7 @@ static int queue_interrupt(struct fuse_iqueue *fiq, struct fuse_req *req) list_add_tail(&req->intr_entry, &fiq->interrupts); /* * Pairs with smp_mb() implied by test_and_set_bit() - * from request_end(). + * from fuse_request_end(). */ smp_mb(); if (test_bit(FR_FINISHED, &req->flags)) { From f849366aaceb10af4e9fd65b97352950a9f54637 Mon Sep 17 00:00:00 2001 From: Eryu Guan Date: Tue, 12 May 2020 10:29:04 +0800 Subject: [PATCH 016/148] UPSTREAM: fuse: invalidate inode attr in writeback cache mode Under writeback mode, inode->i_blocks is not updated, making utils du read st.blocks as 0. For example, when using virtiofs (cache=always & nondax mode) with writeback_cache enabled, writing a new file and check its disk usage with du, du reports 0 usage. # uname -r 5.6.0-rc6+ # mount -t virtiofs virtiofs /mnt/virtiofs # rm -f /mnt/virtiofs/testfile # create new file and do extend write # xfs_io -fc "pwrite 0 4k" /mnt/virtiofs/testfile wrote 4096/4096 bytes at offset 0 4 KiB, 1 ops; 0.0001 sec (28.103 MiB/sec and 7194.2446 ops/sec) # du -k /mnt/virtiofs/testfile 0 <==== disk usage is 0 # stat -c %s,%b /mnt/virtiofs/testfile 4096,0 <==== i_size is correct, but st_blocks is 0 Fix it by invalidating attr in fuse_flush(), so we get up-to-date attr from server on next getattr. Change-Id: I524134d71b0b304ce7998604f8263c0462b6c4de Signed-off-by: Eryu Guan Signed-off-by: Miklos Szeredi --- fs/fuse/file.c | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 0875cf9d85b3..05c36833f6ee 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -454,8 +454,9 @@ static int fuse_flush(struct file *file, fl_owner_t id) if (fuse_is_bad(inode)) return -EIO; + err = 0; if (fc->no_flush) - return 0; + goto inval_attr_out; err = write_inode_now(inode, 1); if (err) @@ -484,6 +485,14 @@ static int fuse_flush(struct file *file, fl_owner_t id) fc->no_flush = 1; err = 0; } + +inval_attr_out: + /* + * In memory i_blocks is not maintained by fuse, if writeback cache is + * enabled, i_blocks from cached attr may not be accurate. + */ + if (!err && fc->writeback_cache) + fuse_invalidate_attr(inode); return err; } From 3f0ad2922878b24867645b1921d7734116690edb Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Tue, 19 May 2020 14:50:37 +0200 Subject: [PATCH 017/148] UPSTREAM: fuse: always flush dirty data on close(2) We want cached data to synced with the userspace filesystem on close(), for example to allow getting correct st_blocks value. Do this regardless of whether the userspace filesystem implements a FLUSH method or not. Change-Id: I0587ed52299cbe0caee7b7ff6a265641cf37714f Signed-off-by: Miklos Szeredi --- fs/fuse/file.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 05c36833f6ee..ee8233fb74db 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -454,10 +454,6 @@ static int fuse_flush(struct file *file, fl_owner_t id) if (fuse_is_bad(inode)) return -EIO; - err = 0; - if (fc->no_flush) - goto inval_attr_out; - err = write_inode_now(inode, 1); if (err) return err; @@ -470,6 +466,10 @@ static int fuse_flush(struct file *file, fl_owner_t id) if (err) return err; + err = 0; + if (fc->no_flush) + goto inval_attr_out; + memset(&inarg, 0, sizeof(inarg)); inarg.fh = ff->fh; inarg.lock_owner = fuse_lock_owner_id(fc, id); From 48e94fe1d244f3115f9e99f6b00c28be4a354734 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Tue, 19 May 2020 14:50:37 +0200 Subject: [PATCH 018/148] UPSTREAM: fuse: always allow query of st_dev Fuse mounts without "allow_other" are off-limits to all non-owners. Yet it makes sense to allow querying st_dev on the root, since this value is provided by the kernel, not the userspace filesystem. Allow statx(2) with a zero request mask to succeed on a fuse mounts for all users. Reported-by: Nikolaus Rath Change-Id: Ie7215acbfa5ee7cbfe512b3e12a00d1045f8fbc3 Signed-off-by: Miklos Szeredi --- fs/fuse/dir.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 97fb62426a18..de70d6983a96 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -1754,8 +1754,18 @@ static int fuse_getattr(const struct path *path, struct kstat *stat, if (fuse_is_bad(inode)) return -EIO; - if (!fuse_allow_current_process(fc)) + if (!fuse_allow_current_process(fc)) { + if (!request_mask) { + /* + * If user explicitly requested *nothing* then don't + * error out, but return st_dev only. + */ + stat->result_mask = 0; + stat->dev = inode->i_sb->s_dev; + return 0; + } return -EACCES; + } return fuse_update_get_attr(inode, NULL, stat, request_mask, flags); } From e86391af094d9d1408bc7a5dacf8908f7f8be8ce Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Tue, 19 May 2020 14:50:37 +0200 Subject: [PATCH 019/148] UPSTREAM: fuse: use dump_page Instead of custom page dumping, use the standard helper. Reported-by: Matthew Wilcox Change-Id: Ia4cb115a3265ad613b66c5c953f2bf917a0b270f Signed-off-by: Miklos Szeredi --- fs/fuse/dev.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index 59ecb5f3f955..ad7a72cf6bfd 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -780,8 +780,7 @@ static int fuse_check_page(struct page *page) 1 << PG_workingset | 1 << PG_reclaim | 1 << PG_waiters))) { - pr_warn("trying to steal weird page\n"); - pr_warn(" page=%p index=%li flags=%08lx, count=%i, mapcount=%i, mapping=%p\n", page, page->index, page->flags, page_count(page), page_mapcount(page), page->mapping); + dump_page(page, "fuse: trying to steal weird page"); return 1; } return 0; From cd15da23c6d8aabd737b3b9d83e38d38226de252 Mon Sep 17 00:00:00 2001 From: Maxim Patlasov Date: Thu, 19 Sep 2019 17:11:20 +0300 Subject: [PATCH 020/148] UPSTREAM: fuse: optimize writepages search Re-work fi->writepages, replacing list with rb-tree. This improves performance because kernel fuse iterates through fi->writepages for each writeback page and typical number of entries is about 800 (for 100MB of fuse writeback). Before patch: 10240+0 records in 10240+0 records out 10737418240 bytes (11 GB) copied, 41.3473 s, 260 MB/s 2 1 0 57445400 40416 6323676 0 0 33 374743 8633 19210 1 8 88 3 0 29.86% [kernel] [k] _raw_spin_lock 26.62% [fuse] [k] fuse_page_is_writeback After patch: 10240+0 records in 10240+0 records out 10737418240 bytes (11 GB) copied, 21.4954 s, 500 MB/s 2 9 0 53676040 31744 10265984 0 0 64 854790 10956 48387 1 6 88 6 0 23.55% [kernel] [k] copy_user_enhanced_fast_string 9.87% [kernel] [k] __memcpy 3.10% [kernel] [k] _raw_spin_lock Change-Id: I381c5356f90332181f427de0feeddf9fcfc08f83 Signed-off-by: Maxim Patlasov Signed-off-by: Vasily Averin Signed-off-by: Miklos Szeredi --- fs/fuse/file.c | 62 ++++++++++++++++++++++++++++++++++++++---------- fs/fuse/fuse_i.h | 2 +- 2 files changed, 50 insertions(+), 14 deletions(-) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index ee8233fb74db..2d80d0f65d7e 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -366,7 +366,7 @@ u64 fuse_lock_owner_id(struct fuse_conn *fc, fl_owner_t id) struct fuse_writepage_args { struct fuse_io_args ia; - struct list_head writepages_entry; + struct rb_node writepages_entry; struct list_head queue_entry; struct fuse_writepage_args *next; struct inode *inode; @@ -375,17 +375,23 @@ struct fuse_writepage_args { static struct fuse_writepage_args *fuse_find_writeback(struct fuse_inode *fi, pgoff_t idx_from, pgoff_t idx_to) { - struct fuse_writepage_args *wpa; + struct rb_node *n; - list_for_each_entry(wpa, &fi->writepages, writepages_entry) { + n = fi->writepages.rb_node; + + while (n) { + struct fuse_writepage_args *wpa; pgoff_t curr_index; + wpa = rb_entry(n, struct fuse_writepage_args, writepages_entry); WARN_ON(get_fuse_inode(wpa->inode) != fi); curr_index = wpa->ia.write.in.offset >> PAGE_SHIFT; - if (idx_from < curr_index + wpa->ia.ap.num_pages && - curr_index <= idx_to) { + if (idx_from >= curr_index + wpa->ia.ap.num_pages) + n = n->rb_right; + else if (idx_to < curr_index) + n = n->rb_left; + else return wpa; - } } return NULL; } @@ -1655,7 +1661,7 @@ static void fuse_writepage_finish(struct fuse_conn *fc, struct backing_dev_info *bdi = inode_to_bdi(inode); int i; - list_del(&wpa->writepages_entry); + rb_erase(&wpa->writepages_entry, &fi->writepages); for (i = 0; i < ap->num_pages; i++) { dec_wb_stat(&bdi->wb, WB_WRITEBACK); dec_node_page_state(ap->pages[i], NR_WRITEBACK_TEMP); @@ -1749,6 +1755,36 @@ __acquires(fi->lock) } } +static void tree_insert(struct rb_root *root, struct fuse_writepage_args *wpa) +{ + pgoff_t idx_from = wpa->ia.write.in.offset >> PAGE_SHIFT; + pgoff_t idx_to = idx_from + wpa->ia.ap.num_pages - 1; + struct rb_node **p = &root->rb_node; + struct rb_node *parent = NULL; + + WARN_ON(!wpa->ia.ap.num_pages); + while (*p) { + struct fuse_writepage_args *curr; + pgoff_t curr_index; + + parent = *p; + curr = rb_entry(parent, struct fuse_writepage_args, + writepages_entry); + WARN_ON(curr->inode != wpa->inode); + curr_index = curr->ia.write.in.offset >> PAGE_SHIFT; + + if (idx_from >= curr_index + curr->ia.ap.num_pages) + p = &(*p)->rb_right; + else if (idx_to < curr_index) + p = &(*p)->rb_left; + else + return (void) WARN_ON(true); + } + + rb_link_node(&wpa->writepages_entry, parent, p); + rb_insert_color(&wpa->writepages_entry, root); +} + static void fuse_writepage_end(struct fuse_conn *fc, struct fuse_args *args, int error) { @@ -1767,7 +1803,7 @@ static void fuse_writepage_end(struct fuse_conn *fc, struct fuse_args *args, wpa->next = next->next; next->next = NULL; next->ia.ff = fuse_file_get(wpa->ia.ff); - list_add(&next->writepages_entry, &fi->writepages); + tree_insert(&fi->writepages, next); /* * Skip fuse_flush_writepages() to make it easy to crop requests @@ -1902,7 +1938,7 @@ static int fuse_writepage_locked(struct page *page) inc_node_page_state(tmp_page, NR_WRITEBACK_TEMP); spin_lock(&fi->lock); - list_add(&wpa->writepages_entry, &fi->writepages); + tree_insert(&fi->writepages, wpa); list_add_tail(&wpa->queue_entry, &fi->queued_writes); fuse_flush_writepages(inode); spin_unlock(&fi->lock); @@ -2014,10 +2050,10 @@ static bool fuse_writepage_in_flight(struct fuse_writepage_args *new_wpa, WARN_ON(new_ap->num_pages != 0); spin_lock(&fi->lock); - list_del(&new_wpa->writepages_entry); + rb_erase(&new_wpa->writepages_entry, &fi->writepages); old_wpa = fuse_find_writeback(fi, page->index, page->index); if (!old_wpa) { - list_add(&new_wpa->writepages_entry, &fi->writepages); + tree_insert(&fi->writepages, new_wpa); spin_unlock(&fi->lock); return false; } @@ -2132,7 +2168,7 @@ static int fuse_writepages_fill(struct page *page, wpa->inode = inode; spin_lock(&fi->lock); - list_add(&wpa->writepages_entry, &fi->writepages); + tree_insert(&fi->writepages, wpa); spin_unlock(&fi->lock); data->wpa = wpa; @@ -3470,5 +3506,5 @@ void fuse_init_file_inode(struct inode *inode) INIT_LIST_HEAD(&fi->queued_writes); fi->writectr = 0; init_waitqueue_head(&fi->page_waitq); - INIT_LIST_HEAD(&fi->writepages); + fi->writepages = RB_ROOT; } diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index cc554464a218..0c0182dda0d2 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -111,7 +111,7 @@ struct fuse_inode { wait_queue_head_t page_waitq; /* List of writepage requestst (pending or sent) */ - struct list_head writepages; + struct rb_root writepages; }; /* readdir cache (directory only) */ From 232516a946bc70c3b8e4db8c64fd823c1789294c Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Tue, 14 Jul 2020 14:45:41 +0200 Subject: [PATCH 021/148] UPSTREAM: fuse: move rb_erase() before tree_insert() In fuse_writepage_end() the old writepages entry needs to be removed from the rbtree before inserting the new one, otherwise tree_insert() would fail. This is a very rare codepath and no reproducer exists. Change-Id: Ifd7bb409f2e23e12054a2681920f9dc3ccc96930 Signed-off-by: Miklos Szeredi --- fs/fuse/file.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 2d80d0f65d7e..a5624d9eab7e 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1661,7 +1661,6 @@ static void fuse_writepage_finish(struct fuse_conn *fc, struct backing_dev_info *bdi = inode_to_bdi(inode); int i; - rb_erase(&wpa->writepages_entry, &fi->writepages); for (i = 0; i < ap->num_pages; i++) { dec_wb_stat(&bdi->wb, WB_WRITEBACK); dec_node_page_state(ap->pages[i], NR_WRITEBACK_TEMP); @@ -1712,6 +1711,7 @@ __acquires(fi->lock) out_free: fi->writectr--; + rb_erase(&wpa->writepages_entry, &fi->writepages); fuse_writepage_finish(fc, wpa); spin_unlock(&fi->lock); @@ -1795,6 +1795,7 @@ static void fuse_writepage_end(struct fuse_conn *fc, struct fuse_args *args, mapping_set_error(inode->i_mapping, error); spin_lock(&fi->lock); + rb_erase(&wpa->writepages_entry, &fi->writepages); while (wpa->next) { struct fuse_conn *fc = get_fuse_conn(inode); struct fuse_write_in *inarg = &wpa->ia.write.in; From 5cf159f2dd9918877418ddbcc5c854cde0aef10e Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Tue, 14 Jul 2020 14:45:41 +0200 Subject: [PATCH 022/148] UPSTREAM: fuse: fix warning in tree_insert() and clean up writepage insertion fuse_writepages_fill() calls tree_insert() with ap->num_pages = 0 which triggers the following warning: WARNING: CPU: 1 PID: 17211 at fs/fuse/file.c:1728 tree_insert+0xab/0xc0 [fuse] RIP: 0010:tree_insert+0xab/0xc0 [fuse] Call Trace: fuse_writepages_fill+0x5da/0x6a0 [fuse] write_cache_pages+0x171/0x470 fuse_writepages+0x8a/0x100 [fuse] do_writepages+0x43/0xe0 Fix up the warning and clean up the code around rb-tree insertion: - Rename tree_insert() to fuse_insert_writeback() and make it return the conflicting entry in case of failure - Re-add tree_insert() as a wrapper around fuse_insert_writeback() - Rename fuse_writepage_in_flight() to fuse_writepage_add() and reverse the meaning of the return value to mean + "true" in case the writepage entry was successfully added + "false" in case it was in-fligt queued on an existing writepage entry's auxiliary list or the existing writepage entry's temporary page updated Switch from fuse_find_writeback() + tree_insert() to fuse_insert_writeback() - Move setting orig_pages to before inserting/updating the entry; this may result in the orig_pages value being discarded later in case of an in-flight request - In case of a new writepage entry use fuse_writepage_add() unconditionally, only set data->wpa if the entry was added. Fixes: 6b2fb79963fb ("fuse: optimize writepages search") Reported-by: kernel test robot Original-path-by: Vasily Averin Change-Id: I837fe9290f1e026e90175f4bbd3ff831b85c5e46 Signed-off-by: Miklos Szeredi --- fs/fuse/file.c | 62 ++++++++++++++++++++++++-------------------------- 1 file changed, 30 insertions(+), 32 deletions(-) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index a5624d9eab7e..b7f8588fc753 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1755,7 +1755,8 @@ __acquires(fi->lock) } } -static void tree_insert(struct rb_root *root, struct fuse_writepage_args *wpa) +static struct fuse_writepage_args *fuse_insert_writeback(struct rb_root *root, + struct fuse_writepage_args *wpa) { pgoff_t idx_from = wpa->ia.write.in.offset >> PAGE_SHIFT; pgoff_t idx_to = idx_from + wpa->ia.ap.num_pages - 1; @@ -1778,11 +1779,17 @@ static void tree_insert(struct rb_root *root, struct fuse_writepage_args *wpa) else if (idx_to < curr_index) p = &(*p)->rb_left; else - return (void) WARN_ON(true); + return curr; } rb_link_node(&wpa->writepages_entry, parent, p); rb_insert_color(&wpa->writepages_entry, root); + return NULL; +} + +static void tree_insert(struct rb_root *root, struct fuse_writepage_args *wpa) +{ + WARN_ON(fuse_insert_writeback(root, wpa)); } static void fuse_writepage_end(struct fuse_conn *fc, struct fuse_args *args, @@ -2034,14 +2041,14 @@ static void fuse_writepages_send(struct fuse_fill_wb_data *data) } /* - * First recheck under fi->lock if the offending offset is still under - * writeback. If yes, then iterate auxiliary write requests, to see if there's + * Check under fi->lock if the page is under writeback, and insert it onto the + * rb_tree if not. Otherwise iterate auxiliary write requests, to see if there's * one already added for a page at this offset. If there's none, then insert * this new request onto the auxiliary list, otherwise reuse the existing one by - * copying the new page contents over to the old temporary page. + * swapping the new temp page with the old one. */ -static bool fuse_writepage_in_flight(struct fuse_writepage_args *new_wpa, - struct page *page) +static bool fuse_writepage_add(struct fuse_writepage_args *new_wpa, + struct page *page) { struct fuse_inode *fi = get_fuse_inode(new_wpa->inode); struct fuse_writepage_args *tmp; @@ -2049,17 +2056,15 @@ static bool fuse_writepage_in_flight(struct fuse_writepage_args *new_wpa, struct fuse_args_pages *new_ap = &new_wpa->ia.ap; WARN_ON(new_ap->num_pages != 0); + new_ap->num_pages = 1; spin_lock(&fi->lock); - rb_erase(&new_wpa->writepages_entry, &fi->writepages); - old_wpa = fuse_find_writeback(fi, page->index, page->index); + old_wpa = fuse_insert_writeback(&fi->writepages, new_wpa); if (!old_wpa) { - tree_insert(&fi->writepages, new_wpa); spin_unlock(&fi->lock); - return false; + return true; } - new_ap->num_pages = 1; for (tmp = old_wpa->next; tmp; tmp = tmp->next) { pgoff_t curr_index; @@ -2088,7 +2093,7 @@ static bool fuse_writepage_in_flight(struct fuse_writepage_args *new_wpa, fuse_writepage_free(new_wpa); } - return true; + return false; } static int fuse_writepages_fill(struct page *page, @@ -2167,12 +2172,6 @@ static int fuse_writepages_fill(struct page *page, ap->args.end = fuse_writepage_end; ap->num_pages = 0; wpa->inode = inode; - - spin_lock(&fi->lock); - tree_insert(&fi->writepages, wpa); - spin_unlock(&fi->lock); - - data->wpa = wpa; } set_page_writeback(page); @@ -2180,26 +2179,25 @@ static int fuse_writepages_fill(struct page *page, ap->pages[ap->num_pages] = tmp_page; ap->descs[ap->num_pages].offset = 0; ap->descs[ap->num_pages].length = PAGE_SIZE; + data->orig_pages[ap->num_pages] = page; inc_wb_stat(&inode_to_bdi(inode)->wb, WB_WRITEBACK); inc_node_page_state(tmp_page, NR_WRITEBACK_TEMP); err = 0; - if (is_writeback && fuse_writepage_in_flight(wpa, page)) { + if (data->wpa) { + /* + * Protected by fi->lock against concurrent access by + * fuse_page_is_writeback(). + */ + spin_lock(&fi->lock); + ap->num_pages++; + spin_unlock(&fi->lock); + } else if (fuse_writepage_add(wpa, page)) { + data->wpa = wpa; + } else { end_page_writeback(page); - data->wpa = NULL; - goto out_unlock; } - data->orig_pages[ap->num_pages] = page; - - /* - * Protected by fi->lock against concurrent access by - * fuse_page_is_writeback(). - */ - spin_lock(&fi->lock); - ap->num_pages++; - spin_unlock(&fi->lock); - out_unlock: unlock_page(page); From abf9dfb0e1d772e7335da5f6639242e979cfe607 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Tue, 14 Jul 2020 14:45:41 +0200 Subject: [PATCH 023/148] UPSTREAM: fuse: reject options on reconfigure via fsconfig(2) Previous patch changed handling of remount/reconfigure to ignore all options, including those that are unknown to the fuse kernel fs. This was done for backward compatibility, but this likely only affects the old mount(2) API. The new fsconfig(2) based reconfiguration could possibly be improved. This would make the new API less of a drop in replacement for the old, OTOH this is a good chance to get rid of some weirdnesses in the old API. Several other behaviors might make sense: 1) unknown options are rejected, known options are ignored 2) unknown options are rejected, known options are rejected if the value is changed, allowed otherwise 3) all options are rejected Prior to the backward compatibility fix to ignore all options all known options were accepted (1), even if they change the value of a mount parameter; fuse_reconfigure() does not look at the config values set by fuse_parse_param(). To fix that we'd need to verify that the value provided is the same as set in the initial configuration (2). The major drawback is that this is much more complex than just rejecting all attempts at changing options (3); i.e. all options signify initial configuration values and don't make sense on reconfigure. This patch opts for (3) with the rationale that no mount options are reconfigurable in fuse. Change-Id: I1eb25670fd437473a1a6c431fed1bb68a80c1046 Signed-off-by: Miklos Szeredi --- fs/fuse/inode.c | 16 ++++++++++------ fs/namespace.c | 1 + include/linux/fs_context.h | 1 + 3 files changed, 12 insertions(+), 6 deletions(-) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 858b9ce472be..521f0b7317c3 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -491,12 +491,16 @@ static int fuse_parse_param(struct fs_context *fc, struct fs_parameter *param) struct fuse_fs_context *ctx = fc->fs_private; int opt; - /* - * Ignore options coming from mount(MS_REMOUNT) for backward - * compatibility. - */ - if (fc->purpose == FS_CONTEXT_FOR_RECONFIGURE) - return 0; + if (fc->purpose == FS_CONTEXT_FOR_RECONFIGURE) { + /* + * Ignore options coming from mount(MS_REMOUNT) for backward + * compatibility. + */ + if (fc->oldapi) + return 0; + + return invalf(fc, "No changes allowed in reconfigure"); + } opt = fs_parse(fc, &fuse_fs_parameters, param, &result); if (opt < 0) diff --git a/fs/namespace.c b/fs/namespace.c index 2fd0bf91aa77..176bccdf6f30 100644 --- a/fs/namespace.c +++ b/fs/namespace.c @@ -2618,6 +2618,7 @@ static int do_remount(struct path *path, int ms_flags, int sb_flags, if (IS_ERR(fc)) return PTR_ERR(fc); + fc->oldapi = true; err = parse_monolithic_mount_data(fc, data); if (!err) { down_write(&sb->s_umount); diff --git a/include/linux/fs_context.h b/include/linux/fs_context.h index ebcb91a57e86..d5a73cd318f8 100644 --- a/include/linux/fs_context.h +++ b/include/linux/fs_context.h @@ -105,6 +105,7 @@ struct fs_context { enum fs_context_phase phase:8; /* The phase the context is in */ bool need_free:1; /* Need to call ops->free() */ bool global:1; /* Goes into &init_user_ns */ + bool oldapi:1; /* Coming from mount(2) */ }; struct fs_context_operations { From 01906dcb712d7c48a7a70995c282d83163575e63 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Tue, 14 Jul 2020 14:45:41 +0200 Subject: [PATCH 024/148] UPSTREAM: fuse: clean up condition for writepage sending fuse_writepages_fill uses following construction: if (wpa && ap->num_pages && (A || B || C)) { action; } else if (wpa && D) { if (E) { the same action; } } - ap->num_pages check is always true and can be removed - "if" and "else if" calls the same action and can be merged. Move checking A, B, C, D, E conditions to a helper, add comments. Original-patch-by: Vasily Averin Change-Id: Ibb9755ad4f025cf027b998205d6838727dad9c72 Signed-off-by: Miklos Szeredi --- fs/fuse/file.c | 53 +++++++++++++++++++++++++++++++++----------------- 1 file changed, 35 insertions(+), 18 deletions(-) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index b7f8588fc753..a19cc87dac49 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -2096,6 +2096,40 @@ static bool fuse_writepage_add(struct fuse_writepage_args *new_wpa, return false; } +static bool fuse_writepage_need_send(struct fuse_conn *fc, struct page *page, + struct fuse_args_pages *ap, + struct fuse_fill_wb_data *data) +{ + WARN_ON(!ap->num_pages); + + /* + * Being under writeback is unlikely but possible. For example direct + * read to an mmaped fuse file will set the page dirty twice; once when + * the pages are faulted with get_user_pages(), and then after the read + * completed. + */ + if (fuse_page_is_writeback(data->inode, page->index)) + return true; + + /* Reached max pages */ + if (ap->num_pages == fc->max_pages) + return true; + + /* Reached max write bytes */ + if ((ap->num_pages + 1) * PAGE_SIZE > fc->max_write) + return true; + + /* Discontinuity */ + if (data->orig_pages[ap->num_pages - 1]->index + 1 != page->index) + return true; + + /* Need to grow the pages array? If so, did the expansion fail? */ + if (ap->num_pages == data->max_pages && !fuse_pages_realloc(data)) + return true; + + return false; +} + static int fuse_writepages_fill(struct page *page, struct writeback_control *wbc, void *_data) { @@ -2106,7 +2140,6 @@ static int fuse_writepages_fill(struct page *page, struct fuse_inode *fi = get_fuse_inode(inode); struct fuse_conn *fc = get_fuse_conn(inode); struct page *tmp_page; - bool is_writeback; int err; if (!data->ff) { @@ -2116,25 +2149,9 @@ static int fuse_writepages_fill(struct page *page, goto out_unlock; } - /* - * Being under writeback is unlikely but possible. For example direct - * read to an mmaped fuse file will set the page dirty twice; once when - * the pages are faulted with get_user_pages(), and then after the read - * completed. - */ - is_writeback = fuse_page_is_writeback(inode, page->index); - - if (wpa && ap->num_pages && - (is_writeback || ap->num_pages == fc->max_pages || - (ap->num_pages + 1) * PAGE_SIZE > fc->max_write || - data->orig_pages[ap->num_pages - 1]->index + 1 != page->index)) { + if (wpa && fuse_writepage_need_send(fc, page, ap, data)) { fuse_writepages_send(data); data->wpa = NULL; - } else if (wpa && ap->num_pages == data->max_pages) { - if (!fuse_pages_realloc(data)) { - fuse_writepages_send(data); - data->wpa = NULL; - } } err = -ENOMEM; From 6c16f6636eba1c2d0fe8f8bbbadc40ce06f0b336 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Almeida?= Date: Thu, 23 Jul 2020 13:43:11 -0300 Subject: [PATCH 025/148] UPSTREAM: fuse: update project homepage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit As stated in https://sourceforge.net/projects/fuse/, "the FUSE project has moved to https://github.com/libfuse/" in 22-Dec-2015. Update URLs to reflect this. Change-Id: Ic0e5f3ea47ea4a49ea0a2946ba9db6035f31c49f Signed-off-by: André Almeida Signed-off-by: Miklos Szeredi --- Documentation/filesystems/fuse.txt | 6 ++---- MAINTAINERS | 3 ++- fs/fuse/Kconfig | 2 +- 3 files changed, 5 insertions(+), 6 deletions(-) diff --git a/Documentation/filesystems/fuse.txt b/Documentation/filesystems/fuse.txt index 13af4a49e7db..d32457766947 100644 --- a/Documentation/filesystems/fuse.txt +++ b/Documentation/filesystems/fuse.txt @@ -46,10 +46,8 @@ non-privileged mounts. This opens up new possibilities for the use of filesystems. A good example is sshfs: a secure network filesystem using the sftp protocol. -The userspace library and utilities are available from the FUSE -homepage: - - http://fuse.sourceforge.net/ +The userspace library and utilities are available from the +`FUSE homepage: `_ Filesystem type ~~~~~~~~~~~~~~~ diff --git a/MAINTAINERS b/MAINTAINERS index 0bc36d8a3c63..7b73fc0469c8 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -6742,7 +6742,8 @@ F: drivers/platform/x86/fujitsu-tablet.c FUSE: FILESYSTEM IN USERSPACE M: Miklos Szeredi L: linux-fsdevel@vger.kernel.org -W: http://fuse.sourceforge.net/ +S: Maintained +W: https://github.com/libfuse/ T: git git://git.kernel.org/pub/scm/linux/kernel/git/mszeredi/fuse.git S: Maintained F: fs/fuse/ diff --git a/fs/fuse/Kconfig b/fs/fuse/Kconfig index e3a2a61c5ae9..aed9b703973a 100644 --- a/fs/fuse/Kconfig +++ b/fs/fuse/Kconfig @@ -8,7 +8,7 @@ config FUSE_FS There's also a companion library: libfuse2. This library is available from the FUSE homepage: - + although chances are your distribution already has that library installed if you've installed the "fuse" package itself. From 9f496602dd8ee709dc1d02806c5a93e8a0a168d9 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Fri, 28 Feb 2020 11:34:51 -0500 Subject: [PATCH 026/148] UPSTREAM: pmem: Add functions for reading/writing page to/from pmem This splits pmem_do_bvec() into pmem_do_read() and pmem_do_write(). pmem_do_write() will be used by pmem zero_page_range() as well. Hence sharing the same code. Change-Id: I5fe20ed0dd24028345b6cff111b79f829cba6fbf Suggested-by: Christoph Hellwig Signed-off-by: Vivek Goyal Reviewed-by: Christoph Hellwig Reviewed-by: Pankaj Gupta Link: https://lore.kernel.org/r/20200228163456.1587-2-vgoyal@redhat.com Signed-off-by: Dan Williams --- drivers/nvdimm/pmem.c | 86 +++++++++++++++++++++++++------------------ 1 file changed, 50 insertions(+), 36 deletions(-) diff --git a/drivers/nvdimm/pmem.c b/drivers/nvdimm/pmem.c index 7e65306b2bf2..e638e63f44b5 100644 --- a/drivers/nvdimm/pmem.c +++ b/drivers/nvdimm/pmem.c @@ -137,9 +137,25 @@ static blk_status_t read_pmem(struct page *page, unsigned int off, return BLK_STS_OK; } -static blk_status_t pmem_do_bvec(struct pmem_device *pmem, struct page *page, - unsigned int len, unsigned int off, unsigned int op, - sector_t sector) +static blk_status_t pmem_do_read(struct pmem_device *pmem, + struct page *page, unsigned int page_off, + sector_t sector, unsigned int len) +{ + blk_status_t rc; + phys_addr_t pmem_off = sector * 512 + pmem->data_offset; + void *pmem_addr = pmem->virt_addr + pmem_off; + + if (unlikely(is_bad_pmem(&pmem->bb, sector, len))) + return BLK_STS_IOERR; + + rc = read_pmem(page, page_off, pmem_addr, len); + flush_dcache_page(page); + return rc; +} + +static blk_status_t pmem_do_write(struct pmem_device *pmem, + struct page *page, unsigned int page_off, + sector_t sector, unsigned int len) { blk_status_t rc = BLK_STS_OK; bool bad_pmem = false; @@ -149,34 +165,25 @@ static blk_status_t pmem_do_bvec(struct pmem_device *pmem, struct page *page, if (unlikely(is_bad_pmem(&pmem->bb, sector, len))) bad_pmem = true; - if (!op_is_write(op)) { - if (unlikely(bad_pmem)) - rc = BLK_STS_IOERR; - else { - rc = read_pmem(page, off, pmem_addr, len); - flush_dcache_page(page); - } - } else { - /* - * Note that we write the data both before and after - * clearing poison. The write before clear poison - * handles situations where the latest written data is - * preserved and the clear poison operation simply marks - * the address range as valid without changing the data. - * In this case application software can assume that an - * interrupted write will either return the new good - * data or an error. - * - * However, if pmem_clear_poison() leaves the data in an - * indeterminate state we need to perform the write - * after clear poison. - */ - flush_dcache_page(page); - write_pmem(pmem_addr, page, off, len); - if (unlikely(bad_pmem)) { - rc = pmem_clear_poison(pmem, pmem_off, len); - write_pmem(pmem_addr, page, off, len); - } + /* + * Note that we write the data both before and after + * clearing poison. The write before clear poison + * handles situations where the latest written data is + * preserved and the clear poison operation simply marks + * the address range as valid without changing the data. + * In this case application software can assume that an + * interrupted write will either return the new good + * data or an error. + * + * However, if pmem_clear_poison() leaves the data in an + * indeterminate state we need to perform the write + * after clear poison. + */ + flush_dcache_page(page); + write_pmem(pmem_addr, page, page_off, len); + if (unlikely(bad_pmem)) { + rc = pmem_clear_poison(pmem, pmem_off, len); + write_pmem(pmem_addr, page, page_off, len); } return rc; @@ -198,8 +205,12 @@ static blk_qc_t pmem_make_request(struct request_queue *q, struct bio *bio) do_acct = nd_iostat_start(bio, &start); bio_for_each_segment(bvec, bio, iter) { - rc = pmem_do_bvec(pmem, bvec.bv_page, bvec.bv_len, - bvec.bv_offset, bio_op(bio), iter.bi_sector); + if (op_is_write(bio_op(bio))) + rc = pmem_do_write(pmem, bvec.bv_page, bvec.bv_offset, + iter.bi_sector, bvec.bv_len); + else + rc = pmem_do_read(pmem, bvec.bv_page, bvec.bv_offset, + iter.bi_sector, bvec.bv_len); if (rc) { bio->bi_status = rc; break; @@ -224,9 +235,12 @@ static int pmem_rw_page(struct block_device *bdev, sector_t sector, struct pmem_device *pmem = bdev->bd_queue->queuedata; blk_status_t rc; - rc = pmem_do_bvec(pmem, page, hpage_nr_pages(page) * PAGE_SIZE, - 0, op, sector); - + if (op_is_write(op)) + rc = pmem_do_write(pmem, page, 0, sector, + hpage_nr_pages(page) * PAGE_SIZE); + else + rc = pmem_do_read(pmem, page, 0, sector, + hpage_nr_pages(page) * PAGE_SIZE); /* * The ->rw_page interface is subtle and tricky. The core * retries on any error, so we can only invoke page_endio() in From 316199322414b43dcc4a535f61d005b2c5ec1a03 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Fri, 28 Feb 2020 11:34:52 -0500 Subject: [PATCH 027/148] UPSTREAM: dax, pmem: Add a dax operation zero_page_range Add a dax operation zero_page_range, to zero a page. This will also clear any known poison in the page being zeroed. As of now, zeroing of one page is allowed in a single call. There are no callers which are trying to zero more than a page in a single call. Once we grow the callers which zero more than a page in single call, we can add that support. Primary reason for not doing that yet is that this will add little complexity in dm implementation where a range might be spanning multiple underlying targets and one will have to split the range into multiple sub ranges and call zero_page_range() on individual targets. Suggested-by: Christoph Hellwig Change-Id: Id86c5bac0824101755c69eb8822760fd05007210 Signed-off-by: Vivek Goyal Reviewed-by: Pankaj Gupta Link: https://lore.kernel.org/r/20200228163456.1587-3-vgoyal@redhat.com Signed-off-by: Dan Williams --- drivers/dax/super.c | 20 ++++++++++++++++++++ drivers/nvdimm/pmem.c | 11 +++++++++++ include/linux/dax.h | 4 ++++ 3 files changed, 35 insertions(+) diff --git a/drivers/dax/super.c b/drivers/dax/super.c index 5005b92b197d..c54bb58de20c 100644 --- a/drivers/dax/super.c +++ b/drivers/dax/super.c @@ -348,6 +348,26 @@ size_t dax_copy_to_iter(struct dax_device *dax_dev, pgoff_t pgoff, void *addr, } EXPORT_SYMBOL_GPL(dax_copy_to_iter); +int dax_zero_page_range(struct dax_device *dax_dev, pgoff_t pgoff, + size_t nr_pages) +{ + if (!dax_alive(dax_dev)) + return -ENXIO; + + if (!dax_dev->ops->zero_page_range) + return -EOPNOTSUPP; + /* + * There are no callers that want to zero more than one page as of now. + * Once users are there, this check can be removed after the + * device mapper code has been updated to split ranges across targets. + */ + if (nr_pages != 1) + return -EIO; + + return dax_dev->ops->zero_page_range(dax_dev, pgoff, nr_pages); +} +EXPORT_SYMBOL_GPL(dax_zero_page_range); + #ifdef CONFIG_ARCH_HAS_PMEM_API void arch_wb_cache_pmem(void *addr, size_t size); void dax_flush(struct dax_device *dax_dev, void *addr, size_t size) diff --git a/drivers/nvdimm/pmem.c b/drivers/nvdimm/pmem.c index e638e63f44b5..0f94658af4e7 100644 --- a/drivers/nvdimm/pmem.c +++ b/drivers/nvdimm/pmem.c @@ -283,6 +283,16 @@ static const struct block_device_operations pmem_fops = { .revalidate_disk = nvdimm_revalidate_disk, }; +static int pmem_dax_zero_page_range(struct dax_device *dax_dev, pgoff_t pgoff, + size_t nr_pages) +{ + struct pmem_device *pmem = dax_get_private(dax_dev); + + return blk_status_to_errno(pmem_do_write(pmem, ZERO_PAGE(0), 0, + PFN_PHYS(pgoff) >> SECTOR_SHIFT, + PAGE_SIZE)); +} + static long pmem_dax_direct_access(struct dax_device *dax_dev, pgoff_t pgoff, long nr_pages, void **kaddr, pfn_t *pfn) { @@ -314,6 +324,7 @@ static const struct dax_operations pmem_dax_ops = { .dax_supported = generic_fsdax_supported, .copy_from_iter = pmem_copy_from_iter, .copy_to_iter = pmem_copy_to_iter, + .zero_page_range = pmem_dax_zero_page_range, }; static const struct attribute_group *pmem_attribute_groups[] = { diff --git a/include/linux/dax.h b/include/linux/dax.h index 72a7f03a59f4..d307f6cc7e0b 100644 --- a/include/linux/dax.h +++ b/include/linux/dax.h @@ -34,6 +34,8 @@ struct dax_operations { /* copy_to_iter: required operation for fs-dax direct-i/o */ size_t (*copy_to_iter)(struct dax_device *, pgoff_t, void *, size_t, struct iov_iter *); + /* zero_page_range: required operation. Zero page range */ + int (*zero_page_range)(struct dax_device *, pgoff_t, size_t); }; extern struct attribute_group dax_attribute_group; @@ -226,6 +228,8 @@ size_t dax_copy_from_iter(struct dax_device *dax_dev, pgoff_t pgoff, void *addr, size_t bytes, struct iov_iter *i); size_t dax_copy_to_iter(struct dax_device *dax_dev, pgoff_t pgoff, void *addr, size_t bytes, struct iov_iter *i); +int dax_zero_page_range(struct dax_device *dax_dev, pgoff_t pgoff, + size_t nr_pages); void dax_flush(struct dax_device *dax_dev, void *addr, size_t size); ssize_t dax_iomap_rw(struct kiocb *iocb, struct iov_iter *iter, From 481a6af47b9a09cecbdf27c21ee82eef4c8b01e5 Mon Sep 17 00:00:00 2001 From: Sebastien Boeuf Date: Wed, 19 Aug 2020 18:19:41 -0400 Subject: [PATCH 028/148] UPSTREAM: virtio: Add get_shm_region method Virtio defines 'shared memory regions' that provide a continuously shared region between the host and guest. Provide a method to find a particular region on a device. Change-Id: I18f5c3b72fd83ab27bf346de3f24aeb61a9f4ab8 Signed-off-by: Sebastien Boeuf Signed-off-by: Dr. David Alan Gilbert Acked-by: Michael S. Tsirkin Cc: kvm@vger.kernel.org Cc: virtualization@lists.linux-foundation.org Cc: "Michael S. Tsirkin" Signed-off-by: Miklos Szeredi --- include/linux/virtio_config.h | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/include/linux/virtio_config.h b/include/linux/virtio_config.h index ad4d4697a167..3469d890d9e1 100644 --- a/include/linux/virtio_config.h +++ b/include/linux/virtio_config.h @@ -10,6 +10,11 @@ struct irq_affinity; +struct virtio_shm_region { + u64 addr; + u64 len; +}; + /** * virtio_config_ops - operations for configuring a virtio device * Note: Do not assume that a transport implements all of the operations @@ -66,6 +71,7 @@ struct irq_affinity; * the caller can then copy. * @set_vq_affinity: set the affinity for a virtqueue (optional). * @get_vq_affinity: get the affinity for a virtqueue (optional). + * @get_shm_region: get a shared memory region based on the index. */ typedef void vq_callback_t(struct virtqueue *); struct virtio_config_ops { @@ -89,6 +95,8 @@ struct virtio_config_ops { const struct cpumask *cpu_mask); const struct cpumask *(*get_vq_affinity)(struct virtio_device *vdev, int index); + bool (*get_shm_region)(struct virtio_device *vdev, + struct virtio_shm_region *region, u8 id); }; /* If driver didn't advertise the feature, it will never appear. */ @@ -251,6 +259,15 @@ int virtqueue_set_affinity(struct virtqueue *vq, const struct cpumask *cpu_mask) return 0; } +static inline +bool virtio_get_shm_region(struct virtio_device *vdev, + struct virtio_shm_region *region, u8 id) +{ + if (!vdev->config->get_shm_region) + return false; + return vdev->config->get_shm_region(vdev, region, id); +} + static inline bool virtio_is_little_endian(struct virtio_device *vdev) { return virtio_has_feature(vdev, VIRTIO_F_VERSION_1) || From 0c4fa1ea4cdbe56ee11c04b684e93c8a8ded4589 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:48 -0400 Subject: [PATCH 029/148] UPSTREAM: virtiofs: keep a list of free dax memory ranges Divide the dax memory range into fixed size ranges (2MB for now) and put them in a list. This will track free ranges. Once an inode requires a free range, we will take one from here and put it in interval-tree of ranges assigned to inode. Change-Id: Icb27b111f8e72d55aa0e12e679946e84af79a285 Signed-off-by: Vivek Goyal Signed-off-by: Peng Tao Signed-off-by: Miklos Szeredi --- fs/fuse/dax.c | 91 ++++++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 90 insertions(+), 1 deletion(-) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index 9660d01f49a5..031106020f75 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -7,20 +7,104 @@ #include "fuse_i.h" #include +#include + +/* Default memory range size, 2MB */ +#define FUSE_DAX_SHIFT 21 +#define FUSE_DAX_SZ (1 << FUSE_DAX_SHIFT) +#define FUSE_DAX_PAGES (FUSE_DAX_SZ / PAGE_SIZE) + +/** Translation information for file offsets to DAX window offsets */ +struct fuse_dax_mapping { + /* Will connect in fcd->free_ranges to keep track of free memory */ + struct list_head list; + + /** Position in DAX window */ + u64 window_offset; + + /** Length of mapping, in bytes */ + loff_t length; +}; struct fuse_conn_dax { /* DAX device */ struct dax_device *dev; + + /* DAX Window Free Ranges */ + long nr_free_ranges; + struct list_head free_ranges; }; +static void fuse_free_dax_mem_ranges(struct list_head *mem_list) +{ + struct fuse_dax_mapping *range, *temp; + + /* Free All allocated elements */ + list_for_each_entry_safe(range, temp, mem_list, list) { + list_del(&range->list); + kfree(range); + } +} + void fuse_dax_conn_free(struct fuse_conn *fc) { - kfree(fc->dax); + if (fc->dax) { + fuse_free_dax_mem_ranges(&fc->dax->free_ranges); + kfree(fc->dax); + } +} + +static int fuse_dax_mem_range_init(struct fuse_conn_dax *fcd) +{ + long nr_pages, nr_ranges; + void *kaddr; + pfn_t pfn; + struct fuse_dax_mapping *range; + int ret, id; + size_t dax_size = -1; + unsigned long i; + + INIT_LIST_HEAD(&fcd->free_ranges); + id = dax_read_lock(); + nr_pages = dax_direct_access(fcd->dev, 0, PHYS_PFN(dax_size), &kaddr, + &pfn); + dax_read_unlock(id); + if (nr_pages < 0) { + pr_debug("dax_direct_access() returned %ld\n", nr_pages); + return nr_pages; + } + + nr_ranges = nr_pages/FUSE_DAX_PAGES; + pr_debug("%s: dax mapped %ld pages. nr_ranges=%ld\n", + __func__, nr_pages, nr_ranges); + + for (i = 0; i < nr_ranges; i++) { + range = kzalloc(sizeof(struct fuse_dax_mapping), GFP_KERNEL); + ret = -ENOMEM; + if (!range) + goto out_err; + + /* TODO: This offset only works if virtio-fs driver is not + * having some memory hidden at the beginning. This needs + * better handling + */ + range->window_offset = i * FUSE_DAX_SZ; + range->length = FUSE_DAX_SZ; + list_add_tail(&range->list, &fcd->free_ranges); + } + + fcd->nr_free_ranges = nr_ranges; + return 0; +out_err: + /* Free All allocated elements */ + fuse_free_dax_mem_ranges(&fcd->free_ranges); + return ret; } int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev) { struct fuse_conn_dax *fcd; + int err; if (!dax_dev) return 0; @@ -30,6 +114,11 @@ int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev) return -ENOMEM; fcd->dev = dax_dev; + err = fuse_dax_mem_range_init(fcd); + if (err) { + kfree(fcd); + return err; + } fc->dax = fcd; return 0; From 12d56dffdc6b2163670d9be408086c013a8798a5 Mon Sep 17 00:00:00 2001 From: Stefan Hajnoczi Date: Wed, 19 Aug 2020 18:19:49 -0400 Subject: [PATCH 030/148] UPSTREAM: virtiofs: implement FUSE_INIT map_alignment field The device communicates FUSE_SETUPMAPPING/FUSE_REMOVMAPPING alignment constraints via the FUST_INIT map_alignment field. Parse this field and ensure our DAX mappings meet the alignment constraints. We don't actually align anything differently since our mappings are already 2MB aligned. Just check the value when the connection is established. If it becomes necessary to honor arbitrary alignments in the future we'll have to adjust how mappings are sized. The upshot of this commit is that we can be confident that mappings will work even when emulating x86 on Power and similar combinations where the host page sizes are different. Change-Id: I8c7b002aa413dbdae43dc159d82cba672e671351 Signed-off-by: Stefan Hajnoczi Signed-off-by: Vivek Goyal Signed-off-by: Miklos Szeredi --- fs/fuse/dax.c | 15 ++++++++++++++- fs/fuse/fuse_i.h | 1 + fs/fuse/inode.c | 17 ++++++++++++++++- include/uapi/linux/fuse.h | 4 +++- 4 files changed, 34 insertions(+), 3 deletions(-) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index 031106020f75..fec8a2bd75b3 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -9,7 +9,10 @@ #include #include -/* Default memory range size, 2MB */ +/* + * Default memory range size. A power of 2 so it agrees with common FUSE_INIT + * map_alignment values 4KB and 64KB. + */ #define FUSE_DAX_SHIFT 21 #define FUSE_DAX_SZ (1 << FUSE_DAX_SHIFT) #define FUSE_DAX_PAGES (FUSE_DAX_SZ / PAGE_SIZE) @@ -123,3 +126,13 @@ int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev) fc->dax = fcd; return 0; } + +bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment) +{ + if (fc->dax && (map_alignment > FUSE_DAX_SHIFT)) { + pr_warn("FUSE: map_alignment %u incompatible with dax mem range size %u\n", + map_alignment, FUSE_DAX_SZ); + return false; + } + return true; +} diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 0c0182dda0d2..42caa2061864 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1146,6 +1146,7 @@ void fuse_free_conn(struct fuse_conn *fc); int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev); void fuse_dax_conn_free(struct fuse_conn *fc); +bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment); /* passthrough.c */ int fuse_passthrough_open(struct fuse_dev *fud, u32 lower_fd); diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 521f0b7317c3..38f3ac5eadec 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -924,9 +924,10 @@ static void process_init_reply(struct fuse_conn *fc, struct fuse_args *args, { struct fuse_init_args *ia = container_of(args, typeof(*ia), args); struct fuse_init_out *arg = &ia->out; + bool ok = true; if (error || arg->major != FUSE_KERNEL_VERSION) - fc->conn_error = 1; + ok = false; else { unsigned long ra_pages; @@ -989,6 +990,11 @@ static void process_init_reply(struct fuse_conn *fc, struct fuse_args *args, min_t(unsigned int, FUSE_MAX_MAX_PAGES, max_t(unsigned int, arg->max_pages, 1)); } + if (IS_ENABLED(CONFIG_FUSE_DAX) && + arg->flags & FUSE_MAP_ALIGNMENT && + !fuse_dax_check_alignment(fc, arg->map_alignment)) { + ok = false; + } if (arg->flags & FUSE_PASSTHROUGH) { fc->passthrough = 1; /* Prevent further stacking */ @@ -1010,6 +1016,11 @@ static void process_init_reply(struct fuse_conn *fc, struct fuse_args *args, } kfree(ia); + if (!ok) { + fc->conn_init = 0; + fc->conn_error = 1; + } + fuse_set_initialized(fc); wake_up_all(&fc->blocked_waitq); } @@ -1034,6 +1045,10 @@ void fuse_send_init(struct fuse_conn *fc) FUSE_ABORT_ERROR | FUSE_MAX_PAGES | FUSE_CACHE_SYMLINKS | FUSE_NO_OPENDIR_SUPPORT | FUSE_EXPLICIT_INVAL_DATA | FUSE_PASSTHROUGH; +#ifdef CONFIG_FUSE_DAX + if (fc->dax) + ia->in.flags |= FUSE_MAP_ALIGNMENT; +#endif ia->args.opcode = FUSE_INIT; ia->args.in_numargs = 1; ia->args.in_args[0].size = sizeof(ia->in); diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index a62498fbad3e..38446ed6e44c 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -313,7 +313,9 @@ struct fuse_file_lock { * FUSE_CACHE_SYMLINKS: cache READLINK responses * FUSE_NO_OPENDIR_SUPPORT: kernel supports zero-message opendir * FUSE_EXPLICIT_INVAL_DATA: only invalidate cached pages on explicit request - * FUSE_MAP_ALIGNMENT: map_alignment field is valid + * FUSE_MAP_ALIGNMENT: init_out.map_alignment contains log2(byte alignment) for + * foffset and moffset fields in struct + * fuse_setupmapping_out and fuse_removemapping_one. */ #define FUSE_ASYNC_READ (1 << 0) #define FUSE_POSIX_LOCKS (1 << 1) From 6672c1f1453ba855b94a2da79d2d32edecd363d7 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:50 -0400 Subject: [PATCH 031/148] UPSTREAM: virtiofs: introduce setupmapping/removemapping commands Introduce two new fuse commands to setup/remove memory mappings. This will be used to setup/tear down file mapping in dax window. Change-Id: I9a5686146f83a72bdef0147bd1a96ec4dd1a278d Signed-off-by: Vivek Goyal Signed-off-by: Peng Tao Signed-off-by: Miklos Szeredi --- include/uapi/linux/fuse.h | 29 +++++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index 38446ed6e44c..aafafe7a58db 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -900,4 +900,33 @@ struct fuse_copy_file_range_in { uint64_t flags; }; +#define FUSE_SETUPMAPPING_FLAG_WRITE (1ull << 0) +struct fuse_setupmapping_in { + /* An already open handle */ + uint64_t fh; + /* Offset into the file to start the mapping */ + uint64_t foffset; + /* Length of mapping required */ + uint64_t len; + /* Flags, FUSE_SETUPMAPPING_FLAG_* */ + uint64_t flags; + /* Offset in Memory Window */ + uint64_t moffset; +}; + +struct fuse_removemapping_in { + /* number of fuse_removemapping_one follows */ + uint32_t count; +}; + +struct fuse_removemapping_one { + /* Offset into the dax window start the unmapping */ + uint64_t moffset; + /* Length of mapping required */ + uint64_t len; +}; + +#define FUSE_REMOVEMAPPING_MAX_ENTRY \ + (PAGE_SIZE / sizeof(struct fuse_removemapping_one)) + #endif /* _LINUX_FUSE_H */ From 9fe4e93da24b894868615911e4e88dad109dc7a8 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:51 -0400 Subject: [PATCH 032/148] UPSTREAM: virtiofs: implement dax read/write operations This patch implements basic DAX support. mmap() is not implemented yet and will come in later patches. This patch looks into implemeting read/write. We make use of interval tree to keep track of per inode dax mappings. Do not use dax for file extending writes, instead just send WRITE message to daemon (like we do for direct I/O path). This will keep write and i_size change atomic w.r.t crash. Change-Id: Ia135d8554026720dc87e4ef5858b230e91533de8 Signed-off-by: Stefan Hajnoczi Signed-off-by: Dr. David Alan Gilbert Signed-off-by: Vivek Goyal Signed-off-by: Liu Bo Signed-off-by: Peng Tao Cc: Dave Chinner Signed-off-by: Miklos Szeredi --- fs/fuse/Kconfig | 1 + fs/fuse/dax.c | 565 ++++++++++++++++++++++++++++++++++++++ fs/fuse/file.c | 15 +- fs/fuse/fuse_i.h | 15 + fs/fuse/inode.c | 21 +- include/uapi/linux/fuse.h | 1 + 6 files changed, 612 insertions(+), 6 deletions(-) diff --git a/fs/fuse/Kconfig b/fs/fuse/Kconfig index aed9b703973a..ac32f7c956a8 100644 --- a/fs/fuse/Kconfig +++ b/fs/fuse/Kconfig @@ -42,6 +42,7 @@ config VIRTIO_FS config FUSE_DAX bool "Virtio Filesystem Direct Host Memory Access support" default y + select INTERVAL_TREE depends on VIRTIO_FS depends on FS_DAX depends on DAX_DRIVER diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index fec8a2bd75b3..a8d311b2db8e 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -7,7 +7,10 @@ #include "fuse_i.h" #include +#include #include +#include +#include /* * Default memory range size. A power of 2 so it agrees with common FUSE_INIT @@ -22,22 +25,556 @@ struct fuse_dax_mapping { /* Will connect in fcd->free_ranges to keep track of free memory */ struct list_head list; + /* For interval tree in file/inode */ + struct interval_tree_node itn; + /** Position in DAX window */ u64 window_offset; /** Length of mapping, in bytes */ loff_t length; + + /* Is this mapping read-only or read-write */ + bool writable; +}; + +/* Per-inode dax map */ +struct fuse_inode_dax { + /* Semaphore to protect modifications to the dmap tree */ + struct rw_semaphore sem; + + /* Sorted rb tree of struct fuse_dax_mapping elements */ + struct rb_root_cached tree; + unsigned long nr; }; struct fuse_conn_dax { /* DAX device */ struct dax_device *dev; + /* Lock protecting accessess to members of this structure */ + spinlock_t lock; + /* DAX Window Free Ranges */ long nr_free_ranges; struct list_head free_ranges; }; +static inline struct fuse_dax_mapping * +node_to_dmap(struct interval_tree_node *node) +{ + if (!node) + return NULL; + + return container_of(node, struct fuse_dax_mapping, itn); +} + +static struct fuse_dax_mapping *alloc_dax_mapping(struct fuse_conn_dax *fcd) +{ + struct fuse_dax_mapping *dmap; + + spin_lock(&fcd->lock); + dmap = list_first_entry_or_null(&fcd->free_ranges, + struct fuse_dax_mapping, list); + if (dmap) { + list_del_init(&dmap->list); + WARN_ON(fcd->nr_free_ranges <= 0); + fcd->nr_free_ranges--; + } + spin_unlock(&fcd->lock); + return dmap; +} + +/* This assumes fcd->lock is held */ +static void __dmap_add_to_free_pool(struct fuse_conn_dax *fcd, + struct fuse_dax_mapping *dmap) +{ + list_add_tail(&dmap->list, &fcd->free_ranges); + fcd->nr_free_ranges++; +} + +static void dmap_add_to_free_pool(struct fuse_conn_dax *fcd, + struct fuse_dax_mapping *dmap) +{ + /* Return fuse_dax_mapping to free list */ + spin_lock(&fcd->lock); + __dmap_add_to_free_pool(fcd, dmap); + spin_unlock(&fcd->lock); +} + +static int fuse_setup_one_mapping(struct inode *inode, unsigned long start_idx, + struct fuse_dax_mapping *dmap, bool writable, + bool upgrade) +{ + struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_conn_dax *fcd = fc->dax; + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_setupmapping_in inarg; + loff_t offset = start_idx << FUSE_DAX_SHIFT; + FUSE_ARGS(args); + ssize_t err; + + WARN_ON(fcd->nr_free_ranges < 0); + + /* Ask fuse daemon to setup mapping */ + memset(&inarg, 0, sizeof(inarg)); + inarg.foffset = offset; + inarg.fh = -1; + inarg.moffset = dmap->window_offset; + inarg.len = FUSE_DAX_SZ; + inarg.flags |= FUSE_SETUPMAPPING_FLAG_READ; + if (writable) + inarg.flags |= FUSE_SETUPMAPPING_FLAG_WRITE; + args.opcode = FUSE_SETUPMAPPING; + args.nodeid = fi->nodeid; + args.in_numargs = 1; + args.in_args[0].size = sizeof(inarg); + args.in_args[0].value = &inarg; + err = fuse_simple_request(fc, &args); + if (err < 0) + return err; + dmap->writable = writable; + if (!upgrade) { + dmap->itn.start = dmap->itn.last = start_idx; + /* Protected by fi->dax->sem */ + interval_tree_insert(&dmap->itn, &fi->dax->tree); + fi->dax->nr++; + } + return 0; +} + +static int fuse_send_removemapping(struct inode *inode, + struct fuse_removemapping_in *inargp, + struct fuse_removemapping_one *remove_one) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_conn *fc = get_fuse_conn(inode); + FUSE_ARGS(args); + + args.opcode = FUSE_REMOVEMAPPING; + args.nodeid = fi->nodeid; + args.in_numargs = 2; + args.in_args[0].size = sizeof(*inargp); + args.in_args[0].value = inargp; + args.in_args[1].size = inargp->count * sizeof(*remove_one); + args.in_args[1].value = remove_one; + return fuse_simple_request(fc, &args); +} + +static int dmap_removemapping_list(struct inode *inode, unsigned int num, + struct list_head *to_remove) +{ + struct fuse_removemapping_one *remove_one, *ptr; + struct fuse_removemapping_in inarg; + struct fuse_dax_mapping *dmap; + int ret, i = 0, nr_alloc; + + nr_alloc = min_t(unsigned int, num, FUSE_REMOVEMAPPING_MAX_ENTRY); + remove_one = kmalloc_array(nr_alloc, sizeof(*remove_one), GFP_NOFS); + if (!remove_one) + return -ENOMEM; + + ptr = remove_one; + list_for_each_entry(dmap, to_remove, list) { + ptr->moffset = dmap->window_offset; + ptr->len = dmap->length; + ptr++; + i++; + num--; + if (i >= nr_alloc || num == 0) { + memset(&inarg, 0, sizeof(inarg)); + inarg.count = i; + ret = fuse_send_removemapping(inode, &inarg, + remove_one); + if (ret) + goto out; + ptr = remove_one; + i = 0; + } + } +out: + kfree(remove_one); + return ret; +} + +/* + * Cleanup dmap entry and add back to free list. This should be called with + * fcd->lock held. + */ +static void dmap_reinit_add_to_free_pool(struct fuse_conn_dax *fcd, + struct fuse_dax_mapping *dmap) +{ + pr_debug("fuse: freeing memory range start_idx=0x%lx end_idx=0x%lx window_offset=0x%llx length=0x%llx\n", + dmap->itn.start, dmap->itn.last, dmap->window_offset, + dmap->length); + dmap->itn.start = dmap->itn.last = 0; + __dmap_add_to_free_pool(fcd, dmap); +} + +/* + * Free inode dmap entries whose range falls inside [start, end]. + * Does not take any locks. At this point of time it should only be + * called from evict_inode() path where we know all dmap entries can be + * reclaimed. + */ +static void inode_reclaim_dmap_range(struct fuse_conn_dax *fcd, + struct inode *inode, + loff_t start, loff_t end) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_dax_mapping *dmap, *n; + int err, num = 0; + LIST_HEAD(to_remove); + unsigned long start_idx = start >> FUSE_DAX_SHIFT; + unsigned long end_idx = end >> FUSE_DAX_SHIFT; + struct interval_tree_node *node; + + while (1) { + node = interval_tree_iter_first(&fi->dax->tree, start_idx, + end_idx); + if (!node) + break; + dmap = node_to_dmap(node); + interval_tree_remove(&dmap->itn, &fi->dax->tree); + num++; + list_add(&dmap->list, &to_remove); + } + + /* Nothing to remove */ + if (list_empty(&to_remove)) + return; + + WARN_ON(fi->dax->nr < num); + fi->dax->nr -= num; + err = dmap_removemapping_list(inode, num, &to_remove); + if (err && err != -ENOTCONN) { + pr_warn("Failed to removemappings. start=0x%llx end=0x%llx\n", + start, end); + } + spin_lock(&fcd->lock); + list_for_each_entry_safe(dmap, n, &to_remove, list) { + list_del_init(&dmap->list); + dmap_reinit_add_to_free_pool(fcd, dmap); + } + spin_unlock(&fcd->lock); +} + +/* + * It is called from evict_inode() and by that time inode is going away. So + * this function does not take any locks like fi->dax->sem for traversing + * that fuse inode interval tree. If that lock is taken then lock validator + * complains of deadlock situation w.r.t fs_reclaim lock. + */ +void fuse_dax_inode_cleanup(struct inode *inode) +{ + struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_inode *fi = get_fuse_inode(inode); + + /* + * fuse_evict_inode() has already called truncate_inode_pages_final() + * before we arrive here. So we should not have to worry about any + * pages/exception entries still associated with inode. + */ + inode_reclaim_dmap_range(fc->dax, inode, 0, -1); + WARN_ON(fi->dax->nr); +} + +static void fuse_fill_iomap_hole(struct iomap *iomap, loff_t length) +{ + iomap->addr = IOMAP_NULL_ADDR; + iomap->length = length; + iomap->type = IOMAP_HOLE; +} + +static void fuse_fill_iomap(struct inode *inode, loff_t pos, loff_t length, + struct iomap *iomap, struct fuse_dax_mapping *dmap, + unsigned int flags) +{ + loff_t offset, len; + loff_t i_size = i_size_read(inode); + + offset = pos - (dmap->itn.start << FUSE_DAX_SHIFT); + len = min(length, dmap->length - offset); + + /* If length is beyond end of file, truncate further */ + if (pos + len > i_size) + len = i_size - pos; + + if (len > 0) { + iomap->addr = dmap->window_offset + offset; + iomap->length = len; + if (flags & IOMAP_FAULT) + iomap->length = ALIGN(len, PAGE_SIZE); + iomap->type = IOMAP_MAPPED; + } else { + /* Mapping beyond end of file is hole */ + fuse_fill_iomap_hole(iomap, length); + } +} + +static int fuse_setup_new_dax_mapping(struct inode *inode, loff_t pos, + loff_t length, unsigned int flags, + struct iomap *iomap) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_conn_dax *fcd = fc->dax; + struct fuse_dax_mapping *dmap, *alloc_dmap = NULL; + int ret; + bool writable = flags & IOMAP_WRITE; + unsigned long start_idx = pos >> FUSE_DAX_SHIFT; + struct interval_tree_node *node; + + alloc_dmap = alloc_dax_mapping(fcd); + if (!alloc_dmap) + return -EIO; + + /* + * Take write lock so that only one caller can try to setup mapping + * and other waits. + */ + down_write(&fi->dax->sem); + /* + * We dropped lock. Check again if somebody else setup + * mapping already. + */ + node = interval_tree_iter_first(&fi->dax->tree, start_idx, start_idx); + if (node) { + dmap = node_to_dmap(node); + fuse_fill_iomap(inode, pos, length, iomap, dmap, flags); + dmap_add_to_free_pool(fcd, alloc_dmap); + up_write(&fi->dax->sem); + return 0; + } + + /* Setup one mapping */ + ret = fuse_setup_one_mapping(inode, pos >> FUSE_DAX_SHIFT, alloc_dmap, + writable, false); + if (ret < 0) { + dmap_add_to_free_pool(fcd, alloc_dmap); + up_write(&fi->dax->sem); + return ret; + } + fuse_fill_iomap(inode, pos, length, iomap, alloc_dmap, flags); + up_write(&fi->dax->sem); + return 0; +} + +static int fuse_upgrade_dax_mapping(struct inode *inode, loff_t pos, + loff_t length, unsigned int flags, + struct iomap *iomap) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_dax_mapping *dmap; + int ret; + unsigned long idx = pos >> FUSE_DAX_SHIFT; + struct interval_tree_node *node; + + /* + * Take exclusive lock so that only one caller can try to setup + * mapping and others wait. + */ + down_write(&fi->dax->sem); + node = interval_tree_iter_first(&fi->dax->tree, idx, idx); + + /* We are holding either inode lock or i_mmap_sem, and that should + * ensure that dmap can't reclaimed or truncated and it should still + * be there in tree despite the fact we dropped and re-acquired the + * lock. + */ + ret = -EIO; + if (WARN_ON(!node)) + goto out_err; + + dmap = node_to_dmap(node); + + /* Maybe another thread already upgraded mapping while we were not + * holding lock. + */ + if (dmap->writable) { + ret = 0; + goto out_fill_iomap; + } + + ret = fuse_setup_one_mapping(inode, pos >> FUSE_DAX_SHIFT, dmap, true, + true); + if (ret < 0) + goto out_err; +out_fill_iomap: + fuse_fill_iomap(inode, pos, length, iomap, dmap, flags); +out_err: + up_write(&fi->dax->sem); + return ret; +} + +/* This is just for DAX and the mapping is ephemeral, do not use it for other + * purposes since there is no block device with a permanent mapping. + */ +static int fuse_iomap_begin(struct inode *inode, loff_t pos, loff_t length, + unsigned int flags, struct iomap *iomap, + struct iomap *srcmap) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_dax_mapping *dmap; + bool writable = flags & IOMAP_WRITE; + unsigned long start_idx = pos >> FUSE_DAX_SHIFT; + struct interval_tree_node *node; + + /* We don't support FIEMAP */ + if (WARN_ON(flags & IOMAP_REPORT)) + return -EIO; + + iomap->offset = pos; + iomap->flags = 0; + iomap->bdev = NULL; + iomap->dax_dev = fc->dax->dev; + + /* + * Both read/write and mmap path can race here. So we need something + * to make sure if we are setting up mapping, then other path waits + * + * For now, use a semaphore for this. It probably needs to be + * optimized later. + */ + down_read(&fi->dax->sem); + node = interval_tree_iter_first(&fi->dax->tree, start_idx, start_idx); + if (node) { + dmap = node_to_dmap(node); + if (writable && !dmap->writable) { + /* Upgrade read-only mapping to read-write. This will + * require exclusive fi->dax->sem lock as we don't want + * two threads to be trying to this simultaneously + * for same dmap. So drop shared lock and acquire + * exclusive lock. + */ + up_read(&fi->dax->sem); + pr_debug("%s: Upgrading mapping at offset 0x%llx length 0x%llx\n", + __func__, pos, length); + return fuse_upgrade_dax_mapping(inode, pos, length, + flags, iomap); + } else { + fuse_fill_iomap(inode, pos, length, iomap, dmap, flags); + up_read(&fi->dax->sem); + return 0; + } + } else { + up_read(&fi->dax->sem); + pr_debug("%s: no mapping at offset 0x%llx length 0x%llx\n", + __func__, pos, length); + if (pos >= i_size_read(inode)) + goto iomap_hole; + + return fuse_setup_new_dax_mapping(inode, pos, length, flags, + iomap); + } + + /* + * If read beyond end of file happnes, fs code seems to return + * it as hole + */ +iomap_hole: + fuse_fill_iomap_hole(iomap, length); + pr_debug("%s returning hole mapping. pos=0x%llx length_asked=0x%llx length_returned=0x%llx\n", + __func__, pos, length, iomap->length); + return 0; +} + +static int fuse_iomap_end(struct inode *inode, loff_t pos, loff_t length, + ssize_t written, unsigned int flags, + struct iomap *iomap) +{ + /* DAX writes beyond end-of-file aren't handled using iomap, so the + * file size is unchanged and there is nothing to do here. + */ + return 0; +} + +static const struct iomap_ops fuse_iomap_ops = { + .iomap_begin = fuse_iomap_begin, + .iomap_end = fuse_iomap_end, +}; + +ssize_t fuse_dax_read_iter(struct kiocb *iocb, struct iov_iter *to) +{ + struct inode *inode = file_inode(iocb->ki_filp); + ssize_t ret; + + if (iocb->ki_flags & IOCB_NOWAIT) { + if (!inode_trylock_shared(inode)) + return -EAGAIN; + } else { + inode_lock_shared(inode); + } + + ret = dax_iomap_rw(iocb, to, &fuse_iomap_ops); + inode_unlock_shared(inode); + + /* TODO file_accessed(iocb->f_filp) */ + return ret; +} + +static bool file_extending_write(struct kiocb *iocb, struct iov_iter *from) +{ + struct inode *inode = file_inode(iocb->ki_filp); + + return (iov_iter_rw(from) == WRITE && + ((iocb->ki_pos) >= i_size_read(inode) || + (iocb->ki_pos + iov_iter_count(from) > i_size_read(inode)))); +} + +static ssize_t fuse_dax_direct_write(struct kiocb *iocb, struct iov_iter *from) +{ + struct inode *inode = file_inode(iocb->ki_filp); + struct fuse_io_priv io = FUSE_IO_PRIV_SYNC(iocb); + ssize_t ret; + + ret = fuse_direct_io(&io, from, &iocb->ki_pos, FUSE_DIO_WRITE); + if (ret < 0) + return ret; + + fuse_invalidate_attr(inode); + fuse_write_update_size(inode, iocb->ki_pos); + return ret; +} + +ssize_t fuse_dax_write_iter(struct kiocb *iocb, struct iov_iter *from) +{ + struct inode *inode = file_inode(iocb->ki_filp); + ssize_t ret; + + if (iocb->ki_flags & IOCB_NOWAIT) { + if (!inode_trylock(inode)) + return -EAGAIN; + } else { + inode_lock(inode); + } + + ret = generic_write_checks(iocb, from); + if (ret <= 0) + goto out; + + ret = file_remove_privs(iocb->ki_filp); + if (ret) + goto out; + /* TODO file_update_time() but we don't want metadata I/O */ + + /* Do not use dax for file extending writes as write and on + * disk i_size increase are not atomic otherwise. + */ + if (file_extending_write(iocb, from)) + ret = fuse_dax_direct_write(iocb, from); + else + ret = dax_iomap_rw(iocb, from, &fuse_iomap_ops); + +out: + inode_unlock(inode); + + if (ret > 0) + ret = generic_write_sync(iocb, ret); + return ret; +} + static void fuse_free_dax_mem_ranges(struct list_head *mem_list) { struct fuse_dax_mapping *range, *temp; @@ -116,6 +653,7 @@ int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev) if (!fcd) return -ENOMEM; + spin_lock_init(&fcd->lock); fcd->dev = dax_dev; err = fuse_dax_mem_range_init(fcd); if (err) { @@ -127,6 +665,33 @@ int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev) return 0; } +bool fuse_dax_inode_alloc(struct super_block *sb, struct fuse_inode *fi) +{ + struct fuse_conn *fc = get_fuse_conn_super(sb); + + fi->dax = NULL; + if (fc->dax) { + fi->dax = kzalloc(sizeof(*fi->dax), GFP_KERNEL_ACCOUNT); + if (!fi->dax) + return false; + + init_rwsem(&fi->dax->sem); + fi->dax->tree = RB_ROOT_CACHED; + } + + return true; +} + +void fuse_dax_inode_init(struct inode *inode) +{ + struct fuse_conn *fc = get_fuse_conn(inode); + + if (!fc->dax) + return; + + inode->i_flags |= S_DAX; +} + bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment) { if (fc->dax && (map_alignment > FUSE_DAX_SHIFT)) { diff --git a/fs/fuse/file.c b/fs/fuse/file.c index a19cc87dac49..6a4d35f13b52 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1609,10 +1609,14 @@ static ssize_t fuse_file_read_iter(struct kiocb *iocb, struct iov_iter *to) { struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; + struct inode *inode = file_inode(file); - if (fuse_is_bad(file_inode(file))) + if (fuse_is_bad(inode)) return -EIO; + if (FUSE_IS_DAX(inode)) + return fuse_dax_read_iter(iocb, to); + if (ff->passthrough.filp) return fuse_passthrough_read_iter(iocb, to); else if (!(ff->open_flags & FOPEN_DIRECT_IO)) @@ -1625,10 +1629,14 @@ static ssize_t fuse_file_write_iter(struct kiocb *iocb, struct iov_iter *from) { struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; + struct inode *inode = file_inode(file); - if (fuse_is_bad(file_inode(file))) + if (fuse_is_bad(inode)) return -EIO; + if (FUSE_IS_DAX(inode)) + return fuse_dax_write_iter(iocb, from); + if (ff->passthrough.filp) return fuse_passthrough_write_iter(iocb, from); else if (!(ff->open_flags & FOPEN_DIRECT_IO)) @@ -3523,4 +3531,7 @@ void fuse_init_file_inode(struct inode *inode) fi->writectr = 0; init_waitqueue_head(&fi->page_waitq); fi->writepages = RB_ROOT; + + if (IS_ENABLED(CONFIG_FUSE_DAX)) + fuse_dax_inode_init(inode); } diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 42caa2061864..e77a010593fa 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -148,6 +148,13 @@ struct fuse_inode { /** Lock to protect write related fields */ spinlock_t lock; + +#ifdef CONFIG_FUSE_DAX + /* + * Dax specific inode data + */ + struct fuse_inode_dax *dax; +#endif }; /** FUSE inode state bits */ @@ -1144,8 +1151,16 @@ void fuse_free_conn(struct fuse_conn *fc); /* dax.c */ +#define FUSE_IS_DAX(inode) (IS_ENABLED(CONFIG_FUSE_DAX) && IS_DAX(inode)) + +ssize_t fuse_dax_read_iter(struct kiocb *iocb, struct iov_iter *to); +ssize_t fuse_dax_write_iter(struct kiocb *iocb, struct iov_iter *from); +int fuse_dax_mmap(struct file *file, struct vm_area_struct *vma); int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev); void fuse_dax_conn_free(struct fuse_conn *fc); +bool fuse_dax_inode_alloc(struct super_block *sb, struct fuse_inode *fi); +void fuse_dax_inode_init(struct inode *inode); +void fuse_dax_inode_cleanup(struct inode *inode); bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment); /* passthrough.c */ diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 38f3ac5eadec..652f5b2aa8e0 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -87,12 +87,19 @@ static struct inode *fuse_alloc_inode(struct super_block *sb) mutex_init(&fi->mutex); spin_lock_init(&fi->lock); fi->forget = fuse_alloc_forget(); - if (!fi->forget) { - kmem_cache_free(fuse_inode_cachep, fi); - return NULL; - } + if (!fi->forget) + goto out_free; + + if (IS_ENABLED(CONFIG_FUSE_DAX) && !fuse_dax_inode_alloc(sb, fi)) + goto out_free_forget; return &fi->inode; + +out_free_forget: + kfree(fi->forget); +out_free: + kmem_cache_free(fuse_inode_cachep, fi); + return NULL; } static void fuse_free_inode(struct inode *inode) @@ -101,6 +108,9 @@ static void fuse_free_inode(struct inode *inode) mutex_destroy(&fi->mutex); kfree(fi->forget); +#ifdef CONFIG_FUSE_DAX + kfree(fi->dax); +#endif kmem_cache_free(fuse_inode_cachep, fi); } @@ -112,6 +122,9 @@ static void fuse_evict_inode(struct inode *inode) clear_inode(inode); if (inode->i_sb->s_flags & SB_ACTIVE) { struct fuse_conn *fc = get_fuse_conn(inode); + + if (FUSE_IS_DAX(inode)) + fuse_dax_inode_cleanup(inode); fuse_queue_forget(fc, fi->forget, fi->nodeid, fi->nlookup); fi->forget = NULL; } diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index aafafe7a58db..37467e470242 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -901,6 +901,7 @@ struct fuse_copy_file_range_in { }; #define FUSE_SETUPMAPPING_FLAG_WRITE (1ull << 0) +#define FUSE_SETUPMAPPING_FLAG_READ (1ull << 1) struct fuse_setupmapping_in { /* An already open handle */ uint64_t fh; From f0e895b3e9424539acec2270b0d3244e7f77da2b Mon Sep 17 00:00:00 2001 From: Stefan Hajnoczi Date: Wed, 19 Aug 2020 18:19:52 -0400 Subject: [PATCH 033/148] UPSTREAM: virtiofs: add DAX mmap support Add DAX mmap() support. Change-Id: I80e46a9054db40ae56e797c070bf7ecc3c84e343 Signed-off-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi --- fs/fuse/dax.c | 59 ++++++++++++++++++++++++++++++++++++++++++++++++++ fs/fuse/file.c | 5 +++++ 2 files changed, 64 insertions(+) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index a8d311b2db8e..b8ec7d64aa68 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -575,6 +575,65 @@ out: return ret; } +static vm_fault_t __fuse_dax_fault(struct vm_fault *vmf, + enum page_entry_size pe_size, bool write) +{ + vm_fault_t ret; + struct inode *inode = file_inode(vmf->vma->vm_file); + struct super_block *sb = inode->i_sb; + pfn_t pfn; + + if (write) + sb_start_pagefault(sb); + + ret = dax_iomap_fault(vmf, pe_size, &pfn, NULL, &fuse_iomap_ops); + + if (ret & VM_FAULT_NEEDDSYNC) + ret = dax_finish_sync_fault(vmf, pe_size, pfn); + + if (write) + sb_end_pagefault(sb); + + return ret; +} + +static vm_fault_t fuse_dax_fault(struct vm_fault *vmf) +{ + return __fuse_dax_fault(vmf, PE_SIZE_PTE, + vmf->flags & FAULT_FLAG_WRITE); +} + +static vm_fault_t fuse_dax_huge_fault(struct vm_fault *vmf, + enum page_entry_size pe_size) +{ + return __fuse_dax_fault(vmf, pe_size, vmf->flags & FAULT_FLAG_WRITE); +} + +static vm_fault_t fuse_dax_page_mkwrite(struct vm_fault *vmf) +{ + return __fuse_dax_fault(vmf, PE_SIZE_PTE, true); +} + +static vm_fault_t fuse_dax_pfn_mkwrite(struct vm_fault *vmf) +{ + return __fuse_dax_fault(vmf, PE_SIZE_PTE, true); +} + +static const struct vm_operations_struct fuse_dax_vm_ops = { + .fault = fuse_dax_fault, + .huge_fault = fuse_dax_huge_fault, + .page_mkwrite = fuse_dax_page_mkwrite, + .pfn_mkwrite = fuse_dax_pfn_mkwrite, +}; + +int fuse_dax_mmap(struct file *file, struct vm_area_struct *vma) +{ + file_accessed(file); + vma->vm_ops = &fuse_dax_vm_ops; + vma->vm_flags |= VM_MIXEDMAP | VM_HUGEPAGE; + return 0; +} + static void fuse_free_dax_mem_ranges(struct list_head *mem_list) { struct fuse_dax_mapping *range, *temp; diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 6a4d35f13b52..fdf2806b860d 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -2405,6 +2405,10 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma) { struct fuse_file *ff = file->private_data; + /* DAX mmap is superior to direct_io mmap */ + if (FUSE_IS_DAX(file_inode(file))) + return fuse_dax_mmap(file, vma); + if (ff->passthrough.filp) return fuse_passthrough_mmap(file, vma); @@ -3496,6 +3500,7 @@ static const struct file_operations fuse_file_operations = { .release = fuse_release, .fsync = fuse_fsync, .lock = fuse_file_lock, + .get_unmapped_area = thp_get_unmapped_area, .flock = fuse_file_flock, .splice_read = generic_file_splice_read, .splice_write = iter_file_splice_write, From 58ba749543192344b30039fd4e211f688a739966 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:53 -0400 Subject: [PATCH 034/148] UPSTREAM: virtiofs: define dax address space operations This is done along the lines of ext4 and xfs. I primarily wanted ->writepages hook at this time so that I could call into dax_writeback_mapping_range(). This in turn will decide which pfns need to be written back. Change-Id: If7c2e595df71f433368a0a4a5a77eca6fcb7549d Signed-off-by: Vivek Goyal Signed-off-by: Miklos Szeredi --- fs/fuse/dax.c | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index b8ec7d64aa68..c21305409191 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -575,6 +575,16 @@ out: return ret; } +static int fuse_dax_writepages(struct address_space *mapping, + struct writeback_control *wbc) +{ + + struct inode *inode = mapping->host; + struct fuse_conn *fc = get_fuse_conn(inode); + + return dax_writeback_mapping_range(mapping, fc->dax->dev, wbc); +} + static vm_fault_t __fuse_dax_fault(struct vm_fault *vmf, enum page_entry_size pe_size, bool write) { @@ -741,6 +751,13 @@ bool fuse_dax_inode_alloc(struct super_block *sb, struct fuse_inode *fi) return true; } +static const struct address_space_operations fuse_dax_file_aops = { + .writepages = fuse_dax_writepages, + .direct_IO = noop_direct_IO, + .set_page_dirty = noop_set_page_dirty, + .invalidatepage = noop_invalidatepage, +}; + void fuse_dax_inode_init(struct inode *inode) { struct fuse_conn *fc = get_fuse_conn(inode); @@ -749,6 +766,7 @@ void fuse_dax_inode_init(struct inode *inode) return; inode->i_flags |= S_DAX; + inode->i_data.a_ops = &fuse_dax_file_aops; } bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment) From 811ed4644f3f9875bdd453288a6edd9318701077 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:54 -0400 Subject: [PATCH 035/148] virtiofs: serialize truncate/punch_hole and dax fault path Currently in fuse we don't seem have any lock which can serialize fault path with truncate/punch_hole path. With dax support I need one for following reasons. 1. Dax requirement DAX fault code relies on inode size being stable for the duration of fault and want to serialize with truncate/punch_hole and they explicitly mention it. static vm_fault_t dax_iomap_pmd_fault(struct vm_fault *vmf, pfn_t *pfnp, const struct iomap_ops *ops) /* * Check whether offset isn't beyond end of file now. Caller is * supposed to hold locks serializing us with truncate / punch hole so * this is a reliable test. */ max_pgoff = DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE); 2. Make sure there are no users of pages being truncated/punch_hole get_user_pages() might take references to page and then do some DMA to said pages. Filesystem might truncate those pages without knowing that a DMA is in progress or some I/O is in progress. So use dax_layout_busy_page() to make sure there are no such references and I/O is not in progress on said pages before moving ahead with truncation. 3. Limitation of kvm page fault error reporting If we are truncating file on host first and then removing mappings in guest lateter (truncate page cache etc), then this could lead to a problem with KVM. Say a mapping is in place in guest and truncation happens on host. Now if guest accesses that mapping, then host will take a fault and kvm will either exit to qemu or spin infinitely. IOW, before we do truncation on host, we need to make sure that guest inode does not have any mapping in that region or whole file. 4. virtiofs memory range reclaim Soon I will introduce the notion of being able to reclaim dax memory ranges from a fuse dax inode. There also I need to make sure that no I/O or fault is going on in the reclaimed range and nobody is using it so that range can be reclaimed without issues. Currently if we take inode lock, that serializes read/write. But it does not do anything for faults. So I add another semaphore fuse_inode->i_mmap_sem for this purpose. It can be used to serialize with faults. As of now, I am adding taking this semaphore only in dax fault path and not regular fault path because existing code does not have one. May be existing code can benefit from it as well to take care of some races, but that we can fix later if need be. For now, I am just focussing only on DAX path which is new path. Also added logic to take fuse_inode->i_mmap_sem in truncate/punch_hole/open(O_TRUNC) path to make sure file truncation and fuse dax fault are mutually exlusive and avoid all the above problems. Change-Id: I156a034a6d2e70fb90372815dfb5f61f033b7cb8 Signed-off-by: Vivek Goyal Cc: Dave Chinner Signed-off-by: Miklos Szeredi --- fs/fuse/dax.c | 49 ++++++++++++++++++++++++++++++++++++++++++++++++ fs/fuse/dir.c | 32 ++++++++++++++++++++++++------- fs/fuse/file.c | 32 ++++++++++++++++++++++++++++--- fs/fuse/fuse_i.h | 8 ++++++++ fs/fuse/inode.c | 1 + 5 files changed, 112 insertions(+), 10 deletions(-) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index c21305409191..efc1a8ef5639 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -495,6 +495,47 @@ static const struct iomap_ops fuse_iomap_ops = { .iomap_end = fuse_iomap_end, }; +static void fuse_wait_dax_page(struct inode *inode) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + + up_write(&fi->i_mmap_sem); + schedule(); + down_write(&fi->i_mmap_sem); +} + +/* Should be called with fi->i_mmap_sem lock held exclusively */ +static int __fuse_dax_break_layouts(struct inode *inode, bool *retry, + loff_t start, loff_t end) +{ + struct page *page; + + page = dax_layout_busy_page_range(inode->i_mapping, start, end); + if (!page) + return 0; + + *retry = true; + return ___wait_var_event(&page->_refcount, + atomic_read(&page->_refcount) == 1, TASK_INTERRUPTIBLE, + 0, 0, fuse_wait_dax_page(inode)); +} + +/* dmap_end == 0 leads to unmapping of whole file */ +int fuse_dax_break_layouts(struct inode *inode, u64 dmap_start, + u64 dmap_end) +{ + bool retry; + int ret; + + do { + retry = false; + ret = __fuse_dax_break_layouts(inode, &retry, dmap_start, + dmap_end); + } while (ret == 0 && retry); + + return ret; +} + ssize_t fuse_dax_read_iter(struct kiocb *iocb, struct iov_iter *to) { struct inode *inode = file_inode(iocb->ki_filp); @@ -596,10 +637,18 @@ static vm_fault_t __fuse_dax_fault(struct vm_fault *vmf, if (write) sb_start_pagefault(sb); + /* + * We need to serialize against not only truncate but also against + * fuse dax memory range reclaim. While a range is being reclaimed, + * we do not want any read/write/mmap to make progress and try + * to populate page cache or access memory we are trying to free. + */ + down_read(&get_fuse_inode(inode)->i_mmap_sem); ret = dax_iomap_fault(vmf, pe_size, &pfn, NULL, &fuse_iomap_ops); if (ret & VM_FAULT_NEEDDSYNC) ret = dax_finish_sync_fault(vmf, pe_size, pfn); + up_read(&get_fuse_inode(inode)->i_mmap_sem); if (write) sb_end_pagefault(sb); diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index de70d6983a96..60f7affcbde0 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -1560,6 +1560,7 @@ int fuse_do_setattr(struct dentry *dentry, struct iattr *attr, loff_t oldsize; int err; bool trust_local_cmtime = is_wb && S_ISREG(inode->i_mode); + bool fault_blocked = false; if (!fc->default_permissions) attr->ia_valid |= ATTR_FORCE; @@ -1568,6 +1569,22 @@ int fuse_do_setattr(struct dentry *dentry, struct iattr *attr, if (err) return err; + if (attr->ia_valid & ATTR_SIZE) { + if (WARN_ON(!S_ISREG(inode->i_mode))) + return -EIO; + is_truncate = true; + } + + if (FUSE_IS_DAX(inode) && is_truncate) { + down_write(&fi->i_mmap_sem); + fault_blocked = true; + err = fuse_dax_break_layouts(inode, 0, 0); + if (err) { + up_write(&fi->i_mmap_sem); + return err; + } + } + if (attr->ia_valid & ATTR_OPEN) { /* This is coming from open(..., ... | O_TRUNC); */ WARN_ON(!(attr->ia_valid & ATTR_SIZE)); @@ -1580,17 +1597,11 @@ int fuse_do_setattr(struct dentry *dentry, struct iattr *attr, */ i_size_write(inode, 0); truncate_pagecache(inode, 0); - return 0; + goto out; } file = NULL; } - if (attr->ia_valid & ATTR_SIZE) { - if (WARN_ON(!S_ISREG(inode->i_mode))) - return -EIO; - is_truncate = true; - } - /* Flush dirty data/metadata before non-truncate SETATTR */ if (is_wb && S_ISREG(inode->i_mode) && attr->ia_valid & @@ -1673,6 +1684,10 @@ int fuse_do_setattr(struct dentry *dentry, struct iattr *attr, } clear_bit(FUSE_I_SIZE_UNSTABLE, &fi->state); +out: + if (fault_blocked) + up_write(&fi->i_mmap_sem); + return 0; error: @@ -1680,6 +1695,9 @@ error: fuse_release_nowrite(inode); clear_bit(FUSE_I_SIZE_UNSTABLE, &fi->state); + + if (fault_blocked) + up_write(&fi->i_mmap_sem); return err; } diff --git a/fs/fuse/file.c b/fs/fuse/file.c index fdf2806b860d..a3ce2e78b7d5 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -224,6 +224,8 @@ int fuse_open_common(struct inode *inode, struct file *file, bool isdir) bool is_wb_truncate = (file->f_flags & O_TRUNC) && fc->atomic_o_trunc && fc->writeback_cache; + bool dax_truncate = (file->f_flags & O_TRUNC) && + fc->atomic_o_trunc && FUSE_IS_DAX(inode); if (fuse_is_bad(inode)) return -EIO; @@ -232,17 +234,27 @@ int fuse_open_common(struct inode *inode, struct file *file, bool isdir) if (err) return err; - if (is_wb_truncate) { + if (is_wb_truncate || dax_truncate) { inode_lock(inode); fuse_set_nowrite(inode); } - err = fuse_do_open(fc, get_node_id(inode), file, isdir); + if (dax_truncate) { + down_write(&get_fuse_inode(inode)->i_mmap_sem); + err = fuse_dax_break_layouts(inode, 0, 0); + if (err) + goto out; + } + err = fuse_do_open(fc, get_node_id(inode), file, isdir); if (!err) fuse_finish_open(inode, file); - if (is_wb_truncate) { +out: + if (dax_truncate) + up_write(&get_fuse_inode(inode)->i_mmap_sem); + + if (is_wb_truncate | dax_truncate) { fuse_release_nowrite(inode); inode_unlock(inode); } @@ -3300,6 +3312,10 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, .mode = mode }; int err; + bool block_faults = FUSE_IS_DAX(inode) && + (!(mode & FALLOC_FL_KEEP_SIZE) || + (mode & FALLOC_FL_PUNCH_HOLE)); + if (mode & ~(FALLOC_FL_KEEP_SIZE | FALLOC_FL_PUNCH_HOLE)) return -EOPNOTSUPP; @@ -3307,6 +3323,13 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, return -EOPNOTSUPP; inode_lock(inode); + if (block_faults) { + down_write(&fi->i_mmap_sem); + err = fuse_dax_break_layouts(inode, 0, 0); + if (err) + goto out; + } + if (mode & FALLOC_FL_PUNCH_HOLE) { loff_t endbyte = offset + length - 1; @@ -3359,6 +3382,9 @@ out: if (!(mode & FALLOC_FL_KEEP_SIZE)) clear_bit(FUSE_I_SIZE_UNSTABLE, &fi->state); + if (block_faults) + up_write(&fi->i_mmap_sem); + inode_unlock(inode); return err; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index e77a010593fa..94a9b0b9150b 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -149,6 +149,13 @@ struct fuse_inode { /** Lock to protect write related fields */ spinlock_t lock; + /** + * Can't take inode lock in fault path (leads to circular dependency). + * Introduce another semaphore which can be taken in fault path and + * then other filesystem paths can take this to block faults. + */ + struct rw_semaphore i_mmap_sem; + #ifdef CONFIG_FUSE_DAX /* * Dax specific inode data @@ -1156,6 +1163,7 @@ void fuse_free_conn(struct fuse_conn *fc); ssize_t fuse_dax_read_iter(struct kiocb *iocb, struct iov_iter *to); ssize_t fuse_dax_write_iter(struct kiocb *iocb, struct iov_iter *from); int fuse_dax_mmap(struct file *file, struct vm_area_struct *vma); +int fuse_dax_break_layouts(struct inode *inode, u64 dmap_start, u64 dmap_end); int fuse_dax_conn_alloc(struct fuse_conn *fc, struct dax_device *dax_dev); void fuse_dax_conn_free(struct fuse_conn *fc); bool fuse_dax_inode_alloc(struct super_block *sb, struct fuse_inode *fi); diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 652f5b2aa8e0..708e6e757dfe 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -85,6 +85,7 @@ static struct inode *fuse_alloc_inode(struct super_block *sb) fi->orig_ino = 0; fi->state = 0; mutex_init(&fi->mutex); + init_rwsem(&fi->i_mmap_sem); spin_lock_init(&fi->lock); fi->forget = fuse_alloc_forget(); if (!fi->forget) From f5b74e7e643486a7464d7a79fd65e8cf65315629 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:55 -0400 Subject: [PATCH 036/148] UPSTREAM: virtiofs: maintain a list of busy elements This list will be used selecting fuse_dax_mapping to free when number of free mappings drops below a threshold. Change-Id: I311dba8e084206ed4b983779674ee1b160d7a757 Signed-off-by: Vivek Goyal Signed-off-by: Miklos Szeredi --- fs/fuse/dax.c | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index efc1a8ef5639..e6407839c538 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -28,6 +28,9 @@ struct fuse_dax_mapping { /* For interval tree in file/inode */ struct interval_tree_node itn; + /* Will connect in fc->busy_ranges to keep track busy memory */ + struct list_head busy_list; + /** Position in DAX window */ u64 window_offset; @@ -55,6 +58,10 @@ struct fuse_conn_dax { /* Lock protecting accessess to members of this structure */ spinlock_t lock; + /* List of memory ranges which are busy */ + unsigned long nr_busy_ranges; + struct list_head busy_ranges; + /* DAX Window Free Ranges */ long nr_free_ranges; struct list_head free_ranges; @@ -85,6 +92,15 @@ static struct fuse_dax_mapping *alloc_dax_mapping(struct fuse_conn_dax *fcd) return dmap; } +/* This assumes fcd->lock is held */ +static void __dmap_remove_busy_list(struct fuse_conn_dax *fcd, + struct fuse_dax_mapping *dmap) +{ + list_del_init(&dmap->busy_list); + WARN_ON(fcd->nr_busy_ranges == 0); + fcd->nr_busy_ranges--; +} + /* This assumes fcd->lock is held */ static void __dmap_add_to_free_pool(struct fuse_conn_dax *fcd, struct fuse_dax_mapping *dmap) @@ -139,6 +155,10 @@ static int fuse_setup_one_mapping(struct inode *inode, unsigned long start_idx, /* Protected by fi->dax->sem */ interval_tree_insert(&dmap->itn, &fi->dax->tree); fi->dax->nr++; + spin_lock(&fcd->lock); + list_add_tail(&dmap->busy_list, &fcd->busy_ranges); + fcd->nr_busy_ranges++; + spin_unlock(&fcd->lock); } return 0; } @@ -207,6 +227,7 @@ static void dmap_reinit_add_to_free_pool(struct fuse_conn_dax *fcd, pr_debug("fuse: freeing memory range start_idx=0x%lx end_idx=0x%lx window_offset=0x%llx length=0x%llx\n", dmap->itn.start, dmap->itn.last, dmap->window_offset, dmap->length); + __dmap_remove_busy_list(fcd, dmap); dmap->itn.start = dmap->itn.last = 0; __dmap_add_to_free_pool(fcd, dmap); } @@ -700,6 +721,8 @@ static void fuse_free_dax_mem_ranges(struct list_head *mem_list) /* Free All allocated elements */ list_for_each_entry_safe(range, temp, mem_list, list) { list_del(&range->list); + if (!list_empty(&range->busy_list)) + list_del(&range->busy_list); kfree(range); } } @@ -723,6 +746,7 @@ static int fuse_dax_mem_range_init(struct fuse_conn_dax *fcd) unsigned long i; INIT_LIST_HEAD(&fcd->free_ranges); + INIT_LIST_HEAD(&fcd->busy_ranges); id = dax_read_lock(); nr_pages = dax_direct_access(fcd->dev, 0, PHYS_PFN(dax_size), &kaddr, &pfn); @@ -748,6 +772,7 @@ static int fuse_dax_mem_range_init(struct fuse_conn_dax *fcd) */ range->window_offset = i * FUSE_DAX_SZ; range->length = FUSE_DAX_SZ; + INIT_LIST_HEAD(&range->busy_list); list_add_tail(&range->list, &fcd->free_ranges); } From 99d630bf582b9f8af53ad76dc7b3144132b93b9b Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Wed, 19 Aug 2020 18:19:56 -0400 Subject: [PATCH 037/148] UPSTREAM: virtiofs: add logic to free up a memory range Add logic to free up a busy memory range. Freed memory range will be returned to free pool. Add a worker which can be started to select and free some busy memory ranges. Process can also steal one of its busy dax ranges if free range is not available. I will refer it to as direct reclaim. If free range is not available and nothing can't be stolen from same inode, caller waits on a waitq for free range to become available. For reclaiming a range, as of now we need to hold following locks in specified order. down_write(&fi->i_mmap_sem); down_write(&fi->dax->sem); We look for a free range in following order. A. Try to get a free range. B. If not, try direct reclaim. C. If not, wait for a memory range to become free Change-Id: Idea3464f4c14faabce59fdd493c10f31b9dbd0ae Signed-off-by: Vivek Goyal Signed-off-by: Liu Bo Signed-off-by: Miklos Szeredi --- fs/fuse/dax.c | 523 +++++++++++++++++++++++++++++++++++++++++++- fs/fuse/fuse_i.h | 1 + fs/fuse/virtio_fs.c | 6 + 3 files changed, 524 insertions(+), 6 deletions(-) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index e6407839c538..e394dba08cc4 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -6,6 +6,7 @@ #include "fuse_i.h" +#include #include #include #include @@ -20,8 +21,21 @@ #define FUSE_DAX_SZ (1 << FUSE_DAX_SHIFT) #define FUSE_DAX_PAGES (FUSE_DAX_SZ / PAGE_SIZE) +/* Number of ranges reclaimer will try to free in one invocation */ +#define FUSE_DAX_RECLAIM_CHUNK (10) + +/* + * Dax memory reclaim threshold in percetage of total ranges. When free + * number of free ranges drops below this threshold, reclaim can trigger + * Default is 20% + */ +#define FUSE_DAX_RECLAIM_THRESHOLD (20) + /** Translation information for file offsets to DAX window offsets */ struct fuse_dax_mapping { + /* Pointer to inode where this memory range is mapped */ + struct inode *inode; + /* Will connect in fcd->free_ranges to keep track of free memory */ struct list_head list; @@ -39,6 +53,9 @@ struct fuse_dax_mapping { /* Is this mapping read-only or read-write */ bool writable; + + /* reference count when the mapping is used by dax iomap. */ + refcount_t refcnt; }; /* Per-inode dax map */ @@ -62,9 +79,17 @@ struct fuse_conn_dax { unsigned long nr_busy_ranges; struct list_head busy_ranges; + /* Worker to free up memory ranges */ + struct delayed_work free_work; + + /* Wait queue for a dax range to become free */ + wait_queue_head_t range_waitq; + /* DAX Window Free Ranges */ long nr_free_ranges; struct list_head free_ranges; + + unsigned long nr_ranges; }; static inline struct fuse_dax_mapping * @@ -76,6 +101,30 @@ node_to_dmap(struct interval_tree_node *node) return container_of(node, struct fuse_dax_mapping, itn); } +static struct fuse_dax_mapping * +alloc_dax_mapping_reclaim(struct fuse_conn_dax *fcd, struct inode *inode); + +static void +__kick_dmap_free_worker(struct fuse_conn_dax *fcd, unsigned long delay_ms) +{ + unsigned long free_threshold; + + /* If number of free ranges are below threshold, start reclaim */ + free_threshold = max_t(unsigned long, fcd->nr_ranges * FUSE_DAX_RECLAIM_THRESHOLD / 100, + 1); + if (fcd->nr_free_ranges < free_threshold) + queue_delayed_work(system_long_wq, &fcd->free_work, + msecs_to_jiffies(delay_ms)); +} + +static void kick_dmap_free_worker(struct fuse_conn_dax *fcd, + unsigned long delay_ms) +{ + spin_lock(&fcd->lock); + __kick_dmap_free_worker(fcd, delay_ms); + spin_unlock(&fcd->lock); +} + static struct fuse_dax_mapping *alloc_dax_mapping(struct fuse_conn_dax *fcd) { struct fuse_dax_mapping *dmap; @@ -89,6 +138,8 @@ static struct fuse_dax_mapping *alloc_dax_mapping(struct fuse_conn_dax *fcd) fcd->nr_free_ranges--; } spin_unlock(&fcd->lock); + + kick_dmap_free_worker(fcd, 0); return dmap; } @@ -101,12 +152,21 @@ static void __dmap_remove_busy_list(struct fuse_conn_dax *fcd, fcd->nr_busy_ranges--; } +static void dmap_remove_busy_list(struct fuse_conn_dax *fcd, + struct fuse_dax_mapping *dmap) +{ + spin_lock(&fcd->lock); + __dmap_remove_busy_list(fcd, dmap); + spin_unlock(&fcd->lock); +} + /* This assumes fcd->lock is held */ static void __dmap_add_to_free_pool(struct fuse_conn_dax *fcd, struct fuse_dax_mapping *dmap) { list_add_tail(&dmap->list, &fcd->free_ranges); fcd->nr_free_ranges++; + wake_up(&fcd->range_waitq); } static void dmap_add_to_free_pool(struct fuse_conn_dax *fcd, @@ -151,6 +211,12 @@ static int fuse_setup_one_mapping(struct inode *inode, unsigned long start_idx, return err; dmap->writable = writable; if (!upgrade) { + /* + * We don't take a refernce on inode. inode is valid right now + * and when inode is going away, cleanup logic should first + * cleanup dmap entries. + */ + dmap->inode = inode; dmap->itn.start = dmap->itn.last = start_idx; /* Protected by fi->dax->sem */ interval_tree_insert(&dmap->itn, &fi->dax->tree); @@ -228,6 +294,7 @@ static void dmap_reinit_add_to_free_pool(struct fuse_conn_dax *fcd, dmap->itn.start, dmap->itn.last, dmap->window_offset, dmap->length); __dmap_remove_busy_list(fcd, dmap); + dmap->inode = NULL; dmap->itn.start = dmap->itn.last = 0; __dmap_add_to_free_pool(fcd, dmap); } @@ -256,6 +323,8 @@ static void inode_reclaim_dmap_range(struct fuse_conn_dax *fcd, if (!node) break; dmap = node_to_dmap(node); + /* inode is going away. There should not be any users of dmap */ + WARN_ON(refcount_read(&dmap->refcnt) > 1); interval_tree_remove(&dmap->itn, &fi->dax->tree); num++; list_add(&dmap->list, &to_remove); @@ -280,6 +349,21 @@ static void inode_reclaim_dmap_range(struct fuse_conn_dax *fcd, spin_unlock(&fcd->lock); } +static int dmap_removemapping_one(struct inode *inode, + struct fuse_dax_mapping *dmap) +{ + struct fuse_removemapping_one forget_one; + struct fuse_removemapping_in inarg; + + memset(&inarg, 0, sizeof(inarg)); + inarg.count = 1; + memset(&forget_one, 0, sizeof(forget_one)); + forget_one.moffset = dmap->window_offset; + forget_one.len = dmap->length; + + return fuse_send_removemapping(inode, &inarg, &forget_one); +} + /* * It is called from evict_inode() and by that time inode is going away. So * this function does not take any locks like fi->dax->sem for traversing @@ -327,6 +411,16 @@ static void fuse_fill_iomap(struct inode *inode, loff_t pos, loff_t length, if (flags & IOMAP_FAULT) iomap->length = ALIGN(len, PAGE_SIZE); iomap->type = IOMAP_MAPPED; + /* + * increace refcnt so that reclaim code knows this dmap is in + * use. This assumes fi->dax->sem mutex is held either + * shared/exclusive. + */ + refcount_inc(&dmap->refcnt); + + /* iomap->private should be NULL */ + WARN_ON_ONCE(iomap->private); + iomap->private = dmap; } else { /* Mapping beyond end of file is hole */ fuse_fill_iomap_hole(iomap, length); @@ -346,8 +440,28 @@ static int fuse_setup_new_dax_mapping(struct inode *inode, loff_t pos, unsigned long start_idx = pos >> FUSE_DAX_SHIFT; struct interval_tree_node *node; - alloc_dmap = alloc_dax_mapping(fcd); - if (!alloc_dmap) + /* + * Can't do inline reclaim in fault path. We call + * dax_layout_busy_page() before we free a range. And + * fuse_wait_dax_page() drops fi->i_mmap_sem lock and requires it. + * In fault path we enter with fi->i_mmap_sem held and can't drop + * it. Also in fault path we hold fi->i_mmap_sem shared and not + * exclusive, so that creates further issues with fuse_wait_dax_page(). + * Hence return -EAGAIN and fuse_dax_fault() will wait for a memory + * range to become free and retry. + */ + if (flags & IOMAP_FAULT) { + alloc_dmap = alloc_dax_mapping(fcd); + if (!alloc_dmap) + return -EAGAIN; + } else { + alloc_dmap = alloc_dax_mapping_reclaim(fcd, inode); + if (IS_ERR(alloc_dmap)) + return PTR_ERR(alloc_dmap); + } + + /* If we are here, we should have memory allocated */ + if (WARN_ON(!alloc_dmap)) return -EIO; /* @@ -399,9 +513,10 @@ static int fuse_upgrade_dax_mapping(struct inode *inode, loff_t pos, node = interval_tree_iter_first(&fi->dax->tree, idx, idx); /* We are holding either inode lock or i_mmap_sem, and that should - * ensure that dmap can't reclaimed or truncated and it should still - * be there in tree despite the fact we dropped and re-acquired the - * lock. + * ensure that dmap can't be truncated. We are holding a reference + * on dmap and that should make sure it can't be reclaimed. So dmap + * should still be there in tree despite the fact we dropped and + * re-acquired the fi->dax->sem lock. */ ret = -EIO; if (WARN_ON(!node)) @@ -409,6 +524,17 @@ static int fuse_upgrade_dax_mapping(struct inode *inode, loff_t pos, dmap = node_to_dmap(node); + /* We took an extra reference on dmap to make sure its not reclaimd. + * Now we hold fi->dax->sem lock and that reference is not needed + * anymore. Drop it. + */ + if (refcount_dec_and_test(&dmap->refcnt)) { + /* refcount should not hit 0. This object only goes + * away when fuse connection goes away + */ + WARN_ON_ONCE(1); + } + /* Maybe another thread already upgraded mapping while we were not * holding lock. */ @@ -468,7 +594,11 @@ static int fuse_iomap_begin(struct inode *inode, loff_t pos, loff_t length, * two threads to be trying to this simultaneously * for same dmap. So drop shared lock and acquire * exclusive lock. + * + * Before dropping fi->dax->sem lock, take reference + * on dmap so that its not freed by range reclaim. */ + refcount_inc(&dmap->refcnt); up_read(&fi->dax->sem); pr_debug("%s: Upgrading mapping at offset 0x%llx length 0x%llx\n", __func__, pos, length); @@ -505,6 +635,17 @@ static int fuse_iomap_end(struct inode *inode, loff_t pos, loff_t length, ssize_t written, unsigned int flags, struct iomap *iomap) { + struct fuse_dax_mapping *dmap = iomap->private; + + if (dmap) { + if (refcount_dec_and_test(&dmap->refcnt)) { + /* refcount should not hit 0. This object only goes + * away when fuse connection goes away + */ + WARN_ON_ONCE(1); + } + } + /* DAX writes beyond end-of-file aren't handled using iomap, so the * file size is unchanged and there is nothing to do here. */ @@ -654,9 +795,16 @@ static vm_fault_t __fuse_dax_fault(struct vm_fault *vmf, struct inode *inode = file_inode(vmf->vma->vm_file); struct super_block *sb = inode->i_sb; pfn_t pfn; + int error = 0; + struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_conn_dax *fcd = fc->dax; + bool retry = false; if (write) sb_start_pagefault(sb); +retry: + if (retry && !(fcd->nr_free_ranges > 0)) + wait_event(fcd->range_waitq, (fcd->nr_free_ranges > 0)); /* * We need to serialize against not only truncate but also against @@ -665,7 +813,13 @@ static vm_fault_t __fuse_dax_fault(struct vm_fault *vmf, * to populate page cache or access memory we are trying to free. */ down_read(&get_fuse_inode(inode)->i_mmap_sem); - ret = dax_iomap_fault(vmf, pe_size, &pfn, NULL, &fuse_iomap_ops); + ret = dax_iomap_fault(vmf, pe_size, &pfn, &error, &fuse_iomap_ops); + if ((ret & VM_FAULT_ERROR) && error == -EAGAIN) { + error = 0; + retry = true; + up_read(&get_fuse_inode(inode)->i_mmap_sem); + goto retry; + } if (ret & VM_FAULT_NEEDDSYNC) ret = dax_finish_sync_fault(vmf, pe_size, pfn); @@ -714,6 +868,348 @@ int fuse_dax_mmap(struct file *file, struct vm_area_struct *vma) return 0; } +static int dmap_writeback_invalidate(struct inode *inode, + struct fuse_dax_mapping *dmap) +{ + int ret; + loff_t start_pos = dmap->itn.start << FUSE_DAX_SHIFT; + loff_t end_pos = (start_pos + FUSE_DAX_SZ - 1); + + ret = filemap_fdatawrite_range(inode->i_mapping, start_pos, end_pos); + if (ret) { + pr_debug("fuse: filemap_fdatawrite_range() failed. err=%d start_pos=0x%llx, end_pos=0x%llx\n", + ret, start_pos, end_pos); + return ret; + } + + ret = invalidate_inode_pages2_range(inode->i_mapping, + start_pos >> PAGE_SHIFT, + end_pos >> PAGE_SHIFT); + if (ret) + pr_debug("fuse: invalidate_inode_pages2_range() failed err=%d\n", + ret); + + return ret; +} + +static int reclaim_one_dmap_locked(struct inode *inode, + struct fuse_dax_mapping *dmap) +{ + int ret; + struct fuse_inode *fi = get_fuse_inode(inode); + + /* + * igrab() was done to make sure inode won't go under us, and this + * further avoids the race with evict(). + */ + ret = dmap_writeback_invalidate(inode, dmap); + if (ret) + return ret; + + /* Remove dax mapping from inode interval tree now */ + interval_tree_remove(&dmap->itn, &fi->dax->tree); + fi->dax->nr--; + + /* It is possible that umount/shutdown has killed the fuse connection + * and worker thread is trying to reclaim memory in parallel. Don't + * warn in that case. + */ + ret = dmap_removemapping_one(inode, dmap); + if (ret && ret != -ENOTCONN) { + pr_warn("Failed to remove mapping. offset=0x%llx len=0x%llx ret=%d\n", + dmap->window_offset, dmap->length, ret); + } + return 0; +} + +/* Find first mapped dmap for an inode and return file offset. Caller needs + * to hold fi->dax->sem lock either shared or exclusive. + */ +static struct fuse_dax_mapping *inode_lookup_first_dmap(struct inode *inode) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_dax_mapping *dmap; + struct interval_tree_node *node; + + for (node = interval_tree_iter_first(&fi->dax->tree, 0, -1); node; + node = interval_tree_iter_next(node, 0, -1)) { + dmap = node_to_dmap(node); + /* still in use. */ + if (refcount_read(&dmap->refcnt) > 1) + continue; + + return dmap; + } + + return NULL; +} + +/* + * Find first mapping in the tree and free it and return it. Do not add + * it back to free pool. + */ +static struct fuse_dax_mapping * +inode_inline_reclaim_one_dmap(struct fuse_conn_dax *fcd, struct inode *inode, + bool *retry) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_dax_mapping *dmap; + u64 dmap_start, dmap_end; + unsigned long start_idx; + int ret; + struct interval_tree_node *node; + + down_write(&fi->i_mmap_sem); + + /* Lookup a dmap and corresponding file offset to reclaim. */ + down_read(&fi->dax->sem); + dmap = inode_lookup_first_dmap(inode); + if (dmap) { + start_idx = dmap->itn.start; + dmap_start = start_idx << FUSE_DAX_SHIFT; + dmap_end = dmap_start + FUSE_DAX_SZ - 1; + } + up_read(&fi->dax->sem); + + if (!dmap) + goto out_mmap_sem; + /* + * Make sure there are no references to inode pages using + * get_user_pages() + */ + ret = fuse_dax_break_layouts(inode, dmap_start, dmap_end); + if (ret) { + pr_debug("fuse: fuse_dax_break_layouts() failed. err=%d\n", + ret); + dmap = ERR_PTR(ret); + goto out_mmap_sem; + } + + down_write(&fi->dax->sem); + node = interval_tree_iter_first(&fi->dax->tree, start_idx, start_idx); + /* Range already got reclaimed by somebody else */ + if (!node) { + if (retry) + *retry = true; + goto out_write_dmap_sem; + } + + dmap = node_to_dmap(node); + /* still in use. */ + if (refcount_read(&dmap->refcnt) > 1) { + dmap = NULL; + if (retry) + *retry = true; + goto out_write_dmap_sem; + } + + ret = reclaim_one_dmap_locked(inode, dmap); + if (ret < 0) { + dmap = ERR_PTR(ret); + goto out_write_dmap_sem; + } + + /* Clean up dmap. Do not add back to free list */ + dmap_remove_busy_list(fcd, dmap); + dmap->inode = NULL; + dmap->itn.start = dmap->itn.last = 0; + + pr_debug("fuse: %s: inline reclaimed memory range. inode=%p, window_offset=0x%llx, length=0x%llx\n", + __func__, inode, dmap->window_offset, dmap->length); + +out_write_dmap_sem: + up_write(&fi->dax->sem); +out_mmap_sem: + up_write(&fi->i_mmap_sem); + return dmap; +} + +static struct fuse_dax_mapping * +alloc_dax_mapping_reclaim(struct fuse_conn_dax *fcd, struct inode *inode) +{ + struct fuse_dax_mapping *dmap; + struct fuse_inode *fi = get_fuse_inode(inode); + + while (1) { + bool retry = false; + + dmap = alloc_dax_mapping(fcd); + if (dmap) + return dmap; + + dmap = inode_inline_reclaim_one_dmap(fcd, inode, &retry); + /* + * Either we got a mapping or it is an error, return in both + * the cases. + */ + if (dmap) + return dmap; + + /* If we could not reclaim a mapping because it + * had a reference or some other temporary failure, + * Try again. We want to give up inline reclaim only + * if there is no range assigned to this node. Otherwise + * if a deadlock is possible if we sleep with fi->i_mmap_sem + * held and worker to free memory can't make progress due + * to unavailability of fi->i_mmap_sem lock. So sleep + * only if fi->dax->nr=0 + */ + if (retry) + continue; + /* + * There are no mappings which can be reclaimed. Wait for one. + * We are not holding fi->dax->sem. So it is possible + * that range gets added now. But as we are not holding + * fi->i_mmap_sem, worker should still be able to free up + * a range and wake us up. + */ + if (!fi->dax->nr && !(fcd->nr_free_ranges > 0)) { + if (wait_event_killable_exclusive(fcd->range_waitq, + (fcd->nr_free_ranges > 0))) { + return ERR_PTR(-EINTR); + } + } + } +} + +static int lookup_and_reclaim_dmap_locked(struct fuse_conn_dax *fcd, + struct inode *inode, + unsigned long start_idx) +{ + int ret; + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_dax_mapping *dmap; + struct interval_tree_node *node; + + /* Find fuse dax mapping at file offset inode. */ + node = interval_tree_iter_first(&fi->dax->tree, start_idx, start_idx); + + /* Range already got cleaned up by somebody else */ + if (!node) + return 0; + dmap = node_to_dmap(node); + + /* still in use. */ + if (refcount_read(&dmap->refcnt) > 1) + return 0; + + ret = reclaim_one_dmap_locked(inode, dmap); + if (ret < 0) + return ret; + + /* Cleanup dmap entry and add back to free list */ + spin_lock(&fcd->lock); + dmap_reinit_add_to_free_pool(fcd, dmap); + spin_unlock(&fcd->lock); + return ret; +} + +/* + * Free a range of memory. + * Locking: + * 1. Take fi->i_mmap_sem to block dax faults. + * 2. Take fi->dax->sem to protect interval tree and also to make sure + * read/write can not reuse a dmap which we might be freeing. + */ +static int lookup_and_reclaim_dmap(struct fuse_conn_dax *fcd, + struct inode *inode, + unsigned long start_idx, + unsigned long end_idx) +{ + int ret; + struct fuse_inode *fi = get_fuse_inode(inode); + loff_t dmap_start = start_idx << FUSE_DAX_SHIFT; + loff_t dmap_end = (dmap_start + FUSE_DAX_SZ) - 1; + + down_write(&fi->i_mmap_sem); + ret = fuse_dax_break_layouts(inode, dmap_start, dmap_end); + if (ret) { + pr_debug("virtio_fs: fuse_dax_break_layouts() failed. err=%d\n", + ret); + goto out_mmap_sem; + } + + down_write(&fi->dax->sem); + ret = lookup_and_reclaim_dmap_locked(fcd, inode, start_idx); + up_write(&fi->dax->sem); +out_mmap_sem: + up_write(&fi->i_mmap_sem); + return ret; +} + +static int try_to_free_dmap_chunks(struct fuse_conn_dax *fcd, + unsigned long nr_to_free) +{ + struct fuse_dax_mapping *dmap, *pos, *temp; + int ret, nr_freed = 0; + unsigned long start_idx = 0, end_idx = 0; + struct inode *inode = NULL; + + /* Pick first busy range and free it for now*/ + while (1) { + if (nr_freed >= nr_to_free) + break; + + dmap = NULL; + spin_lock(&fcd->lock); + + if (!fcd->nr_busy_ranges) { + spin_unlock(&fcd->lock); + return 0; + } + + list_for_each_entry_safe(pos, temp, &fcd->busy_ranges, + busy_list) { + /* skip this range if it's in use. */ + if (refcount_read(&pos->refcnt) > 1) + continue; + + inode = igrab(pos->inode); + /* + * This inode is going away. That will free + * up all the ranges anyway, continue to + * next range. + */ + if (!inode) + continue; + /* + * Take this element off list and add it tail. If + * this element can't be freed, it will help with + * selecting new element in next iteration of loop. + */ + dmap = pos; + list_move_tail(&dmap->busy_list, &fcd->busy_ranges); + start_idx = end_idx = dmap->itn.start; + break; + } + spin_unlock(&fcd->lock); + if (!dmap) + return 0; + + ret = lookup_and_reclaim_dmap(fcd, inode, start_idx, end_idx); + iput(inode); + if (ret) + return ret; + nr_freed++; + } + return 0; +} + +static void fuse_dax_free_mem_worker(struct work_struct *work) +{ + int ret; + struct fuse_conn_dax *fcd = container_of(work, struct fuse_conn_dax, + free_work.work); + ret = try_to_free_dmap_chunks(fcd, FUSE_DAX_RECLAIM_CHUNK); + if (ret) { + pr_debug("fuse: try_to_free_dmap_chunks() failed with err=%d\n", + ret); + } + + /* If number of free ranges are still below threhold, requeue */ + kick_dmap_free_worker(fcd, 1); +} + static void fuse_free_dax_mem_ranges(struct list_head *mem_list) { struct fuse_dax_mapping *range, *temp; @@ -745,8 +1241,11 @@ static int fuse_dax_mem_range_init(struct fuse_conn_dax *fcd) size_t dax_size = -1; unsigned long i; + init_waitqueue_head(&fcd->range_waitq); INIT_LIST_HEAD(&fcd->free_ranges); INIT_LIST_HEAD(&fcd->busy_ranges); + INIT_DELAYED_WORK(&fcd->free_work, fuse_dax_free_mem_worker); + id = dax_read_lock(); nr_pages = dax_direct_access(fcd->dev, 0, PHYS_PFN(dax_size), &kaddr, &pfn); @@ -773,10 +1272,12 @@ static int fuse_dax_mem_range_init(struct fuse_conn_dax *fcd) range->window_offset = i * FUSE_DAX_SZ; range->length = FUSE_DAX_SZ; INIT_LIST_HEAD(&range->busy_list); + refcount_set(&range->refcnt, 1); list_add_tail(&range->list, &fcd->free_ranges); } fcd->nr_free_ranges = nr_ranges; + fcd->nr_ranges = nr_ranges; return 0; out_err: /* Free All allocated elements */ @@ -852,3 +1353,13 @@ bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment) } return true; } + +void fuse_dax_cancel_work(struct fuse_conn *fc) +{ + struct fuse_conn_dax *fcd = fc->dax; + + if (fcd) + cancel_delayed_work_sync(&fcd->free_work); + +} +EXPORT_SYMBOL_GPL(fuse_dax_cancel_work); diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 94a9b0b9150b..d92d289cd76d 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1170,6 +1170,7 @@ bool fuse_dax_inode_alloc(struct super_block *sb, struct fuse_inode *fi); void fuse_dax_inode_init(struct inode *inode); void fuse_dax_inode_cleanup(struct inode *inode); bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment); +void fuse_dax_cancel_work(struct fuse_conn *fc); /* passthrough.c */ int fuse_passthrough_open(struct fuse_dev *fud, u32 lower_fd); diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index f5830d07342f..ba228531e821 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1356,6 +1356,12 @@ static void virtio_kill_sb(struct super_block *sb) vfs = fc->iq.priv; fsvq = &vfs->vqs[VQ_HIPRIO]; + /* Stop dax worker. Soon evict_inodes() will be called which will + * free all memory ranges belonging to all inodes. + */ + if (IS_ENABLED(CONFIG_FUSE_DAX)) + fuse_dax_cancel_work(fc); + /* Stop forget queue. Soon destroy will be sent */ spin_lock(&fsvq->lock); fsvq->connected = false; From d67035009c3433cfca0287c97c021bb8a04823d8 Mon Sep 17 00:00:00 2001 From: Max Reitz Date: Tue, 8 Sep 2020 18:01:48 +0200 Subject: [PATCH 038/148] UPSTREAM: fuse: add submount support to - Add fuse_attr.flags - Add FUSE_ATTR_SUBMOUNT This is a flag for fuse_attr.flags that indicates that the given entry resides on a different filesystem than the parent, and as such should have a different st_dev. - Add FUSE_SUBMOUNTS The client sets this flag if it supports automounting directories. Change-Id: Ibea2d75d0a93dc3c269555fdb7d0ee58af7860b4 Signed-off-by: Max Reitz Signed-off-by: Miklos Szeredi --- include/uapi/linux/fuse.h | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index 37467e470242..bf3c62dd7112 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -172,6 +172,9 @@ * - add FUSE_WRITE_KILL_PRIV flag * - add FUSE_SETUPMAPPING and FUSE_REMOVEMAPPING * - add map_alignment to fuse_init_out, add FUSE_MAP_ALIGNMENT flag + * + * 7.32 + * - add flags to fuse_attr, add FUSE_ATTR_SUBMOUNT, add FUSE_SUBMOUNTS */ #ifndef _LINUX_FUSE_H @@ -207,7 +210,7 @@ #define FUSE_KERNEL_VERSION 7 /** Minor version number of this interface */ -#define FUSE_KERNEL_MINOR_VERSION 31 +#define FUSE_KERNEL_MINOR_VERSION 32 /** The node ID of the root inode */ #define FUSE_ROOT_ID 1 @@ -231,7 +234,7 @@ struct fuse_attr { uint32_t gid; uint32_t rdev; uint32_t blksize; - uint32_t padding; + uint32_t flags; }; struct fuse_kstatfs { @@ -316,6 +319,7 @@ struct fuse_file_lock { * FUSE_MAP_ALIGNMENT: init_out.map_alignment contains log2(byte alignment) for * foffset and moffset fields in struct * fuse_setupmapping_out and fuse_removemapping_one. + * FUSE_SUBMOUNTS: kernel supports auto-mounting directory submounts */ #define FUSE_ASYNC_READ (1 << 0) #define FUSE_POSIX_LOCKS (1 << 1) @@ -344,6 +348,7 @@ struct fuse_file_lock { #define FUSE_NO_OPENDIR_SUPPORT (1 << 24) #define FUSE_EXPLICIT_INVAL_DATA (1 << 25) #define FUSE_MAP_ALIGNMENT (1 << 26) +#define FUSE_SUBMOUNTS (1 << 27) #define FUSE_PASSTHROUGH (1 << 31) /** @@ -420,6 +425,13 @@ struct fuse_file_lock { */ #define FUSE_FSYNC_FDATASYNC (1 << 0) +/** + * fuse_attr flags + * + * FUSE_ATTR_SUBMOUNT: Object is a submount root + */ +#define FUSE_ATTR_SUBMOUNT (1 << 0) + enum fuse_opcode { FUSE_LOOKUP = 1, FUSE_FORGET = 2, /* no reply */ From 9784782e98d48f43c7e4224977263ad42e921575 Mon Sep 17 00:00:00 2001 From: Max Reitz Date: Mon, 20 Apr 2020 17:54:38 +0200 Subject: [PATCH 039/148] UPSTREAM: fuse: store fuse_conn in fuse_req Every fuse_req belongs to a fuse_conn. Right now, we always know which fuse_conn that is based on the respective device, but we want to allow multiple (sub)mounts per single connection, and then the corresponding filesystem is not going to be so trivial to obtain. Storing a pointer to the associated fuse_conn in every fuse_req will allow us to trivially find any request's superblock (and thus filesystem) even then. Change-Id: I60a5e891535ab5c172d3658c3b6f42f4804e2a21 Signed-off-by: Max Reitz Reviewed-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi --- fs/fuse/dev.c | 13 +++++++------ fs/fuse/fuse_i.h | 3 +++ 2 files changed, 10 insertions(+), 6 deletions(-) diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index ad7a72cf6bfd..a5a5e8aa81e3 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -42,20 +42,21 @@ static struct fuse_dev *fuse_get_dev(struct file *file) return READ_ONCE(file->private_data); } -static void fuse_request_init(struct fuse_req *req) +static void fuse_request_init(struct fuse_conn *fc, struct fuse_req *req) { INIT_LIST_HEAD(&req->list); INIT_LIST_HEAD(&req->intr_entry); init_waitqueue_head(&req->waitq); refcount_set(&req->count, 1); __set_bit(FR_PENDING, &req->flags); + req->fc = fc; } -static struct fuse_req *fuse_request_alloc(gfp_t flags) +static struct fuse_req *fuse_request_alloc(struct fuse_conn *fc, gfp_t flags) { struct fuse_req *req = kmem_cache_zalloc(fuse_req_cachep, flags); if (req) - fuse_request_init(req); + fuse_request_init(fc, req); return req; } @@ -127,7 +128,7 @@ static struct fuse_req *fuse_get_req(struct fuse_conn *fc, bool for_background) if (fc->conn_error) goto out; - req = fuse_request_alloc(GFP_KERNEL); + req = fuse_request_alloc(fc, GFP_KERNEL); err = -ENOMEM; if (!req) { if (for_background) @@ -487,7 +488,7 @@ ssize_t fuse_simple_request(struct fuse_conn *fc, struct fuse_args *args) if (args->force) { atomic_inc(&fc->num_waiting); - req = fuse_request_alloc(GFP_KERNEL | __GFP_NOFAIL); + req = fuse_request_alloc(fc, GFP_KERNEL | __GFP_NOFAIL); if (!args->nocreds) fuse_force_creds(fc, req); @@ -554,7 +555,7 @@ int fuse_simple_background(struct fuse_conn *fc, struct fuse_args *args, if (args->force) { WARN_ON(!args->nocreds); - req = fuse_request_alloc(gfp_flags); + req = fuse_request_alloc(fc, gfp_flags); if (!req) return -ENOMEM; __set_bit(FR_BACKGROUND, &req->flags); diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index d92d289cd76d..89ce51ecf87d 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -394,6 +394,9 @@ struct fuse_req { /** virtio-fs's physically contiguous buffer for in and out args */ void *argbuf; #endif + + /** fuse_conn this request belongs to */ + struct fuse_conn *fc; }; struct fuse_iqueue; From 5d4da270054425ba64460d01a1749b85c20233e3 Mon Sep 17 00:00:00 2001 From: Alistair Delva Date: Fri, 24 Jul 2020 13:51:48 -0700 Subject: [PATCH 040/148] ANDROID: GKI: Don't compact fuse_req when CONFIG_VIRTIO_FS=n Otherwise we cannot enable VIRTIO_FS downstream. Bug: 161843089 Change-Id: I317b8c425ab96a1bd484b85b41ce3cb036327117 Signed-off-by: Alistair Delva --- fs/fuse/fuse_i.h | 2 -- 1 file changed, 2 deletions(-) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 89ce51ecf87d..688226469759 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -390,10 +390,8 @@ struct fuse_req { /** Used to wake up the task waiting for completion of request*/ wait_queue_head_t waitq; -#if IS_ENABLED(CONFIG_VIRTIO_FS) /** virtio-fs's physically contiguous buffer for in and out args */ void *argbuf; -#endif /** fuse_conn this request belongs to */ struct fuse_conn *fc; From 2eff56e2dd09337795c5c642870f6b1a4bb2bb79 Mon Sep 17 00:00:00 2001 From: Max Reitz Date: Mon, 20 Apr 2020 17:59:34 +0200 Subject: [PATCH 041/148] UPSTREAM: fuse: drop fuse_conn parameter where possible With the last commit, all functions that handle some existing fuse_req no longer need to be given the associated fuse_conn, because they can get it from the fuse_req object. Change-Id: I0436ff160047ca7bc1de46681cd4f9e92878561f Signed-off-by: Max Reitz Reviewed-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi --- fs/fuse/dev.c | 70 +++++++++++++++++++++++++-------------------- fs/fuse/fuse_i.h | 2 +- fs/fuse/virtio_fs.c | 8 ++---- 3 files changed, 43 insertions(+), 37 deletions(-) diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index a5a5e8aa81e3..b68346db7419 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -103,7 +103,7 @@ static void fuse_drop_waiting(struct fuse_conn *fc) } } -static void fuse_put_request(struct fuse_conn *fc, struct fuse_req *req); +static void fuse_put_request(struct fuse_req *req); static struct fuse_req *fuse_get_req(struct fuse_conn *fc, bool for_background) { @@ -146,7 +146,7 @@ static struct fuse_req *fuse_get_req(struct fuse_conn *fc, bool for_background) if (unlikely(req->in.h.uid == ((uid_t)-1) || req->in.h.gid == ((gid_t)-1))) { - fuse_put_request(fc, req); + fuse_put_request(req); return ERR_PTR(-EOVERFLOW); } return req; @@ -156,8 +156,10 @@ static struct fuse_req *fuse_get_req(struct fuse_conn *fc, bool for_background) return ERR_PTR(err); } -static void fuse_put_request(struct fuse_conn *fc, struct fuse_req *req) +static void fuse_put_request(struct fuse_req *req) { + struct fuse_conn *fc = req->fc; + if (refcount_dec_and_test(&req->count)) { if (test_bit(FR_BACKGROUND, &req->flags)) { /* @@ -279,8 +281,9 @@ static void flush_bg_queue(struct fuse_conn *fc) * the 'end' callback is called if given, else the reference to the * request is released */ -void fuse_request_end(struct fuse_conn *fc, struct fuse_req *req) +void fuse_request_end(struct fuse_req *req) { + struct fuse_conn *fc = req->fc; struct fuse_iqueue *fiq = &fc->iq; if (test_and_set_bit(FR_FINISHED, &req->flags)) @@ -331,12 +334,14 @@ void fuse_request_end(struct fuse_conn *fc, struct fuse_req *req) if (test_bit(FR_ASYNC, &req->flags)) req->args->end(fc, req->args, req->out.h.error); put_request: - fuse_put_request(fc, req); + fuse_put_request(req); } EXPORT_SYMBOL_GPL(fuse_request_end); -static int queue_interrupt(struct fuse_iqueue *fiq, struct fuse_req *req) +static int queue_interrupt(struct fuse_req *req) { + struct fuse_iqueue *fiq = &req->fc->iq; + spin_lock(&fiq->lock); /* Check for we've sent request to interrupt this req */ if (unlikely(!test_bit(FR_INTERRUPTED, &req->flags))) { @@ -363,8 +368,9 @@ static int queue_interrupt(struct fuse_iqueue *fiq, struct fuse_req *req) return 0; } -static void request_wait_answer(struct fuse_conn *fc, struct fuse_req *req) +static void request_wait_answer(struct fuse_req *req) { + struct fuse_conn *fc = req->fc; struct fuse_iqueue *fiq = &fc->iq; int err; @@ -379,7 +385,7 @@ static void request_wait_answer(struct fuse_conn *fc, struct fuse_req *req) /* matches barrier in fuse_dev_do_read() */ smp_mb__after_atomic(); if (test_bit(FR_SENT, &req->flags)) - queue_interrupt(fiq, req); + queue_interrupt(req); } if (!test_bit(FR_FORCE, &req->flags)) { @@ -410,9 +416,9 @@ static void request_wait_answer(struct fuse_conn *fc, struct fuse_req *req) test_bit(FR_FINISHED, &req->flags)); } -static void __fuse_request_send(struct fuse_conn *fc, struct fuse_req *req) +static void __fuse_request_send(struct fuse_req *req) { - struct fuse_iqueue *fiq = &fc->iq; + struct fuse_iqueue *fiq = &req->fc->iq; BUG_ON(test_bit(FR_BACKGROUND, &req->flags)); spin_lock(&fiq->lock); @@ -426,7 +432,7 @@ static void __fuse_request_send(struct fuse_conn *fc, struct fuse_req *req) __fuse_get_request(req); queue_request_and_unlock(fiq, req, 1); - request_wait_answer(fc, req); + request_wait_answer(req); /* Pairs with smp_wmb() in fuse_request_end() */ smp_rmb(); } @@ -465,8 +471,10 @@ static void fuse_adjust_compat(struct fuse_conn *fc, struct fuse_args *args) } } -static void fuse_force_creds(struct fuse_conn *fc, struct fuse_req *req) +static void fuse_force_creds(struct fuse_req *req) { + struct fuse_conn *fc = req->fc; + req->in.h.uid = from_kuid_munged(fc->user_ns, current_fsuid()); req->in.h.gid = from_kgid_munged(fc->user_ns, current_fsgid()); req->in.h.pid = pid_nr_ns(task_pid(current), fc->pid_ns); @@ -491,7 +499,7 @@ ssize_t fuse_simple_request(struct fuse_conn *fc, struct fuse_args *args) req = fuse_request_alloc(fc, GFP_KERNEL | __GFP_NOFAIL); if (!args->nocreds) - fuse_force_creds(fc, req); + fuse_force_creds(req); __set_bit(FR_WAITING, &req->flags); __set_bit(FR_FORCE, &req->flags); @@ -508,20 +516,20 @@ ssize_t fuse_simple_request(struct fuse_conn *fc, struct fuse_args *args) if (!args->noreply) __set_bit(FR_ISREPLY, &req->flags); - __fuse_request_send(fc, req); + __fuse_request_send(req); ret = req->out.h.error; if (!ret && args->out_argvar) { BUG_ON(args->out_numargs == 0); ret = args->out_args[args->out_numargs - 1].size; } - fuse_put_request(fc, req); + fuse_put_request(req); return ret; } -static bool fuse_request_queue_background(struct fuse_conn *fc, - struct fuse_req *req) +static bool fuse_request_queue_background(struct fuse_req *req) { + struct fuse_conn *fc = req->fc; bool queued = false; WARN_ON(!test_bit(FR_BACKGROUND, &req->flags)); @@ -568,8 +576,8 @@ int fuse_simple_background(struct fuse_conn *fc, struct fuse_args *args, fuse_args_to_req(req, args); - if (!fuse_request_queue_background(fc, req)) { - fuse_put_request(fc, req); + if (!fuse_request_queue_background(req)) { + fuse_put_request(req); return -ENOTCONN; } @@ -599,7 +607,7 @@ static int fuse_simple_notify_reply(struct fuse_conn *fc, } else { err = -ENODEV; spin_unlock(&fiq->lock); - fuse_put_request(fc, req); + fuse_put_request(req); } return err; @@ -1283,7 +1291,7 @@ static ssize_t fuse_dev_do_read(struct fuse_dev *fud, struct file *file, /* SETXATTR is special, since it may contain too large data */ if (args->opcode == FUSE_SETXATTR) req->out.h.error = -E2BIG; - fuse_request_end(fc, req); + fuse_request_end(req); goto restart; } spin_lock(&fpq->lock); @@ -1326,8 +1334,8 @@ static ssize_t fuse_dev_do_read(struct fuse_dev *fud, struct file *file, /* matches barrier in request_wait_answer() */ smp_mb__after_atomic(); if (test_bit(FR_INTERRUPTED, &req->flags)) - queue_interrupt(fiq, req); - fuse_put_request(fc, req); + queue_interrupt(req); + fuse_put_request(req); return reqsize; @@ -1335,7 +1343,7 @@ out_end: if (!test_bit(FR_PRIVATE, &req->flags)) list_del_init(&req->list); spin_unlock(&fpq->lock); - fuse_request_end(fc, req); + fuse_request_end(req); return err; err_unlock: @@ -1919,9 +1927,9 @@ static ssize_t fuse_dev_do_write(struct fuse_dev *fud, else if (oh.error == -ENOSYS) fc->no_interrupt = 1; else if (oh.error == -EAGAIN) - err = queue_interrupt(&fc->iq, req); + err = queue_interrupt(req); - fuse_put_request(fc, req); + fuse_put_request(req); goto copy_finish; } @@ -1957,7 +1965,7 @@ static ssize_t fuse_dev_do_write(struct fuse_dev *fud, list_del_init(&req->list); spin_unlock(&fpq->lock); - fuse_request_end(fc, req); + fuse_request_end(req); out: return err ? err : nbytes; @@ -2093,7 +2101,7 @@ static __poll_t fuse_dev_poll(struct file *file, poll_table *wait) } /* Abort all requests on the given list (pending or processing) */ -static void end_requests(struct fuse_conn *fc, struct list_head *head) +static void end_requests(struct list_head *head) { while (!list_empty(head)) { struct fuse_req *req; @@ -2101,7 +2109,7 @@ static void end_requests(struct fuse_conn *fc, struct list_head *head) req->out.h.error = -ECONNABORTED; clear_bit(FR_SENT, &req->flags); list_del_init(&req->list); - fuse_request_end(fc, req); + fuse_request_end(req); } } @@ -2196,7 +2204,7 @@ void fuse_abort_conn(struct fuse_conn *fc) wake_up_all(&fc->blocked_waitq); spin_unlock(&fc->lock); - end_requests(fc, &to_end); + end_requests(&to_end); } else { spin_unlock(&fc->lock); } @@ -2226,7 +2234,7 @@ int fuse_dev_release(struct inode *inode, struct file *file) list_splice_init(&fpq->processing[i], &to_end); spin_unlock(&fpq->lock); - end_requests(fc, &to_end); + end_requests(&to_end); /* Are we the last open device? */ if (atomic_dec_and_test(&fc->dev_count)) { diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 688226469759..31f0052c4b99 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -987,7 +987,7 @@ int fuse_simple_background(struct fuse_conn *fc, struct fuse_args *args, /** * End a finished request */ -void fuse_request_end(struct fuse_conn *fc, struct fuse_req *req); +void fuse_request_end(struct fuse_req *req); /* Abort all requests */ void fuse_abort_conn(struct fuse_conn *fc); diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index ba228531e821..147467a0aab5 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -354,7 +354,6 @@ static void virtio_fs_request_dispatch_work(struct work_struct *work) struct fuse_req *req; struct virtio_fs_vq *fsvq = container_of(work, struct virtio_fs_vq, dispatch_work.work); - struct fuse_conn *fc = fsvq->fud->fc; int ret; pr_debug("virtio-fs: worker %s called.\n", __func__); @@ -369,7 +368,7 @@ static void virtio_fs_request_dispatch_work(struct work_struct *work) list_del_init(&req->list); spin_unlock(&fsvq->lock); - fuse_request_end(fc, req); + fuse_request_end(req); } /* Dispatch pending requests */ @@ -400,7 +399,7 @@ static void virtio_fs_request_dispatch_work(struct work_struct *work) spin_unlock(&fsvq->lock); pr_err("virtio-fs: virtio_fs_enqueue_req() failed %d\n", ret); - fuse_request_end(fc, req); + fuse_request_end(req); } } } @@ -560,7 +559,6 @@ static void virtio_fs_request_complete(struct fuse_req *req, struct virtio_fs_vq *fsvq) { struct fuse_pqueue *fpq = &fsvq->fud->pq; - struct fuse_conn *fc = fsvq->fud->fc; struct fuse_args *args; struct fuse_args_pages *ap; unsigned int len, i, thislen; @@ -593,7 +591,7 @@ static void virtio_fs_request_complete(struct fuse_req *req, clear_bit(FR_SENT, &req->flags); spin_unlock(&fpq->lock); - fuse_request_end(fc, req); + fuse_request_end(req); spin_lock(&fsvq->lock); dec_in_flight_req(fsvq); spin_unlock(&fsvq->lock); From 5d830cf84db33c7aae22b51e3069aed6ea1f8e44 Mon Sep 17 00:00:00 2001 From: Max Reitz Date: Wed, 6 May 2020 17:44:12 +0200 Subject: [PATCH 042/148] BACKPORT: fuse: split fuse_mount off of fuse_conn We want to allow submounts for the same fuse_conn, but with different superblocks so that each of the submounts has its own device ID. To do so, we need to split all mount-specific information off of fuse_conn into a new fuse_mount structure, so that multiple mounts can share a single fuse_conn. We need to take care only to perform connection-level actions once (i.e. when the fuse_conn and thus the first fuse_mount are established, or when the last fuse_mount and thus the fuse_conn are destroyed). For example, fuse_sb_destroy() must invoke fuse_send_destroy() until the last superblock is released. To do so, we keep track of which fuse_mount is the root mount and perform all fuse_conn-level actions only when this fuse_mount is involved. Fixed the code for lots of 5.xx to 4.19 backports that google had previously applied without this commit. Change-Id: I4169dac1eb30b96c5f43ecadd339746be16f77fc Signed-off-by: Max Reitz Reviewed-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi --- fs/fuse/control.c | 20 +++-- fs/fuse/cuse.c | 21 +++-- fs/fuse/dax.c | 10 +-- fs/fuse/dev.c | 106 +++++++++++------------ fs/fuse/dir.c | 114 ++++++++++++------------ fs/fuse/file.c | 207 ++++++++++++++++++++++---------------------- fs/fuse/fuse_i.h | 128 ++++++++++++++++++++------- fs/fuse/inode.c | 174 ++++++++++++++++++++++++++----------- fs/fuse/readdir.c | 10 +-- fs/fuse/virtio_fs.c | 67 +++++++++----- fs/fuse/xattr.c | 34 ++++---- 11 files changed, 530 insertions(+), 361 deletions(-) diff --git a/fs/fuse/control.c b/fs/fuse/control.c index 7092958c43ed..24b4d9db231d 100644 --- a/fs/fuse/control.c +++ b/fs/fuse/control.c @@ -164,6 +164,7 @@ static ssize_t fuse_conn_congestion_threshold_write(struct file *file, { unsigned val; struct fuse_conn *fc; + struct fuse_mount *fm; ssize_t ret; ret = fuse_conn_limit_write(file, buf, count, ppos, &val, @@ -174,18 +175,27 @@ static ssize_t fuse_conn_congestion_threshold_write(struct file *file, if (!fc) goto out; + down_read(&fc->killsb); spin_lock(&fc->bg_lock); fc->congestion_threshold = val; - if (fc->sb) { + + /* + * Get any fuse_mount belonging to this fuse_conn; s_bdi is + * shared between all of them + */ + + if (!list_empty(&fc->mounts)) { + fm = list_first_entry(&fc->mounts, struct fuse_mount, fc_entry); if (fc->num_background < fc->congestion_threshold) { - clear_bdi_congested(fc->sb->s_bdi, BLK_RW_SYNC); - clear_bdi_congested(fc->sb->s_bdi, BLK_RW_ASYNC); + clear_bdi_congested(fm->sb->s_bdi, BLK_RW_SYNC); + clear_bdi_congested(fm->sb->s_bdi, BLK_RW_ASYNC); } else { - set_bdi_congested(fc->sb->s_bdi, BLK_RW_SYNC); - set_bdi_congested(fc->sb->s_bdi, BLK_RW_ASYNC); + set_bdi_congested(fm->sb->s_bdi, BLK_RW_SYNC); + set_bdi_congested(fm->sb->s_bdi, BLK_RW_ASYNC); } } spin_unlock(&fc->bg_lock); + up_read(&fc->killsb); fuse_conn_put(fc); out: return ret; diff --git a/fs/fuse/cuse.c b/fs/fuse/cuse.c index 0e25c31d55a6..a37528b51798 100644 --- a/fs/fuse/cuse.c +++ b/fs/fuse/cuse.c @@ -57,6 +57,7 @@ struct cuse_conn { struct list_head list; /* linked on cuse_conntbl */ + struct fuse_mount fm; /* Dummy mount referencing fc */ struct fuse_conn fc; /* fuse connection */ struct cdev *cdev; /* associated character device */ struct device *dev; /* device representing @cdev */ @@ -134,7 +135,7 @@ static int cuse_open(struct inode *inode, struct file *file) * Generic permission check is already done against the chrdev * file, proceed to open. */ - rc = fuse_do_open(&cc->fc, 0, file, 0); + rc = fuse_do_open(&cc->fm, 0, file, 0); if (rc) fuse_conn_put(&cc->fc); return rc; @@ -143,10 +144,10 @@ static int cuse_open(struct inode *inode, struct file *file) static int cuse_release(struct inode *inode, struct file *file) { struct fuse_file *ff = file->private_data; - struct fuse_conn *fc = ff->fc; + struct fuse_mount *fm = ff->fm; fuse_sync_release(NULL, ff, file->f_flags); - fuse_conn_put(fc); + fuse_conn_put(fm->fc); return 0; } @@ -155,7 +156,7 @@ static long cuse_file_ioctl(struct file *file, unsigned int cmd, unsigned long arg) { struct fuse_file *ff = file->private_data; - struct cuse_conn *cc = fc_to_cc(ff->fc); + struct cuse_conn *cc = fc_to_cc(ff->fm->fc); unsigned int flags = 0; if (cc->unrestricted_ioctl) @@ -168,7 +169,7 @@ static long cuse_file_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg) { struct fuse_file *ff = file->private_data; - struct cuse_conn *cc = fc_to_cc(ff->fc); + struct cuse_conn *cc = fc_to_cc(ff->fm->fc); unsigned int flags = FUSE_IOCTL_COMPAT; if (cc->unrestricted_ioctl) @@ -313,9 +314,10 @@ struct cuse_init_args { * required data structures for it. Please read the comment at the * top of this file for high level overview. */ -static void cuse_process_init_reply(struct fuse_conn *fc, +static void cuse_process_init_reply(struct fuse_mount *fm, struct fuse_args *args, int error) { + struct fuse_conn *fc = fm->fc; struct cuse_init_args *ia = container_of(args, typeof(*ia), ap.args); struct fuse_args_pages *ap = &ia->ap; struct cuse_conn *cc = fc_to_cc(fc), *pos; @@ -424,7 +426,7 @@ static int cuse_send_init(struct cuse_conn *cc) { int rc; struct page *page; - struct fuse_conn *fc = &cc->fc; + struct fuse_mount *fm = &cc->fm; struct cuse_init_args *ia; struct fuse_args_pages *ap; @@ -460,7 +462,7 @@ static int cuse_send_init(struct cuse_conn *cc) ia->desc.length = ap->args.out_args[1].size; ap->args.end = cuse_process_init_reply; - rc = fuse_simple_background(fc, &ap->args, GFP_KERNEL); + rc = fuse_simple_background(fm, &ap->args, GFP_KERNEL); if (rc) { kfree(ia); err_free_page: @@ -506,7 +508,8 @@ static int cuse_channel_open(struct inode *inode, struct file *file) * Limit the cuse channel to requests that can * be represented in file->f_cred->user_ns. */ - fuse_conn_init(&cc->fc, file->f_cred->user_ns, &fuse_dev_fiq_ops, NULL); + fuse_conn_init(&cc->fc, &cc->fm, file->f_cred->user_ns, + &fuse_dev_fiq_ops, NULL); fud = fuse_dev_alloc_install(&cc->fc); if (!fud) { diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index e394dba08cc4..ff99ab2a3c43 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -182,8 +182,8 @@ static int fuse_setup_one_mapping(struct inode *inode, unsigned long start_idx, struct fuse_dax_mapping *dmap, bool writable, bool upgrade) { - struct fuse_conn *fc = get_fuse_conn(inode); - struct fuse_conn_dax *fcd = fc->dax; + struct fuse_mount *fm = get_fuse_mount(inode); + struct fuse_conn_dax *fcd = fm->fc->dax; struct fuse_inode *fi = get_fuse_inode(inode); struct fuse_setupmapping_in inarg; loff_t offset = start_idx << FUSE_DAX_SHIFT; @@ -206,7 +206,7 @@ static int fuse_setup_one_mapping(struct inode *inode, unsigned long start_idx, args.in_numargs = 1; args.in_args[0].size = sizeof(inarg); args.in_args[0].value = &inarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err < 0) return err; dmap->writable = writable; @@ -234,7 +234,7 @@ static int fuse_send_removemapping(struct inode *inode, struct fuse_removemapping_one *remove_one) { struct fuse_inode *fi = get_fuse_inode(inode); - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); args.opcode = FUSE_REMOVEMAPPING; @@ -244,7 +244,7 @@ static int fuse_send_removemapping(struct inode *inode, args.in_args[0].value = inargp; args.in_args[1].size = inargp->count * sizeof(*remove_one); args.in_args[1].value = remove_one; - return fuse_simple_request(fc, &args); + return fuse_simple_request(fm, &args); } static int dmap_removemapping_list(struct inode *inode, unsigned int num, diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index b68346db7419..d40e58bd7c16 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -42,21 +42,21 @@ static struct fuse_dev *fuse_get_dev(struct file *file) return READ_ONCE(file->private_data); } -static void fuse_request_init(struct fuse_conn *fc, struct fuse_req *req) +static void fuse_request_init(struct fuse_mount *fm, struct fuse_req *req) { INIT_LIST_HEAD(&req->list); INIT_LIST_HEAD(&req->intr_entry); init_waitqueue_head(&req->waitq); refcount_set(&req->count, 1); __set_bit(FR_PENDING, &req->flags); - req->fc = fc; + req->fm = fm; } -static struct fuse_req *fuse_request_alloc(struct fuse_conn *fc, gfp_t flags) +static struct fuse_req *fuse_request_alloc(struct fuse_mount *fm, gfp_t flags) { struct fuse_req *req = kmem_cache_zalloc(fuse_req_cachep, flags); if (req) - fuse_request_init(fc, req); + fuse_request_init(fm, req); return req; } @@ -105,8 +105,9 @@ static void fuse_drop_waiting(struct fuse_conn *fc) static void fuse_put_request(struct fuse_req *req); -static struct fuse_req *fuse_get_req(struct fuse_conn *fc, bool for_background) +static struct fuse_req *fuse_get_req(struct fuse_mount *fm, bool for_background) { + struct fuse_conn *fc = fm->fc; struct fuse_req *req; int err; atomic_inc(&fc->num_waiting); @@ -128,7 +129,7 @@ static struct fuse_req *fuse_get_req(struct fuse_conn *fc, bool for_background) if (fc->conn_error) goto out; - req = fuse_request_alloc(fc, GFP_KERNEL); + req = fuse_request_alloc(fm, GFP_KERNEL); err = -ENOMEM; if (!req) { if (for_background) @@ -158,7 +159,7 @@ static struct fuse_req *fuse_get_req(struct fuse_conn *fc, bool for_background) static void fuse_put_request(struct fuse_req *req) { - struct fuse_conn *fc = req->fc; + struct fuse_conn *fc = req->fm->fc; if (refcount_dec_and_test(&req->count)) { if (test_bit(FR_BACKGROUND, &req->flags)) { @@ -283,7 +284,8 @@ static void flush_bg_queue(struct fuse_conn *fc) */ void fuse_request_end(struct fuse_req *req) { - struct fuse_conn *fc = req->fc; + struct fuse_mount *fm = req->fm; + struct fuse_conn *fc = fm->fc; struct fuse_iqueue *fiq = &fc->iq; if (test_and_set_bit(FR_FINISHED, &req->flags)) @@ -318,9 +320,9 @@ void fuse_request_end(struct fuse_req *req) wake_up(&fc->blocked_waitq); } - if (fc->num_background == fc->congestion_threshold && fc->sb) { - clear_bdi_congested(fc->sb->s_bdi, BLK_RW_SYNC); - clear_bdi_congested(fc->sb->s_bdi, BLK_RW_ASYNC); + if (fc->num_background == fc->congestion_threshold && fm->sb) { + clear_bdi_congested(fm->sb->s_bdi, BLK_RW_SYNC); + clear_bdi_congested(fm->sb->s_bdi, BLK_RW_ASYNC); } fc->num_background--; fc->active_background--; @@ -332,7 +334,7 @@ void fuse_request_end(struct fuse_req *req) } if (test_bit(FR_ASYNC, &req->flags)) - req->args->end(fc, req->args, req->out.h.error); + req->args->end(fm, req->args, req->out.h.error); put_request: fuse_put_request(req); } @@ -340,7 +342,7 @@ EXPORT_SYMBOL_GPL(fuse_request_end); static int queue_interrupt(struct fuse_req *req) { - struct fuse_iqueue *fiq = &req->fc->iq; + struct fuse_iqueue *fiq = &req->fm->fc->iq; spin_lock(&fiq->lock); /* Check for we've sent request to interrupt this req */ @@ -370,7 +372,7 @@ static int queue_interrupt(struct fuse_req *req) static void request_wait_answer(struct fuse_req *req) { - struct fuse_conn *fc = req->fc; + struct fuse_conn *fc = req->fm->fc; struct fuse_iqueue *fiq = &fc->iq; int err; @@ -418,7 +420,7 @@ static void request_wait_answer(struct fuse_req *req) static void __fuse_request_send(struct fuse_req *req) { - struct fuse_iqueue *fiq = &req->fc->iq; + struct fuse_iqueue *fiq = &req->fm->fc->iq; BUG_ON(test_bit(FR_BACKGROUND, &req->flags)); spin_lock(&fiq->lock); @@ -473,7 +475,7 @@ static void fuse_adjust_compat(struct fuse_conn *fc, struct fuse_args *args) static void fuse_force_creds(struct fuse_req *req) { - struct fuse_conn *fc = req->fc; + struct fuse_conn *fc = req->fm->fc; req->in.h.uid = from_kuid_munged(fc->user_ns, current_fsuid()); req->in.h.gid = from_kgid_munged(fc->user_ns, current_fsgid()); @@ -489,14 +491,15 @@ static void fuse_args_to_req(struct fuse_req *req, struct fuse_args *args) __set_bit(FR_ASYNC, &req->flags); } -ssize_t fuse_simple_request(struct fuse_conn *fc, struct fuse_args *args) +ssize_t fuse_simple_request(struct fuse_mount *fm, struct fuse_args *args) { + struct fuse_conn *fc = fm->fc; struct fuse_req *req; ssize_t ret; if (args->force) { atomic_inc(&fc->num_waiting); - req = fuse_request_alloc(fc, GFP_KERNEL | __GFP_NOFAIL); + req = fuse_request_alloc(fm, GFP_KERNEL | __GFP_NOFAIL); if (!args->nocreds) fuse_force_creds(req); @@ -505,7 +508,7 @@ ssize_t fuse_simple_request(struct fuse_conn *fc, struct fuse_args *args) __set_bit(FR_FORCE, &req->flags); } else { WARN_ON(args->nocreds); - req = fuse_get_req(fc, false); + req = fuse_get_req(fm, false); if (IS_ERR(req)) return PTR_ERR(req); } @@ -529,7 +532,8 @@ ssize_t fuse_simple_request(struct fuse_conn *fc, struct fuse_args *args) static bool fuse_request_queue_background(struct fuse_req *req) { - struct fuse_conn *fc = req->fc; + struct fuse_mount *fm = req->fm; + struct fuse_conn *fc = fm->fc; bool queued = false; WARN_ON(!test_bit(FR_BACKGROUND, &req->flags)); @@ -543,9 +547,9 @@ static bool fuse_request_queue_background(struct fuse_req *req) fc->num_background++; if (fc->num_background == fc->max_background) fc->blocked = 1; - if (fc->num_background == fc->congestion_threshold && fc->sb) { - set_bdi_congested(fc->sb->s_bdi, BLK_RW_SYNC); - set_bdi_congested(fc->sb->s_bdi, BLK_RW_ASYNC); + if (fc->num_background == fc->congestion_threshold && fm->sb) { + set_bdi_congested(fm->sb->s_bdi, BLK_RW_SYNC); + set_bdi_congested(fm->sb->s_bdi, BLK_RW_ASYNC); } list_add_tail(&req->list, &fc->bg_queue); flush_bg_queue(fc); @@ -556,20 +560,20 @@ static bool fuse_request_queue_background(struct fuse_req *req) return queued; } -int fuse_simple_background(struct fuse_conn *fc, struct fuse_args *args, +int fuse_simple_background(struct fuse_mount *fm, struct fuse_args *args, gfp_t gfp_flags) { struct fuse_req *req; if (args->force) { WARN_ON(!args->nocreds); - req = fuse_request_alloc(fc, gfp_flags); + req = fuse_request_alloc(fm, gfp_flags); if (!req) return -ENOMEM; __set_bit(FR_BACKGROUND, &req->flags); } else { WARN_ON(args->nocreds); - req = fuse_get_req(fc, true); + req = fuse_get_req(fm, true); if (IS_ERR(req)) return PTR_ERR(req); } @@ -585,14 +589,14 @@ int fuse_simple_background(struct fuse_conn *fc, struct fuse_args *args, } EXPORT_SYMBOL_GPL(fuse_simple_background); -static int fuse_simple_notify_reply(struct fuse_conn *fc, +static int fuse_simple_notify_reply(struct fuse_mount *fm, struct fuse_args *args, u64 unique) { struct fuse_req *req; - struct fuse_iqueue *fiq = &fc->iq; + struct fuse_iqueue *fiq = &fm->fc->iq; int err = 0; - req = fuse_get_req(fc, false); + req = fuse_get_req(fm, false); if (IS_ERR(req)) return PTR_ERR(req); @@ -1466,11 +1470,8 @@ static int fuse_notify_inval_inode(struct fuse_conn *fc, unsigned int size, fuse_copy_finish(cs); down_read(&fc->killsb); - err = -ENOENT; - if (fc->sb) { - err = fuse_reverse_inval_inode(fc->sb, outarg.ino, - outarg.off, outarg.len); - } + err = fuse_reverse_inval_inode(fc, outarg.ino, + outarg.off, outarg.len); up_read(&fc->killsb); return err; @@ -1516,9 +1517,7 @@ static int fuse_notify_inval_entry(struct fuse_conn *fc, unsigned int size, buf[outarg.namelen] = 0; down_read(&fc->killsb); - err = -ENOENT; - if (fc->sb) - err = fuse_reverse_inval_entry(fc->sb, outarg.parent, 0, &name); + err = fuse_reverse_inval_entry(fc, outarg.parent, 0, &name); up_read(&fc->killsb); kfree(buf); return err; @@ -1566,10 +1565,7 @@ static int fuse_notify_delete(struct fuse_conn *fc, unsigned int size, buf[outarg.namelen] = 0; down_read(&fc->killsb); - err = -ENOENT; - if (fc->sb) - err = fuse_reverse_inval_entry(fc->sb, outarg.parent, - outarg.child, &name); + err = fuse_reverse_inval_entry(fc, outarg.parent, outarg.child, &name); up_read(&fc->killsb); kfree(buf); return err; @@ -1611,10 +1607,7 @@ static int fuse_notify_store(struct fuse_conn *fc, unsigned int size, down_read(&fc->killsb); err = -ENOENT; - if (!fc->sb) - goto out_up_killsb; - - inode = ilookup5(fc->sb, nodeid, fuse_inode_eq, &nodeid); + inode = fuse_ilookup(fc, nodeid, NULL); if (!inode) goto out_up_killsb; @@ -1673,7 +1666,7 @@ struct fuse_retrieve_args { struct fuse_notify_retrieve_in inarg; }; -static void fuse_retrieve_end(struct fuse_conn *fc, struct fuse_args *args, +static void fuse_retrieve_end(struct fuse_mount *fm, struct fuse_args *args, int error) { struct fuse_retrieve_args *ra = @@ -1683,7 +1676,7 @@ static void fuse_retrieve_end(struct fuse_conn *fc, struct fuse_args *args, kfree(ra); } -static int fuse_retrieve(struct fuse_conn *fc, struct inode *inode, +static int fuse_retrieve(struct fuse_mount *fm, struct inode *inode, struct fuse_notify_retrieve_out *outarg) { int err; @@ -1694,6 +1687,7 @@ static int fuse_retrieve(struct fuse_conn *fc, struct inode *inode, unsigned int offset; size_t total_len = 0; unsigned int num_pages; + struct fuse_conn *fc = fm->fc; struct fuse_retrieve_args *ra; size_t args_size = sizeof(*ra); struct fuse_args_pages *ap; @@ -1755,9 +1749,9 @@ static int fuse_retrieve(struct fuse_conn *fc, struct inode *inode, args->in_args[0].value = &ra->inarg; args->in_args[1].size = total_len; - err = fuse_simple_notify_reply(fc, args, outarg->notify_unique); + err = fuse_simple_notify_reply(fm, args, outarg->notify_unique); if (err) - fuse_retrieve_end(fc, args, err); + fuse_retrieve_end(fm, args, err); return err; } @@ -1766,7 +1760,9 @@ static int fuse_notify_retrieve(struct fuse_conn *fc, unsigned int size, struct fuse_copy_state *cs) { struct fuse_notify_retrieve_out outarg; + struct fuse_mount *fm; struct inode *inode; + u64 nodeid; int err; err = -EINVAL; @@ -1781,14 +1777,12 @@ static int fuse_notify_retrieve(struct fuse_conn *fc, unsigned int size, down_read(&fc->killsb); err = -ENOENT; - if (fc->sb) { - u64 nodeid = outarg.nodeid; + nodeid = outarg.nodeid; - inode = ilookup5(fc->sb, nodeid, fuse_inode_eq, &nodeid); - if (inode) { - err = fuse_retrieve(fc, inode, &outarg); - iput(inode); - } + inode = fuse_ilookup(fc, nodeid, &fm); + if (inode) { + err = fuse_retrieve(fm, inode, &outarg); + iput(inode); } up_read(&fc->killsb); diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 60f7affcbde0..488cf8f633aa 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -196,7 +196,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) { struct inode *inode; struct dentry *parent; - struct fuse_conn *fc; + struct fuse_mount *fm; struct fuse_inode *fi; int ret; @@ -218,19 +218,19 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) if (flags & LOOKUP_RCU) goto out; - fc = get_fuse_conn(inode); + fm = get_fuse_mount(inode); forget = fuse_alloc_forget(); ret = -ENOMEM; if (!forget) goto out; - attr_version = fuse_get_attr_version(fc); + attr_version = fuse_get_attr_version(fm->fc); parent = dget_parent(entry); - fuse_lookup_init(fc, &args, get_node_id(d_inode(parent)), + fuse_lookup_init(fm->fc, &args, get_node_id(d_inode(parent)), &entry->d_name, &outarg); - ret = fuse_simple_request(fc, &args); + ret = fuse_simple_request(fm, &args); dput(parent); /* Zero nodeid is same as -ENOENT */ if (!ret && !outarg.nodeid) @@ -238,7 +238,8 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) if (!ret) { fi = get_fuse_inode(inode); if (outarg.nodeid != get_node_id(inode)) { - fuse_queue_forget(fc, forget, outarg.nodeid, 1); + fuse_queue_forget(fm->fc, forget, + outarg.nodeid, 1); goto invalid; } spin_lock(&fi->lock); @@ -306,7 +307,8 @@ static int fuse_dentry_delete(const struct dentry *dentry) */ static void fuse_dentry_canonical_path(const struct path *path, struct path *canonical_path) { struct inode *inode = d_inode(path->dentry); - struct fuse_conn *fc = get_fuse_conn(inode); + //struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount_super(path->mnt->mnt_sb); FUSE_ARGS(args); char *path_name; int err; @@ -324,7 +326,7 @@ static void fuse_dentry_canonical_path(const struct path *path, struct path *can args.canonical_path = canonical_path; args.out_argvar = 1; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); free_page((unsigned long)path_name); if (err > 0) return; @@ -366,7 +368,7 @@ bool fuse_invalid_attr(struct fuse_attr *attr) int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name, struct fuse_entry_out *outarg, struct inode **inode) { - struct fuse_conn *fc = get_fuse_conn_super(sb); + struct fuse_mount *fm = get_fuse_mount_super(sb); FUSE_ARGS(args); struct fuse_forget_link *forget; u64 attr_version; @@ -383,10 +385,10 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name if (!forget) goto out; - attr_version = fuse_get_attr_version(fc); + attr_version = fuse_get_attr_version(fm->fc); - fuse_lookup_init(fc, &args, nodeid, name, outarg); - err = fuse_simple_request(fc, &args); + fuse_lookup_init(fm->fc, &args, nodeid, name, outarg); + err = fuse_simple_request(fm, &args); /* Zero nodeid is same as -ENOENT, but with valid timeout */ if (err || !outarg->nodeid) goto out_put_forget; @@ -402,7 +404,7 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name attr_version); err = -ENOMEM; if (!*inode) { - fuse_queue_forget(fc, forget, outarg->nodeid, 1); + fuse_queue_forget(fm->fc, forget, outarg->nodeid, 1); goto out; } err = 0; @@ -475,6 +477,7 @@ static int fuse_create_open(struct inode *dir, struct dentry *entry, int err; struct inode *inode; struct fuse_conn *fc = get_fuse_conn(dir); + struct fuse_mount *fm = get_fuse_mount(dir); FUSE_ARGS(args); struct fuse_forget_link *forget; struct fuse_create_in inarg; @@ -492,11 +495,11 @@ static int fuse_create_open(struct inode *dir, struct dentry *entry, goto out_err; err = -ENOMEM; - ff = fuse_file_alloc(fc); + ff = fuse_file_alloc(fm); if (!ff) goto out_put_forget_req; - if (!fc->dont_mask) + if (!fm->fc->dont_mask) mode &= ~current_umask(); flags &= ~O_NOCTTY; @@ -517,7 +520,7 @@ static int fuse_create_open(struct inode *dir, struct dentry *entry, args.out_args[0].value = &outentry; args.out_args[1].size = sizeof(outopen); args.out_args[1].value = &outopen; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err) goto out_free_ff; @@ -535,7 +538,7 @@ static int fuse_create_open(struct inode *dir, struct dentry *entry, if (!inode) { flags &= ~(O_CREAT | O_EXCL | O_TRUNC); fuse_sync_release(NULL, ff, flags); - fuse_queue_forget(fc, forget, outentry.nodeid, 1); + fuse_queue_forget(fm->fc, forget, outentry.nodeid, 1); err = -ENOMEM; goto out_err; } @@ -611,7 +614,7 @@ no_open: /* * Code shared between mknod, mkdir, symlink and link */ -static int create_new_entry(struct fuse_conn *fc, struct fuse_args *args, +static int create_new_entry(struct fuse_mount *fm, struct fuse_args *args, struct inode *dir, struct dentry *entry, umode_t mode) { @@ -633,7 +636,7 @@ static int create_new_entry(struct fuse_conn *fc, struct fuse_args *args, args->out_numargs = 1; args->out_args[0].size = sizeof(outarg); args->out_args[0].value = &outarg; - err = fuse_simple_request(fc, args); + err = fuse_simple_request(fm, args); if (err) goto out_put_forget_req; @@ -647,7 +650,7 @@ static int create_new_entry(struct fuse_conn *fc, struct fuse_args *args, inode = fuse_iget(dir->i_sb, outarg.nodeid, outarg.generation, &outarg.attr, entry_attr_timeout(&outarg), 0); if (!inode) { - fuse_queue_forget(fc, forget, outarg.nodeid, 1); + fuse_queue_forget(fm->fc, forget, outarg.nodeid, 1); return -ENOMEM; } kfree(forget); @@ -675,10 +678,10 @@ static int fuse_mknod(struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) { struct fuse_mknod_in inarg; - struct fuse_conn *fc = get_fuse_conn(dir); + struct fuse_mount *fm = get_fuse_mount(dir); FUSE_ARGS(args); - if (!fc->dont_mask) + if (!fm->fc->dont_mask) mode &= ~current_umask(); memset(&inarg, 0, sizeof(inarg)); @@ -691,7 +694,7 @@ static int fuse_mknod(struct inode *dir, struct dentry *entry, umode_t mode, args.in_args[0].value = &inarg; args.in_args[1].size = entry->d_name.len + 1; args.in_args[1].value = entry->d_name.name; - return create_new_entry(fc, &args, dir, entry, mode); + return create_new_entry(fm, &args, dir, entry, mode); } static int fuse_create(struct inode *dir, struct dentry *entry, umode_t mode, @@ -703,10 +706,10 @@ static int fuse_create(struct inode *dir, struct dentry *entry, umode_t mode, static int fuse_mkdir(struct inode *dir, struct dentry *entry, umode_t mode) { struct fuse_mkdir_in inarg; - struct fuse_conn *fc = get_fuse_conn(dir); + struct fuse_mount *fm = get_fuse_mount(dir); FUSE_ARGS(args); - if (!fc->dont_mask) + if (!fm->fc->dont_mask) mode &= ~current_umask(); memset(&inarg, 0, sizeof(inarg)); @@ -718,13 +721,13 @@ static int fuse_mkdir(struct inode *dir, struct dentry *entry, umode_t mode) args.in_args[0].value = &inarg; args.in_args[1].size = entry->d_name.len + 1; args.in_args[1].value = entry->d_name.name; - return create_new_entry(fc, &args, dir, entry, S_IFDIR); + return create_new_entry(fm, &args, dir, entry, S_IFDIR); } static int fuse_symlink(struct inode *dir, struct dentry *entry, const char *link) { - struct fuse_conn *fc = get_fuse_conn(dir); + struct fuse_mount *fm = get_fuse_mount(dir); unsigned len = strlen(link) + 1; FUSE_ARGS(args); @@ -734,7 +737,7 @@ static int fuse_symlink(struct inode *dir, struct dentry *entry, args.in_args[0].value = entry->d_name.name; args.in_args[1].size = len; args.in_args[1].value = link; - return create_new_entry(fc, &args, dir, entry, S_IFLNK); + return create_new_entry(fm, &args, dir, entry, S_IFLNK); } void fuse_update_ctime(struct inode *inode) @@ -748,7 +751,7 @@ void fuse_update_ctime(struct inode *inode) static int fuse_unlink(struct inode *dir, struct dentry *entry) { int err; - struct fuse_conn *fc = get_fuse_conn(dir); + struct fuse_mount *fm = get_fuse_mount(dir); FUSE_ARGS(args); if (fuse_is_bad(dir)) @@ -759,13 +762,13 @@ static int fuse_unlink(struct inode *dir, struct dentry *entry) args.in_numargs = 1; args.in_args[0].size = entry->d_name.len + 1; args.in_args[0].value = entry->d_name.name; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (!err) { struct inode *inode = d_inode(entry); struct fuse_inode *fi = get_fuse_inode(inode); spin_lock(&fi->lock); - fi->attr_version = atomic64_inc_return(&fc->attr_version); + fi->attr_version = atomic64_inc_return(&fm->fc->attr_version); /* * If i_nlink == 0 then unlink doesn't make sense, yet this can * happen if userspace filesystem is careless. It would be @@ -787,7 +790,7 @@ static int fuse_unlink(struct inode *dir, struct dentry *entry) static int fuse_rmdir(struct inode *dir, struct dentry *entry) { int err; - struct fuse_conn *fc = get_fuse_conn(dir); + struct fuse_mount *fm = get_fuse_mount(dir); FUSE_ARGS(args); if (fuse_is_bad(dir)) @@ -798,7 +801,7 @@ static int fuse_rmdir(struct inode *dir, struct dentry *entry) args.in_numargs = 1; args.in_args[0].size = entry->d_name.len + 1; args.in_args[0].value = entry->d_name.name; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (!err) { clear_nlink(d_inode(entry)); fuse_dir_changed(dir); @@ -814,7 +817,7 @@ static int fuse_rename_common(struct inode *olddir, struct dentry *oldent, { int err; struct fuse_rename2_in inarg; - struct fuse_conn *fc = get_fuse_conn(olddir); + struct fuse_mount *fm = get_fuse_mount(olddir); FUSE_ARGS(args); memset(&inarg, 0, argsize); @@ -829,7 +832,7 @@ static int fuse_rename_common(struct inode *olddir, struct dentry *oldent, args.in_args[1].value = oldent->d_name.name; args.in_args[2].size = newent->d_name.len + 1; args.in_args[2].value = newent->d_name.name; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (!err) { /* ctime changes */ fuse_invalidate_attr(d_inode(oldent)); @@ -903,7 +906,7 @@ static int fuse_link(struct dentry *entry, struct inode *newdir, int err; struct fuse_link_in inarg; struct inode *inode = d_inode(entry); - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); memset(&inarg, 0, sizeof(inarg)); @@ -914,7 +917,7 @@ static int fuse_link(struct dentry *entry, struct inode *newdir, args.in_args[0].value = &inarg; args.in_args[1].size = newent->d_name.len + 1; args.in_args[1].value = newent->d_name.name; - err = create_new_entry(fc, &args, newdir, newent, inode->i_mode); + err = create_new_entry(fm, &args, newdir, newent, inode->i_mode); /* Contrary to "normal" filesystems it can happen that link makes two "logical" inodes point to the same "physical" inode. We invalidate the attributes of the old one, so it @@ -925,7 +928,7 @@ static int fuse_link(struct dentry *entry, struct inode *newdir, struct fuse_inode *fi = get_fuse_inode(inode); spin_lock(&fi->lock); - fi->attr_version = atomic64_inc_return(&fc->attr_version); + fi->attr_version = atomic64_inc_return(&fm->fc->attr_version); if (likely(inode->i_nlink < UINT_MAX)) inc_nlink(inode); spin_unlock(&fi->lock); @@ -982,11 +985,11 @@ static int fuse_do_getattr(struct inode *inode, struct kstat *stat, int err; struct fuse_getattr_in inarg; struct fuse_attr_out outarg; - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); u64 attr_version; - attr_version = fuse_get_attr_version(fc); + attr_version = fuse_get_attr_version(fm->fc); memset(&inarg, 0, sizeof(inarg)); memset(&outarg, 0, sizeof(outarg)); @@ -1005,7 +1008,7 @@ static int fuse_do_getattr(struct inode *inode, struct kstat *stat, args.out_numargs = 1; args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (!err) { if (fuse_invalid_attr(&outarg.attr) || (inode->i_mode ^ outarg.attr.mode) & S_IFMT) { @@ -1058,7 +1061,7 @@ int fuse_update_attributes(struct inode *inode, struct file *file) STATX_BASIC_STATS & ~STATX_ATIME, 0); } -int fuse_reverse_inval_entry(struct super_block *sb, u64 parent_nodeid, +int fuse_reverse_inval_entry(struct fuse_conn *fc, u64 parent_nodeid, u64 child_nodeid, struct qstr *name) { int err = -ENOTDIR; @@ -1066,7 +1069,7 @@ int fuse_reverse_inval_entry(struct super_block *sb, u64 parent_nodeid, struct dentry *dir; struct dentry *entry; - parent = ilookup5(sb, parent_nodeid, fuse_inode_eq, &parent_nodeid); + parent = fuse_ilookup(fc, parent_nodeid, NULL); if (!parent) return -ENOENT; @@ -1158,14 +1161,14 @@ int fuse_allow_current_process(struct fuse_conn *fc) static int fuse_access(struct inode *inode, int mask) { - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); struct fuse_access_in inarg; int err; BUG_ON(mask & MAY_NOT_BLOCK); - if (fc->no_access) + if (fm->fc->no_access) return 0; memset(&inarg, 0, sizeof(inarg)); @@ -1175,9 +1178,9 @@ static int fuse_access(struct inode *inode, int mask) args.in_numargs = 1; args.in_args[0].size = sizeof(inarg); args.in_args[0].value = &inarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err == -ENOSYS) { - fc->no_access = 1; + fm->fc->no_access = 1; err = 0; } return err; @@ -1268,7 +1271,7 @@ static int fuse_permission(struct inode *inode, int mask) static int fuse_readlink_page(struct inode *inode, struct page *page) { - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); struct fuse_page_desc desc = { .length = PAGE_SIZE - 1 }; struct fuse_args_pages ap = { .num_pages = 1, @@ -1285,7 +1288,7 @@ static int fuse_readlink_page(struct inode *inode, struct page *page) ap.args.page_zeroing = true; ap.args.out_numargs = 1; ap.args.out_args[0].size = desc.length; - res = fuse_simple_request(fc, &ap.args); + res = fuse_simple_request(fm, &ap.args); fuse_invalidate_atime(inode); @@ -1513,7 +1516,7 @@ static void fuse_setattr_fill(struct fuse_conn *fc, struct fuse_args *args, */ int fuse_flush_times(struct inode *inode, struct fuse_file *ff) { - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); struct fuse_setattr_in inarg; struct fuse_attr_out outarg; @@ -1524,7 +1527,7 @@ int fuse_flush_times(struct inode *inode, struct fuse_file *ff) inarg.valid = FATTR_MTIME; inarg.mtime = inode->i_mtime.tv_sec; inarg.mtimensec = inode->i_mtime.tv_nsec; - if (fc->minor >= 23) { + if (fm->fc->minor >= 23) { inarg.valid |= FATTR_CTIME; inarg.ctime = inode->i_ctime.tv_sec; inarg.ctimensec = inode->i_ctime.tv_nsec; @@ -1533,9 +1536,9 @@ int fuse_flush_times(struct inode *inode, struct fuse_file *ff) inarg.valid |= FATTR_FH; inarg.fh = ff->fh; } - fuse_setattr_fill(fc, &args, inode, &inarg, &outarg); + fuse_setattr_fill(fm->fc, &args, inode, &inarg, &outarg); - return fuse_simple_request(fc, &args); + return fuse_simple_request(fm, &args); } /* @@ -1550,7 +1553,8 @@ int fuse_do_setattr(struct dentry *dentry, struct iattr *attr, struct file *file) { struct inode *inode = d_inode(dentry); - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); + struct fuse_conn *fc = fm->fc; struct fuse_inode *fi = get_fuse_inode(inode); FUSE_ARGS(args); struct fuse_setattr_in inarg; @@ -1636,7 +1640,7 @@ int fuse_do_setattr(struct dentry *dentry, struct iattr *attr, inarg.lock_owner = fuse_lock_owner_id(fc, current->files); } fuse_setattr_fill(fc, &args, inode, &inarg, &outarg); - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err) { if (err == -EINTR) fuse_invalidate_attr(inode); diff --git a/fs/fuse/file.c b/fs/fuse/file.c index a3ce2e78b7d5..66ccb8ee4889 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -32,7 +32,7 @@ static struct page **fuse_pages_alloc(unsigned int npages, gfp_t flags, return pages; } -static int fuse_send_open(struct fuse_conn *fc, u64 nodeid, struct file *file, +static int fuse_send_open(struct fuse_mount *fm, u64 nodeid, struct file *file, int opcode, struct fuse_open_out *outargp) { struct fuse_open_in inarg; @@ -40,7 +40,7 @@ static int fuse_send_open(struct fuse_conn *fc, u64 nodeid, struct file *file, memset(&inarg, 0, sizeof(inarg)); inarg.flags = file->f_flags & ~(O_CREAT | O_EXCL | O_NOCTTY); - if (!fc->atomic_o_trunc) + if (!fm->fc->atomic_o_trunc) inarg.flags &= ~O_TRUNC; args.opcode = opcode; args.nodeid = nodeid; @@ -51,7 +51,7 @@ static int fuse_send_open(struct fuse_conn *fc, u64 nodeid, struct file *file, args.out_args[0].size = sizeof(*outargp); args.out_args[0].value = outargp; - return fuse_simple_request(fc, &args); + return fuse_simple_request(fm, &args); } struct fuse_release_args { @@ -60,7 +60,7 @@ struct fuse_release_args { struct inode *inode; }; -struct fuse_file *fuse_file_alloc(struct fuse_conn *fc) +struct fuse_file *fuse_file_alloc(struct fuse_mount *fm) { struct fuse_file *ff; @@ -68,7 +68,7 @@ struct fuse_file *fuse_file_alloc(struct fuse_conn *fc) if (unlikely(!ff)) return NULL; - ff->fc = fc; + ff->fm = fm; ff->release_args = kzalloc(sizeof(*ff->release_args), GFP_KERNEL_ACCOUNT); if (!ff->release_args) { @@ -82,7 +82,7 @@ struct fuse_file *fuse_file_alloc(struct fuse_conn *fc) RB_CLEAR_NODE(&ff->polled_node); init_waitqueue_head(&ff->poll_wait); - ff->kh = atomic64_inc_return(&fc->khctr); + ff->kh = atomic64_inc_return(&fm->fc->khctr); return ff; } @@ -100,7 +100,7 @@ static struct fuse_file *fuse_file_get(struct fuse_file *ff) return ff; } -static void fuse_release_end(struct fuse_conn *fc, struct fuse_args *args, +static void fuse_release_end(struct fuse_mount *fm, struct fuse_args *args, int error) { struct fuse_release_args *ra = container_of(args, typeof(*ra), args); @@ -114,29 +114,30 @@ static void fuse_file_put(struct fuse_file *ff, bool sync, bool isdir) if (refcount_dec_and_test(&ff->count)) { struct fuse_args *args = &ff->release_args->args; - if (isdir ? ff->fc->no_opendir : ff->fc->no_open) { + if (isdir ? ff->fm->fc->no_opendir : ff->fm->fc->no_open) { /* Do nothing when client does not implement 'open' */ - fuse_release_end(ff->fc, args, 0); + fuse_release_end(ff->fm, args, 0); } else if (sync) { - fuse_simple_request(ff->fc, args); - fuse_release_end(ff->fc, args, 0); + fuse_simple_request(ff->fm, args); + fuse_release_end(ff->fm, args, 0); } else { args->end = fuse_release_end; - if (fuse_simple_background(ff->fc, args, + if (fuse_simple_background(ff->fm, args, GFP_KERNEL | __GFP_NOFAIL)) - fuse_release_end(ff->fc, args, -ENOTCONN); + fuse_release_end(ff->fm, args, -ENOTCONN); } kfree(ff); } } -int fuse_do_open(struct fuse_conn *fc, u64 nodeid, struct file *file, +int fuse_do_open(struct fuse_mount *fm, u64 nodeid, struct file *file, bool isdir) { + struct fuse_conn *fc = fm->fc; struct fuse_file *ff; int opcode = isdir ? FUSE_OPENDIR : FUSE_OPEN; - ff = fuse_file_alloc(fc); + ff = fuse_file_alloc(fm); if (!ff) return -ENOMEM; @@ -147,7 +148,7 @@ int fuse_do_open(struct fuse_conn *fc, u64 nodeid, struct file *file, struct fuse_open_out outarg; int err; - err = fuse_send_open(fc, nodeid, file, opcode, &outarg); + err = fuse_send_open(fm, nodeid, file, opcode, &outarg); if (!err) { ff->fh = outarg.fh; ff->open_flags = outarg.open_flags; @@ -219,7 +220,8 @@ void fuse_finish_open(struct inode *inode, struct file *file) int fuse_open_common(struct inode *inode, struct file *file, bool isdir) { - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); + struct fuse_conn *fc = fm->fc; int err; bool is_wb_truncate = (file->f_flags & O_TRUNC) && fc->atomic_o_trunc && @@ -246,7 +248,7 @@ int fuse_open_common(struct inode *inode, struct file *file, bool isdir) goto out; } - err = fuse_do_open(fc, get_node_id(inode), file, isdir); + err = fuse_do_open(fm, get_node_id(inode), file, isdir); if (!err) fuse_finish_open(inode, file); @@ -265,7 +267,7 @@ out: static void fuse_prepare_release(struct fuse_inode *fi, struct fuse_file *ff, int flags, int opcode) { - struct fuse_conn *fc = ff->fc; + struct fuse_conn *fc = ff->fm->fc; struct fuse_release_args *ra = ff->release_args; /* Inode is NULL on error path of fuse_create_open() */ @@ -305,7 +307,7 @@ void fuse_release_common(struct file *file, bool isdir) if (ff->flock) { ra->inarg.release_flags |= FUSE_RELEASE_FLOCK_UNLOCK; - ra->inarg.lock_owner = fuse_lock_owner_id(ff->fc, + ra->inarg.lock_owner = fuse_lock_owner_id(ff->fm->fc, (fl_owner_t) file); } /* Hold inode until release is finished */ @@ -320,7 +322,7 @@ void fuse_release_common(struct file *file, bool isdir) * synchronous RELEASE is allowed (and desirable) in this case * because the server can be trusted not to screw up. */ - fuse_file_put(ff, ff->fc->destroy, isdir); + fuse_file_put(ff, ff->fm->fc->destroy, isdir); } static int fuse_open(struct inode *inode, struct file *file) @@ -463,7 +465,7 @@ static void fuse_sync_writes(struct inode *inode) static int fuse_flush(struct file *file, fl_owner_t id) { struct inode *inode = file_inode(file); - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); struct fuse_file *ff = file->private_data; struct fuse_flush_in inarg; FUSE_ARGS(args); @@ -485,12 +487,12 @@ static int fuse_flush(struct file *file, fl_owner_t id) return err; err = 0; - if (fc->no_flush) + if (fm->fc->no_flush) goto inval_attr_out; memset(&inarg, 0, sizeof(inarg)); inarg.fh = ff->fh; - inarg.lock_owner = fuse_lock_owner_id(fc, id); + inarg.lock_owner = fuse_lock_owner_id(fm->fc, id); args.opcode = FUSE_FLUSH; args.nodeid = get_node_id(inode); args.in_numargs = 1; @@ -498,9 +500,9 @@ static int fuse_flush(struct file *file, fl_owner_t id) args.in_args[0].value = &inarg; args.force = true; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err == -ENOSYS) { - fc->no_flush = 1; + fm->fc->no_flush = 1; err = 0; } @@ -509,7 +511,7 @@ inval_attr_out: * In memory i_blocks is not maintained by fuse, if writeback cache is * enabled, i_blocks from cached attr may not be accurate. */ - if (!err && fc->writeback_cache) + if (!err && fm->fc->writeback_cache) fuse_invalidate_attr(inode); return err; } @@ -518,7 +520,7 @@ int fuse_fsync_common(struct file *file, loff_t start, loff_t end, int datasync, int opcode) { struct inode *inode = file->f_mapping->host; - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); struct fuse_file *ff = file->private_data; FUSE_ARGS(args); struct fuse_fsync_in inarg; @@ -531,7 +533,7 @@ int fuse_fsync_common(struct file *file, loff_t start, loff_t end, args.in_numargs = 1; args.in_args[0].size = sizeof(inarg); args.in_args[0].value = &inarg; - return fuse_simple_request(fc, &args); + return fuse_simple_request(fm, &args); } static int fuse_fsync(struct file *file, loff_t start, loff_t end, @@ -706,7 +708,7 @@ static void fuse_io_free(struct fuse_io_args *ia) kfree(ia); } -static void fuse_aio_complete_req(struct fuse_conn *fc, struct fuse_args *args, +static void fuse_aio_complete_req(struct fuse_mount *fm, struct fuse_args *args, int err) { struct fuse_io_args *ia = container_of(args, typeof(*ia), ap.args); @@ -735,7 +737,7 @@ static void fuse_aio_complete_req(struct fuse_conn *fc, struct fuse_args *args, fuse_io_free(ia); } -static ssize_t fuse_async_req_send(struct fuse_conn *fc, +static ssize_t fuse_async_req_send(struct fuse_mount *fm, struct fuse_io_args *ia, size_t num_bytes) { ssize_t err; @@ -749,9 +751,9 @@ static ssize_t fuse_async_req_send(struct fuse_conn *fc, ia->ap.args.end = fuse_aio_complete_req; ia->ap.args.may_block = io->should_dirty; - err = fuse_simple_background(fc, &ia->ap.args, GFP_KERNEL); + err = fuse_simple_background(fm, &ia->ap.args, GFP_KERNEL); if (err) - fuse_aio_complete_req(fc, &ia->ap.args, err); + fuse_aio_complete_req(fm, &ia->ap.args, err); return num_bytes; } @@ -761,18 +763,18 @@ static ssize_t fuse_send_read(struct fuse_io_args *ia, loff_t pos, size_t count, { struct file *file = ia->io->iocb->ki_filp; struct fuse_file *ff = file->private_data; - struct fuse_conn *fc = ff->fc; + struct fuse_mount *fm = ff->fm; fuse_read_args_fill(ia, file, pos, count, FUSE_READ); if (owner != NULL) { ia->read.in.read_flags |= FUSE_READ_LOCKOWNER; - ia->read.in.lock_owner = fuse_lock_owner_id(fc, owner); + ia->read.in.lock_owner = fuse_lock_owner_id(fm->fc, owner); } if (ia->io->async) - return fuse_async_req_send(fc, ia, count); + return fuse_async_req_send(fm, ia, count); - return fuse_simple_request(fc, &ia->ap.args); + return fuse_simple_request(fm, &ia->ap.args); } static void fuse_read_update_size(struct inode *inode, loff_t size, @@ -818,7 +820,7 @@ static void fuse_short_read(struct inode *inode, u64 attr_ver, size_t num_read, static int fuse_do_readpage(struct file *file, struct page *page) { struct inode *inode = page->mapping->host; - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); loff_t pos = page_offset(page); struct fuse_page_desc desc = { .length = PAGE_SIZE }; struct fuse_io_args ia = { @@ -838,14 +840,14 @@ static int fuse_do_readpage(struct file *file, struct page *page) */ fuse_wait_on_page_writeback(inode, page->index); - attr_ver = fuse_get_attr_version(fc); + attr_ver = fuse_get_attr_version(fm->fc); /* Don't overflow end offset */ if (pos + (desc.length - 1) == LLONG_MAX) desc.length--; fuse_read_args_fill(&ia, file, pos, desc.length, FUSE_READ); - res = fuse_simple_request(fc, &ia.ap.args); + res = fuse_simple_request(fm, &ia.ap.args); if (res < 0) return res; /* @@ -875,7 +877,7 @@ static int fuse_readpage(struct file *file, struct page *page) return err; } -static void fuse_readpages_end(struct fuse_conn *fc, struct fuse_args *args, +static void fuse_readpages_end(struct fuse_mount *fm, struct fuse_args *args, int err) { int i; @@ -919,7 +921,7 @@ static void fuse_readpages_end(struct fuse_conn *fc, struct fuse_args *args, static void fuse_send_readpages(struct fuse_io_args *ia, struct file *file) { struct fuse_file *ff = file->private_data; - struct fuse_conn *fc = ff->fc; + struct fuse_mount *fm = ff->fm; struct fuse_args_pages *ap = &ia->ap; loff_t pos = page_offset(ap->pages[0]); size_t count = ap->num_pages << PAGE_SHIFT; @@ -938,18 +940,18 @@ static void fuse_send_readpages(struct fuse_io_args *ia, struct file *file) WARN_ON((loff_t) (pos + count) < 0); fuse_read_args_fill(ia, file, pos, count, FUSE_READ); - ia->read.attr_ver = fuse_get_attr_version(fc); - if (fc->async_read) { + ia->read.attr_ver = fuse_get_attr_version(fm->fc); + if (fm->fc->async_read) { ia->ff = fuse_file_get(ff); ap->args.end = fuse_readpages_end; - err = fuse_simple_background(fc, &ap->args, GFP_KERNEL); + err = fuse_simple_background(fm, &ap->args, GFP_KERNEL); if (!err) return; } else { - res = fuse_simple_request(fc, &ap->args); + res = fuse_simple_request(fm, &ap->args); err = res < 0 ? res : 0; } - fuse_readpages_end(fc, &ap->args, err); + fuse_readpages_end(fm, &ap->args, err); } struct fuse_fill_data { @@ -1064,7 +1066,7 @@ static void fuse_write_args_fill(struct fuse_io_args *ia, struct fuse_file *ff, args->opcode = FUSE_WRITE; args->nodeid = ff->nodeid; args->in_numargs = 2; - if (ff->fc->minor < 9) + if (ff->fm->fc->minor < 9) args->in_args[0].size = FUSE_COMPAT_WRITE_IN_SIZE; else args->in_args[0].size = sizeof(ia->write.in); @@ -1093,7 +1095,7 @@ static ssize_t fuse_send_write(struct fuse_io_args *ia, loff_t pos, struct kiocb *iocb = ia->io->iocb; struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; - struct fuse_conn *fc = ff->fc; + struct fuse_mount *fm = ff->fm; struct fuse_write_in *inarg = &ia->write.in; ssize_t err; @@ -1101,13 +1103,13 @@ static ssize_t fuse_send_write(struct fuse_io_args *ia, loff_t pos, inarg->flags = fuse_write_flags(iocb); if (owner != NULL) { inarg->write_flags |= FUSE_WRITE_LOCKOWNER; - inarg->lock_owner = fuse_lock_owner_id(fc, owner); + inarg->lock_owner = fuse_lock_owner_id(fm->fc, owner); } if (ia->io->async) - return fuse_async_req_send(fc, ia, count); + return fuse_async_req_send(fm, ia, count); - err = fuse_simple_request(fc, &ia->ap.args); + err = fuse_simple_request(fm, &ia->ap.args); if (!err && ia->write.out.size > count) err = -EIO; @@ -1138,7 +1140,7 @@ static ssize_t fuse_send_write_pages(struct fuse_io_args *ia, struct fuse_args_pages *ap = &ia->ap; struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; - struct fuse_conn *fc = ff->fc; + struct fuse_mount *fm = ff->fm; unsigned int offset, i; bool short_write; int err; @@ -1149,7 +1151,7 @@ static ssize_t fuse_send_write_pages(struct fuse_io_args *ia, fuse_write_args_fill(ia, ff, pos, count); ia->write.in.flags = fuse_write_flags(iocb); - err = fuse_simple_request(fc, &ap->args); + err = fuse_simple_request(fm, &ap->args); if (!err && ia->write.out.size > count) err = -EIO; @@ -1481,7 +1483,7 @@ ssize_t fuse_direct_io(struct fuse_io_priv *io, struct iov_iter *iter, struct file *file = io->iocb->ki_filp; struct inode *inode = file->f_mapping->host; struct fuse_file *ff = file->private_data; - struct fuse_conn *fc = ff->fc; + struct fuse_conn *fc = ff->fm->fc; size_t nmax = write ? fc->max_write : fc->max_read; loff_t pos = *ppos; size_t count = iov_iter_count(iter); @@ -1672,7 +1674,7 @@ static void fuse_writepage_free(struct fuse_writepage_args *wpa) kfree(wpa); } -static void fuse_writepage_finish(struct fuse_conn *fc, +static void fuse_writepage_finish(struct fuse_mount *fm, struct fuse_writepage_args *wpa) { struct fuse_args_pages *ap = &wpa->ia.ap; @@ -1690,7 +1692,7 @@ static void fuse_writepage_finish(struct fuse_conn *fc, } /* Called under fi->lock, may release and reacquire it */ -static void fuse_send_writepage(struct fuse_conn *fc, +static void fuse_send_writepage(struct fuse_mount *fm, struct fuse_writepage_args *wpa, loff_t size) __releases(fi->lock) __acquires(fi->lock) @@ -1716,10 +1718,10 @@ __acquires(fi->lock) args->force = true; args->nocreds = true; - err = fuse_simple_background(fc, args, GFP_ATOMIC); + err = fuse_simple_background(fm, args, GFP_ATOMIC); if (err == -ENOMEM) { spin_unlock(&fi->lock); - err = fuse_simple_background(fc, args, GFP_NOFS | __GFP_NOFAIL); + err = fuse_simple_background(fm, args, GFP_NOFS | __GFP_NOFAIL); spin_lock(&fi->lock); } @@ -1732,7 +1734,7 @@ __acquires(fi->lock) out_free: fi->writectr--; rb_erase(&wpa->writepages_entry, &fi->writepages); - fuse_writepage_finish(fc, wpa); + fuse_writepage_finish(fm, wpa); spin_unlock(&fi->lock); /* After rb_erase() aux request list is private */ @@ -1762,7 +1764,7 @@ void fuse_flush_writepages(struct inode *inode) __releases(fi->lock) __acquires(fi->lock) { - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); struct fuse_inode *fi = get_fuse_inode(inode); loff_t crop = i_size_read(inode); struct fuse_writepage_args *wpa; @@ -1771,7 +1773,7 @@ __acquires(fi->lock) wpa = list_entry(fi->queued_writes.next, struct fuse_writepage_args, queue_entry); list_del_init(&wpa->queue_entry); - fuse_send_writepage(fc, wpa, crop); + fuse_send_writepage(fm, wpa, crop); } } @@ -1812,7 +1814,7 @@ static void tree_insert(struct rb_root *root, struct fuse_writepage_args *wpa) WARN_ON(fuse_insert_writeback(root, wpa)); } -static void fuse_writepage_end(struct fuse_conn *fc, struct fuse_args *args, +static void fuse_writepage_end(struct fuse_mount *fm, struct fuse_args *args, int error) { struct fuse_writepage_args *wpa = @@ -1824,7 +1826,7 @@ static void fuse_writepage_end(struct fuse_conn *fc, struct fuse_args *args, spin_lock(&fi->lock); rb_erase(&wpa->writepages_entry, &fi->writepages); while (wpa->next) { - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); struct fuse_write_in *inarg = &wpa->ia.write.in; struct fuse_writepage_args *next = wpa->next; @@ -1856,10 +1858,10 @@ static void fuse_writepage_end(struct fuse_conn *fc, struct fuse_args *args, * no invocations of fuse_writepage_end() while we're in * fuse_set_nowrite..fuse_release_nowrite section. */ - fuse_send_writepage(fc, next, inarg->offset + inarg->size); + fuse_send_writepage(fm, next, inarg->offset + inarg->size); } fi->writectr--; - fuse_writepage_finish(fc, wpa); + fuse_writepage_finish(fm, wpa); spin_unlock(&fi->lock); fuse_writepage_free(wpa); } @@ -2502,7 +2504,7 @@ static void fuse_lk_fill(struct fuse_args *args, struct file *file, static int fuse_getlk(struct file *file, struct file_lock *fl) { struct inode *inode = file_inode(file); - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); struct fuse_lk_in inarg; struct fuse_lk_out outarg; @@ -2512,9 +2514,9 @@ static int fuse_getlk(struct file *file, struct file_lock *fl) args.out_numargs = 1; args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (!err) - err = convert_fuse_file_lock(fc, &outarg.lk, fl); + err = convert_fuse_file_lock(fm->fc, &outarg.lk, fl); return err; } @@ -2522,12 +2524,12 @@ static int fuse_getlk(struct file *file, struct file_lock *fl) static int fuse_setlk(struct file *file, struct file_lock *fl, int flock) { struct inode *inode = file_inode(file); - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); struct fuse_lk_in inarg; int opcode = (fl->fl_flags & FL_SLEEP) ? FUSE_SETLKW : FUSE_SETLK; struct pid *pid = fl->fl_type != F_UNLCK ? task_tgid(current) : NULL; - pid_t pid_nr = pid_nr_ns(pid, fc->pid_ns); + pid_t pid_nr = pid_nr_ns(pid, fm->fc->pid_ns); int err; if (fl->fl_lmops && fl->fl_lmops->lm_grant) { @@ -2540,7 +2542,7 @@ static int fuse_setlk(struct file *file, struct file_lock *fl, int flock) return 0; fuse_lk_fill(&args, file, fl, opcode, pid_nr, flock, &inarg); - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); /* locking is restartable */ if (err == -EINTR) @@ -2594,13 +2596,13 @@ static int fuse_file_flock(struct file *file, int cmd, struct file_lock *fl) static sector_t fuse_bmap(struct address_space *mapping, sector_t block) { struct inode *inode = mapping->host; - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); struct fuse_bmap_in inarg; struct fuse_bmap_out outarg; int err; - if (!inode->i_sb->s_bdev || fc->no_bmap) + if (!inode->i_sb->s_bdev || fm->fc->no_bmap) return 0; memset(&inarg, 0, sizeof(inarg)); @@ -2614,9 +2616,9 @@ static sector_t fuse_bmap(struct address_space *mapping, sector_t block) args.out_numargs = 1; args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err == -ENOSYS) - fc->no_bmap = 1; + fm->fc->no_bmap = 1; return err ? 0 : outarg.block; } @@ -2624,7 +2626,7 @@ static sector_t fuse_bmap(struct address_space *mapping, sector_t block) static loff_t fuse_lseek(struct file *file, loff_t offset, int whence) { struct inode *inode = file->f_mapping->host; - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); struct fuse_file *ff = file->private_data; FUSE_ARGS(args); struct fuse_lseek_in inarg = { @@ -2635,7 +2637,7 @@ static loff_t fuse_lseek(struct file *file, loff_t offset, int whence) struct fuse_lseek_out outarg; int err; - if (fc->no_lseek) + if (fm->fc->no_lseek) goto fallback; args.opcode = FUSE_LSEEK; @@ -2646,10 +2648,10 @@ static loff_t fuse_lseek(struct file *file, loff_t offset, int whence) args.out_numargs = 1; args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err) { if (err == -ENOSYS) { - fc->no_lseek = 1; + fm->fc->no_lseek = 1; goto fallback; } return err; @@ -2835,7 +2837,7 @@ long fuse_do_ioctl(struct file *file, unsigned int cmd, unsigned long arg, unsigned int flags) { struct fuse_file *ff = file->private_data; - struct fuse_conn *fc = ff->fc; + struct fuse_mount *fm = ff->fm; struct fuse_ioctl_in inarg = { .fh = ff->fh, .cmd = cmd, @@ -2868,12 +2870,12 @@ long fuse_do_ioctl(struct file *file, unsigned int cmd, unsigned long arg, BUILD_BUG_ON(sizeof(struct fuse_ioctl_iovec) * FUSE_IOCTL_MAX_IOV > PAGE_SIZE); err = -ENOMEM; - ap.pages = fuse_pages_alloc(fc->max_pages, GFP_KERNEL, &ap.descs); + ap.pages = fuse_pages_alloc(fm->fc->max_pages, GFP_KERNEL, &ap.descs); iov_page = (struct iovec *) __get_free_page(GFP_KERNEL); if (!ap.pages || !iov_page) goto out; - fuse_page_descs_length_init(ap.descs, 0, fc->max_pages); + fuse_page_descs_length_init(ap.descs, 0, fm->fc->max_pages); /* * If restricted, initialize IO parameters as encoded in @cmd. @@ -2918,7 +2920,7 @@ long fuse_do_ioctl(struct file *file, unsigned int cmd, unsigned long arg, /* make sure there are enough buffer pages and init request with them */ err = -ENOMEM; - if (max_pages > fc->max_pages) + if (max_pages > fm->fc->max_pages) goto out; while (ap.num_pages < max_pages) { ap.pages[ap.num_pages] = alloc_page(GFP_KERNEL | __GFP_HIGHMEM); @@ -2955,7 +2957,7 @@ long fuse_do_ioctl(struct file *file, unsigned int cmd, unsigned long arg, ap.args.out_pages = true; ap.args.out_argvar = true; - transferred = fuse_simple_request(fc, &ap.args); + transferred = fuse_simple_request(fm, &ap.args); err = transferred; if (transferred < 0) goto out; @@ -2983,7 +2985,7 @@ long fuse_do_ioctl(struct file *file, unsigned int cmd, unsigned long arg, goto out; vaddr = kmap_atomic(ap.pages[0]); - err = fuse_copy_ioctl_iovec(fc, iov_page, vaddr, + err = fuse_copy_ioctl_iovec(fm->fc, iov_page, vaddr, transferred, in_iovs + out_iovs, (flags & FUSE_IOCTL_COMPAT) != 0); kunmap_atomic(vaddr); @@ -2993,11 +2995,11 @@ long fuse_do_ioctl(struct file *file, unsigned int cmd, unsigned long arg, in_iov = iov_page; out_iov = in_iov + in_iovs; - err = fuse_verify_ioctl_iov(fc, in_iov, in_iovs); + err = fuse_verify_ioctl_iov(fm->fc, in_iov, in_iovs); if (err) goto out; - err = fuse_verify_ioctl_iov(fc, out_iov, out_iovs); + err = fuse_verify_ioctl_iov(fm->fc, out_iov, out_iovs); if (err) goto out; @@ -3107,13 +3109,13 @@ static void fuse_register_polled_file(struct fuse_conn *fc, __poll_t fuse_file_poll(struct file *file, poll_table *wait) { struct fuse_file *ff = file->private_data; - struct fuse_conn *fc = ff->fc; + struct fuse_mount *fm = ff->fm; struct fuse_poll_in inarg = { .fh = ff->fh, .kh = ff->kh }; struct fuse_poll_out outarg; FUSE_ARGS(args); int err; - if (fc->no_poll) + if (fm->fc->no_poll) return DEFAULT_POLLMASK; poll_wait(file, &ff->poll_wait, wait); @@ -3125,7 +3127,7 @@ __poll_t fuse_file_poll(struct file *file, poll_table *wait) */ if (waitqueue_active(&ff->poll_wait)) { inarg.flags |= FUSE_POLL_SCHEDULE_NOTIFY; - fuse_register_polled_file(fc, ff); + fuse_register_polled_file(fm->fc, ff); } args.opcode = FUSE_POLL; @@ -3136,12 +3138,12 @@ __poll_t fuse_file_poll(struct file *file, poll_table *wait) args.out_numargs = 1; args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (!err) return demangle_poll(outarg.revents); if (err == -ENOSYS) { - fc->no_poll = 1; + fm->fc->no_poll = 1; return DEFAULT_POLLMASK; } return EPOLLERR; @@ -3227,13 +3229,13 @@ fuse_direct_IO(struct kiocb *iocb, struct iov_iter *iter) * By default, we want to optimize all I/Os with async request * submission to the client filesystem if supported. */ - io->async = ff->fc->async_dio; + io->async = ff->fm->fc->async_dio; io->iocb = iocb; io->blocking = is_sync_kiocb(iocb); /* optimization for short read */ if (io->async && !io->write && offset + count > i_size) { - iov_iter_truncate(iter, fuse_round_up(ff->fc, i_size - offset)); + iov_iter_truncate(iter, fuse_round_up(ff->fm->fc, i_size - offset)); shortened = count - iov_iter_count(iter); count -= shortened; } @@ -3303,7 +3305,7 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, struct fuse_file *ff = file->private_data; struct inode *inode = file_inode(file); struct fuse_inode *fi = get_fuse_inode(inode); - struct fuse_conn *fc = ff->fc; + struct fuse_mount *fm = ff->fm; FUSE_ARGS(args); struct fuse_fallocate_in inarg = { .fh = ff->fh, @@ -3319,7 +3321,7 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, if (mode & ~(FALLOC_FL_KEEP_SIZE | FALLOC_FL_PUNCH_HOLE)) return -EOPNOTSUPP; - if (fc->no_fallocate) + if (fm->fc->no_fallocate) return -EOPNOTSUPP; inode_lock(inode); @@ -3357,9 +3359,9 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, args.in_numargs = 1; args.in_args[0].size = sizeof(inarg); args.in_args[0].value = &inarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err == -ENOSYS) { - fc->no_fallocate = 1; + fm->fc->no_fallocate = 1; err = -EOPNOTSUPP; } if (err) @@ -3369,7 +3371,7 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, if (!(mode & FALLOC_FL_KEEP_SIZE)) { bool changed = fuse_write_update_size(inode, offset + length); - if (changed && fc->writeback_cache) + if (changed && fm->fc->writeback_cache) file_update_time(file); } @@ -3399,7 +3401,8 @@ static ssize_t __fuse_copy_file_range(struct file *file_in, loff_t pos_in, struct inode *inode_in = file_inode(file_in); struct inode *inode_out = file_inode(file_out); struct fuse_inode *fi_out = get_fuse_inode(inode_out); - struct fuse_conn *fc = ff_in->fc; + struct fuse_mount *fm = ff_in->fm; + struct fuse_conn *fc = fm->fc; FUSE_ARGS(args); struct fuse_copy_file_range_in inarg = { .fh_in = ff_in->fh, @@ -3468,7 +3471,7 @@ static ssize_t __fuse_copy_file_range(struct file *file_in, loff_t pos_in, args.out_numargs = 1; args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err == -ENOSYS) { fc->no_copy_file_range = 1; err = -EOPNOTSUPP; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 31f0052c4b99..fa696ea015aa 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -177,6 +177,7 @@ enum { }; struct fuse_conn; +struct fuse_mount; struct fuse_release_args; /** @@ -193,7 +194,7 @@ struct fuse_passthrough { /** FUSE specific file data */ struct fuse_file { /** Fuse connection for this file */ - struct fuse_conn *fc; + struct fuse_mount *fm; /* Argument space reserved for release */ struct fuse_release_args *release_args; @@ -283,7 +284,7 @@ struct fuse_args { bool may_block:1; struct fuse_in_arg in_args[3]; struct fuse_arg out_args[2]; - void (*end)(struct fuse_conn *fc, struct fuse_args *args, int error); + void (*end)(struct fuse_mount *fm, struct fuse_args *args, int error); /* Path used for completing d_canonical_path */ struct path *canonical_path; @@ -393,8 +394,8 @@ struct fuse_req { /** virtio-fs's physically contiguous buffer for in and out args */ void *argbuf; - /** fuse_conn this request belongs to */ - struct fuse_conn *fc; + /** fuse_mount this request belongs to */ + struct fuse_mount *fm; }; struct fuse_iqueue; @@ -533,9 +534,9 @@ struct fuse_fs_context { /** * A Fuse connection. * - * This structure is created, when the filesystem is mounted, and is - * destroyed, when the client device is closed and the filesystem is - * unmounted. + * This structure is created, when the root filesystem is mounted, and + * is destroyed, when the client device is closed and the last + * fuse_mount is destroyed. */ struct fuse_conn { /** Lock protecting accessess to members of this structure */ @@ -768,10 +769,10 @@ struct fuse_conn { /** Negotiated minor version */ unsigned minor; - /** Entry on the fuse_conn_list */ + /** Entry on the fuse_mount_list */ struct list_head entry; - /** Device ID from super block */ + /** Device ID from the root super block */ dev_t dev; /** Dentries in the control filesystem */ @@ -789,10 +790,10 @@ struct fuse_conn { /** Called on final put */ void (*release)(struct fuse_conn *); - /** Super block for this connection. */ - struct super_block *sb; - - /** Read/write semaphore to hold when accessing sb. */ + /** + * Read/write semaphore to hold when accessing the sb of any + * fuse_mount belonging to this connection + */ struct rw_semaphore killsb; /** List of device instances belonging to this connection */ @@ -803,6 +804,9 @@ struct fuse_conn { struct fuse_conn_dax *dax; #endif + /** List of filesystems using this connection */ + struct list_head mounts; + /** IDR for passthrough requests */ struct idr passthrough_req; @@ -810,14 +814,52 @@ struct fuse_conn { spinlock_t passthrough_req_lock; }; -static inline struct fuse_conn *get_fuse_conn_super(struct super_block *sb) +/* + * Represents a mounted filesystem, potentially a submount. + * + * This object allows sharing a fuse_conn between separate mounts to + * allow submounts with dedicated superblocks and thus separate device + * IDs. + */ +struct fuse_mount { + /* Underlying (potentially shared) connection to the FUSE server */ + struct fuse_conn *fc; + + /* Refcount */ + refcount_t count; + + /* + * Super block for this connection (fc->killsb must be held when + * accessing this). + */ + struct super_block *sb; + + /* Entry on fc->mounts */ + struct list_head fc_entry; +}; + +static inline struct fuse_mount *get_fuse_mount_super(struct super_block *sb) { return sb->s_fs_info; } +static inline struct fuse_conn *get_fuse_conn_super(struct super_block *sb) +{ + struct fuse_mount *fm = get_fuse_mount_super(sb); + + return fm ? fm->fc : NULL; +} + +static inline struct fuse_mount *get_fuse_mount(struct inode *inode) +{ + return get_fuse_mount_super(inode->i_sb); +} + static inline struct fuse_conn *get_fuse_conn(struct inode *inode) { - return get_fuse_conn_super(inode->i_sb); + struct fuse_mount *fm = get_fuse_mount(inode); + + return fm ? fm->fc : NULL; } static inline struct fuse_inode *get_fuse_inode(struct inode *inode) @@ -856,11 +898,6 @@ extern const struct file_operations fuse_dev_operations; extern const struct dentry_operations fuse_dentry_operations; extern const struct dentry_operations fuse_root_dentry_operations; -/** - * Inode to nodeid comparison. - */ -int fuse_inode_eq(struct inode *inode, void *_nodeidp); - /** * Get a filled in inode */ @@ -912,7 +949,7 @@ void fuse_read_args_fill(struct fuse_io_args *ia, struct file *file, loff_t pos, */ int fuse_open_common(struct inode *inode, struct file *file, bool isdir); -struct fuse_file *fuse_file_alloc(struct fuse_conn *fc); +struct fuse_file *fuse_file_alloc(struct fuse_mount *fm); void fuse_file_free(struct fuse_file *ff); void fuse_finish_open(struct inode *inode, struct file *file); @@ -980,8 +1017,8 @@ void __exit fuse_ctl_cleanup(void); /** * Simple request sending that does request allocation and freeing */ -ssize_t fuse_simple_request(struct fuse_conn *fc, struct fuse_args *args); -int fuse_simple_background(struct fuse_conn *fc, struct fuse_args *args, +ssize_t fuse_simple_request(struct fuse_mount *fm, struct fuse_args *args); +int fuse_simple_background(struct fuse_mount *fm, struct fuse_args *args, gfp_t gfp_flags); /** @@ -1013,7 +1050,8 @@ struct fuse_conn *fuse_conn_get(struct fuse_conn *fc); /** * Initialize fuse_conn */ -void fuse_conn_init(struct fuse_conn *fc, struct user_namespace *user_ns, +void fuse_conn_init(struct fuse_conn *fc, struct fuse_mount *fm, + struct user_namespace *user_ns, const struct fuse_iqueue_ops *fiq_ops, void *fiq_priv); /** @@ -1021,11 +1059,21 @@ void fuse_conn_init(struct fuse_conn *fc, struct user_namespace *user_ns, */ void fuse_conn_put(struct fuse_conn *fc); +/** + * Acquire reference to fuse_mount + */ +struct fuse_mount *fuse_mount_get(struct fuse_mount *fm); + +/** + * Release reference to fuse_mount + */ +void fuse_mount_put(struct fuse_mount *fm); + struct fuse_dev *fuse_dev_alloc_install(struct fuse_conn *fc); struct fuse_dev *fuse_dev_alloc(void); void fuse_dev_install(struct fuse_dev *fud, struct fuse_conn *fc); void fuse_dev_free(struct fuse_dev *fud); -void fuse_send_init(struct fuse_conn *fc); +void fuse_send_init(struct fuse_mount *fm); /** * Fill in superblock and initialize fuse connection @@ -1034,12 +1082,18 @@ void fuse_send_init(struct fuse_conn *fc); */ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx); -/** - * Disassociate fuse connection from superblock and kill the superblock + +/* + * Remove the mount from the connection * - * Calls kill_anon_super(), do not use with bdev mounts. + * Returns whether this was the last mount */ -void fuse_kill_sb_anon(struct super_block *sb); +bool fuse_mount_remove(struct fuse_mount *fm); + +/* + * Shut down the connection (possibly sending DESTROY request). + */ +void fuse_conn_destroy(struct fuse_mount *fm); /** * Add connection to control filesystem @@ -1074,10 +1128,20 @@ void fuse_flush_writepages(struct inode *inode); void fuse_set_nowrite(struct inode *inode); void fuse_release_nowrite(struct inode *inode); +/** + * Scan all fuse_mounts belonging to fc to find the first where + * ilookup5() returns a result. Return that result and the + * respective fuse_mount in *fm (unless fm is NULL). + * + * The caller must hold fc->killsb. + */ +struct inode *fuse_ilookup(struct fuse_conn *fc, u64 nodeid, + struct fuse_mount **fm); + /** * File-system tells the kernel to invalidate cache for the given node id. */ -int fuse_reverse_inval_inode(struct super_block *sb, u64 nodeid, +int fuse_reverse_inval_inode(struct fuse_conn *fc, u64 nodeid, loff_t offset, loff_t len); /** @@ -1090,10 +1154,10 @@ int fuse_reverse_inval_inode(struct super_block *sb, u64 nodeid, * - is a file or oan empty directory * then the dentry is unhashed (d_delete()). */ -int fuse_reverse_inval_entry(struct super_block *sb, u64 parent_nodeid, +int fuse_reverse_inval_entry(struct fuse_conn *fc, u64 parent_nodeid, u64 child_nodeid, struct qstr *name); -int fuse_do_open(struct fuse_conn *fc, u64 nodeid, struct file *file, +int fuse_do_open(struct fuse_mount *fm, u64 nodeid, struct file *file, bool isdir); /** diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 708e6e757dfe..6537cee436f2 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -288,7 +288,7 @@ static void fuse_init_inode(struct inode *inode, struct fuse_attr *attr) BUG(); } -int fuse_inode_eq(struct inode *inode, void *_nodeidp) +static int fuse_inode_eq(struct inode *inode, void *_nodeidp) { u64 nodeid = *(u64 *) _nodeidp; if (get_node_id(inode) == nodeid) @@ -343,16 +343,37 @@ struct inode *fuse_iget(struct super_block *sb, u64 nodeid, return inode; } -int fuse_reverse_inval_inode(struct super_block *sb, u64 nodeid, +struct inode *fuse_ilookup(struct fuse_conn *fc, u64 nodeid, + struct fuse_mount **fm) +{ + struct fuse_mount *fm_iter; + struct inode *inode; + + WARN_ON(!rwsem_is_locked(&fc->killsb)); + list_for_each_entry(fm_iter, &fc->mounts, fc_entry) { + if (!fm_iter->sb) + continue; + + inode = ilookup5(fm_iter->sb, nodeid, fuse_inode_eq, &nodeid); + if (inode) { + if (fm) + *fm = fm_iter; + return inode; + } + } + + return NULL; +} + +int fuse_reverse_inval_inode(struct fuse_conn *fc, u64 nodeid, loff_t offset, loff_t len) { - struct fuse_conn *fc = get_fuse_conn_super(sb); struct fuse_inode *fi; struct inode *inode; pgoff_t pg_start; pgoff_t pg_end; - inode = ilookup5(sb, nodeid, fuse_inode_eq, &nodeid); + inode = fuse_ilookup(fc, nodeid, NULL); if (!inode) return -ENOENT; @@ -402,28 +423,23 @@ static void fuse_umount_begin(struct super_block *sb) fuse_abort_conn(fc); } -static void fuse_send_destroy(struct fuse_conn *fc) +static void fuse_send_destroy(struct fuse_mount *fm) { - if (fc->conn_init) { + if (fm->fc->conn_init) { FUSE_ARGS(args); args.opcode = FUSE_DESTROY; args.force = true; args.nocreds = true; - fuse_simple_request(fc, &args); + fuse_simple_request(fm, &args); } } static void fuse_put_super(struct super_block *sb) { - struct fuse_conn *fc = get_fuse_conn_super(sb); + struct fuse_mount *fm = get_fuse_mount_super(sb); - mutex_lock(&fuse_mutex); - list_del(&fc->entry); - fuse_ctl_remove_conn(fc); - mutex_unlock(&fuse_mutex); - - fuse_conn_put(fc); + fuse_mount_put(fm); } static void convert_fuse_statfs(struct kstatfs *stbuf, struct fuse_kstatfs *attr) @@ -443,12 +459,12 @@ static void convert_fuse_statfs(struct kstatfs *stbuf, struct fuse_kstatfs *attr static int fuse_statfs(struct dentry *dentry, struct kstatfs *buf) { struct super_block *sb = dentry->d_sb; - struct fuse_conn *fc = get_fuse_conn_super(sb); + struct fuse_mount *fm = get_fuse_mount_super(sb); FUSE_ARGS(args); struct fuse_statfs_out outarg; int err; - if (!fuse_allow_current_process(fc)) { + if (!fuse_allow_current_process(fm->fc)) { buf->f_type = FUSE_SUPER_MAGIC; return 0; } @@ -460,7 +476,7 @@ static int fuse_statfs(struct dentry *dentry, struct kstatfs *buf) args.out_numargs = 1; args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (!err) convert_fuse_statfs(buf, &outarg.st); return err; @@ -649,7 +665,8 @@ static void fuse_pqueue_init(struct fuse_pqueue *fpq) fpq->connected = 1; } -void fuse_conn_init(struct fuse_conn *fc, struct user_namespace *user_ns, +void fuse_conn_init(struct fuse_conn *fc, struct fuse_mount *fm, + struct user_namespace *user_ns, const struct fuse_iqueue_ops *fiq_ops, void *fiq_priv) { memset(fc, 0, sizeof(*fc)); @@ -678,6 +695,11 @@ void fuse_conn_init(struct fuse_conn *fc, struct user_namespace *user_ns, fc->pid_ns = get_pid_ns(task_active_pid_ns(current)); fc->user_ns = get_user_ns(user_ns); fc->max_pages = FUSE_DEFAULT_MAX_PAGES_PER_REQ; + + INIT_LIST_HEAD(&fc->mounts); + list_add(&fm->fc_entry, &fc->mounts); + fm->fc = fc; + refcount_set(&fm->count, 1); } EXPORT_SYMBOL_GPL(fuse_conn_init); @@ -704,6 +726,23 @@ struct fuse_conn *fuse_conn_get(struct fuse_conn *fc) } EXPORT_SYMBOL_GPL(fuse_conn_get); +void fuse_mount_put(struct fuse_mount *fm) +{ + if (refcount_dec_and_test(&fm->count)) { + if (fm->fc) + fuse_conn_put(fm->fc); + kfree(fm); + } +} +EXPORT_SYMBOL_GPL(fuse_mount_put); + +struct fuse_mount *fuse_mount_get(struct fuse_mount *fm) +{ + refcount_inc(&fm->count); + return fm; +} +EXPORT_SYMBOL_GPL(fuse_mount_get); + static struct inode *fuse_get_root_inode(struct super_block *sb, unsigned mode) { struct fuse_attr attr; @@ -933,9 +972,10 @@ struct fuse_init_args { struct fuse_init_out out; }; -static void process_init_reply(struct fuse_conn *fc, struct fuse_args *args, +static void process_init_reply(struct fuse_mount *fm, struct fuse_args *args, int error) { + struct fuse_conn *fc = fm->fc; struct fuse_init_args *ia = container_of(args, typeof(*ia), args); struct fuse_init_out *arg = &ia->out; bool ok = true; @@ -989,11 +1029,11 @@ static void process_init_reply(struct fuse_conn *fc, struct fuse_args *args, if (arg->flags & FUSE_HANDLE_KILLPRIV) fc->handle_killpriv = 1; if (arg->time_gran && arg->time_gran <= 1000000000) - fc->sb->s_time_gran = arg->time_gran; + fm->sb->s_time_gran = arg->time_gran; if ((arg->flags & FUSE_POSIX_ACL)) { fc->default_permissions = 1; fc->posix_acl = 1; - fc->sb->s_xattr = fuse_acl_xattr_handlers; + fm->sb->s_xattr = fuse_acl_xattr_handlers; } if (arg->flags & FUSE_CACHE_SYMLINKS) fc->cache_symlinks = 1; @@ -1012,7 +1052,7 @@ static void process_init_reply(struct fuse_conn *fc, struct fuse_args *args, if (arg->flags & FUSE_PASSTHROUGH) { fc->passthrough = 1; /* Prevent further stacking */ - fc->sb->s_stack_depth = + fm->sb->s_stack_depth = FILESYSTEM_MAX_STACK_DEPTH; } } else { @@ -1021,8 +1061,8 @@ static void process_init_reply(struct fuse_conn *fc, struct fuse_args *args, fc->no_flock = 1; } - fc->sb->s_bdi->ra_pages = - min(fc->sb->s_bdi->ra_pages, ra_pages); + fm->sb->s_bdi->ra_pages = + min(fm->sb->s_bdi->ra_pages, ra_pages); fc->minor = arg->minor; fc->max_write = arg->minor < 5 ? 4096 : arg->max_write; fc->max_write = max_t(unsigned, 4096, fc->max_write); @@ -1039,7 +1079,7 @@ static void process_init_reply(struct fuse_conn *fc, struct fuse_args *args, wake_up_all(&fc->blocked_waitq); } -void fuse_send_init(struct fuse_conn *fc) +void fuse_send_init(struct fuse_mount *fm) { struct fuse_init_args *ia; @@ -1047,7 +1087,7 @@ void fuse_send_init(struct fuse_conn *fc) ia->in.major = FUSE_KERNEL_VERSION; ia->in.minor = FUSE_KERNEL_MINOR_VERSION; - ia->in.max_readahead = fc->sb->s_bdi->ra_pages * PAGE_SIZE; + ia->in.max_readahead = fm->sb->s_bdi->ra_pages * PAGE_SIZE; ia->in.flags |= FUSE_ASYNC_READ | FUSE_POSIX_LOCKS | FUSE_ATOMIC_O_TRUNC | FUSE_EXPORT_SUPPORT | FUSE_BIG_WRITES | FUSE_DONT_MASK | @@ -1060,7 +1100,7 @@ void fuse_send_init(struct fuse_conn *fc) FUSE_NO_OPENDIR_SUPPORT | FUSE_EXPLICIT_INVAL_DATA | FUSE_PASSTHROUGH; #ifdef CONFIG_FUSE_DAX - if (fc->dax) + if (fm->fc->dax) ia->in.flags |= FUSE_MAP_ALIGNMENT; #endif ia->args.opcode = FUSE_INIT; @@ -1078,8 +1118,8 @@ void fuse_send_init(struct fuse_conn *fc) ia->args.nocreds = true; ia->args.end = process_init_reply; - if (fuse_simple_background(fc, &ia->args, GFP_KERNEL) != 0) - process_init_reply(fc, &ia->args, -ENOTCONN); + if (fuse_simple_background(fm, &ia->args, GFP_KERNEL) != 0) + process_init_reply(fm, &ia->args, -ENOTCONN); } EXPORT_SYMBOL_GPL(fuse_send_init); @@ -1205,7 +1245,8 @@ EXPORT_SYMBOL_GPL(fuse_dev_free); int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) { struct fuse_dev *fud = NULL; - struct fuse_conn *fc = get_fuse_conn_super(sb); + struct fuse_mount *fm = get_fuse_mount_super(sb); + struct fuse_conn *fc = fm->fc; struct inode *root; struct dentry *root_dentry; int err; @@ -1260,7 +1301,7 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) } fc->dev = sb->s_dev; - fc->sb = sb; + fm->sb = sb; err = fuse_bdi_init(fc, sb); if (err) goto err_dev_free; @@ -1325,6 +1366,7 @@ static int fuse_fill_super(struct super_block *sb, struct fs_context *fsc) struct file *file; int err; struct fuse_conn *fc; + struct fuse_mount *fm; err = -EINVAL; file = fget(ctx->fd); @@ -1345,9 +1387,16 @@ static int fuse_fill_super(struct super_block *sb, struct fs_context *fsc) if (!fc) goto err_fput; - fuse_conn_init(fc, sb->s_user_ns, &fuse_dev_fiq_ops, NULL); + fm = kzalloc(sizeof(*fm), GFP_KERNEL); + if (!fm) { + kfree(fc); + goto err_fput; + } + + fuse_conn_init(fc, fm, sb->s_user_ns, &fuse_dev_fiq_ops, NULL); fc->release = fuse_free_conn; - sb->s_fs_info = fc; + + sb->s_fs_info = fm; err = fuse_fill_super_common(sb, ctx); if (err) @@ -1358,11 +1407,11 @@ static int fuse_fill_super(struct super_block *sb, struct fs_context *fsc) * CPUs after this */ fput(file); - fuse_send_init(get_fuse_conn_super(sb)); + fuse_send_init(get_fuse_mount_super(sb)); return 0; err_put_conn: - fuse_conn_put(fc); + fuse_mount_put(fm); sb->s_fs_info = NULL; err_fput: fput(file); @@ -1420,29 +1469,45 @@ static int fuse_init_fs_context(struct fs_context *fc) return 0; } -static void fuse_sb_destroy(struct super_block *sb) +bool fuse_mount_remove(struct fuse_mount *fm) { - struct fuse_conn *fc = get_fuse_conn_super(sb); + struct fuse_conn *fc = fm->fc; + bool last = false; - if (fc) { - if (fc->destroy) - fuse_send_destroy(fc); + down_write(&fc->killsb); + list_del_init(&fm->fc_entry); + if (list_empty(&fc->mounts)) + last = true; + up_write(&fc->killsb); - fuse_abort_conn(fc); - fuse_wait_aborted(fc); - - down_write(&fc->killsb); - fc->sb = NULL; - up_write(&fc->killsb); - } + return last; } +EXPORT_SYMBOL_GPL(fuse_mount_remove); -void fuse_kill_sb_anon(struct super_block *sb) +void fuse_conn_destroy(struct fuse_mount *fm) { - fuse_sb_destroy(sb); + struct fuse_conn *fc = fm->fc; + + if (fc->destroy) + fuse_send_destroy(fm); + + fuse_abort_conn(fc); + fuse_wait_aborted(fc); +} +EXPORT_SYMBOL_GPL(fuse_conn_destroy); + +static void fuse_kill_sb_anon(struct super_block *sb) +{ + struct fuse_mount *fm = get_fuse_mount_super(sb); + bool last; + + if (fm) { + last = fuse_mount_remove(fm); + if (last) + fuse_conn_destroy(fm); + } kill_anon_super(sb); } -EXPORT_SYMBOL_GPL(fuse_kill_sb_anon); static struct file_system_type fuse_fs_type = { .owner = THIS_MODULE, @@ -1457,7 +1522,14 @@ MODULE_ALIAS_FS("fuse"); #ifdef CONFIG_BLOCK static void fuse_kill_sb_blk(struct super_block *sb) { - fuse_sb_destroy(sb); + struct fuse_mount *fm = get_fuse_mount_super(sb); + bool last; + + if (fm) { + last = fuse_mount_remove(fm); + if (last) + fuse_conn_destroy(fm); + } kill_block_super(sb); } diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index c5203b5a0057..b0f0585dbdba 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -260,7 +260,7 @@ retry: static void fuse_force_forget(struct file *file, u64 nodeid) { struct inode *inode = file_inode(file); - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); struct fuse_forget_in inarg; FUSE_ARGS(args); @@ -274,7 +274,7 @@ static void fuse_force_forget(struct file *file, u64 nodeid) args.force = true; args.noreply = true; - fuse_simple_request(fc, &args); + fuse_simple_request(fm, &args); /* ignore errors */ } @@ -328,7 +328,7 @@ static int fuse_readdir_uncached(struct file *file, struct dir_context *ctx) ssize_t res; struct page *page; struct inode *inode = file_inode(file); - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); struct fuse_io_args ia = {}; struct fuse_args_pages *ap = &ia.ap; struct fuse_page_desc desc = { .length = PAGE_SIZE }; @@ -345,7 +345,7 @@ static int fuse_readdir_uncached(struct file *file, struct dir_context *ctx) ap->pages = &page; ap->descs = &desc; if (plus) { - attr_version = fuse_get_attr_version(fc); + attr_version = fuse_get_attr_version(fm->fc); fuse_read_args_fill(&ia, file, ctx->pos, PAGE_SIZE, FUSE_READDIRPLUS); } else { @@ -353,7 +353,7 @@ static int fuse_readdir_uncached(struct file *file, struct dir_context *ctx) FUSE_READDIR); } locked = fuse_lock_inode(inode); - res = fuse_simple_request(fc, &ap->args); + res = fuse_simple_request(fm, &ap->args); fuse_unlock_inode(inode, locked); if (res >= 0) { if (!res) { diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 147467a0aab5..d3b90052a45b 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1285,7 +1285,8 @@ static inline void virtio_fs_ctx_set_defaults(struct fuse_fs_context *ctx) static int virtio_fs_fill_super(struct super_block *sb, struct fs_context *fsc) { - struct fuse_conn *fc = get_fuse_conn_super(sb); + struct fuse_mount *fm = get_fuse_mount_super(sb); + struct fuse_conn *fc = fm->fc; struct virtio_fs *fs = fc->iq.priv; struct fuse_fs_context *ctx = fsc->fs_private; unsigned int i; @@ -1330,7 +1331,7 @@ static int virtio_fs_fill_super(struct super_block *sb, struct fs_context *fsc) /* Previous unmount will stop all queues. Start these again */ virtio_fs_start_all_queues(fs); - fuse_send_init(fc); + fuse_send_init(fm); mutex_unlock(&virtio_fs_mutex); return 0; @@ -1341,21 +1342,14 @@ err: return err; } -static void virtio_kill_sb(struct super_block *sb) +static void virtio_fs_conn_destroy(struct fuse_mount *fm) { - struct fuse_conn *fc = get_fuse_conn_super(sb); - struct virtio_fs *vfs; - struct virtio_fs_vq *fsvq; + struct fuse_conn *fc = fm->fc; + struct virtio_fs *vfs = fc->iq.priv; + struct virtio_fs_vq *fsvq = &vfs->vqs[VQ_HIPRIO]; - /* If mount failed, we can still be called without any fc */ - if (!fc) - return fuse_kill_sb_anon(sb); - - vfs = fc->iq.priv; - fsvq = &vfs->vqs[VQ_HIPRIO]; - - /* Stop dax worker. Soon evict_inodes() will be called which will - * free all memory ranges belonging to all inodes. + /* Stop dax worker. Soon evict_inodes() will be called which + * will free all memory ranges belonging to all inodes. */ if (IS_ENABLED(CONFIG_FUSE_DAX)) fuse_dax_cancel_work(fc); @@ -1366,9 +1360,9 @@ static void virtio_kill_sb(struct super_block *sb) spin_unlock(&fsvq->lock); virtio_fs_drain_all_queues(vfs); - fuse_kill_sb_anon(sb); + fuse_conn_destroy(fm); - /* fuse_kill_sb_anon() must have sent destroy. Stop all queues + /* fuse_conn_destroy() must have sent destroy. Stop all queues * and drain one more time and free fuse devices. Freeing fuse * devices will drop their reference on fuse_conn and that in * turn will drop its reference on virtio_fs object. @@ -1378,12 +1372,27 @@ static void virtio_kill_sb(struct super_block *sb) virtio_fs_free_devs(vfs); } +static void virtio_kill_sb(struct super_block *sb) +{ + struct fuse_mount *fm = get_fuse_mount_super(sb); + bool last; + + /* If mount failed, we can still be called without any fc */ + if (fm) { + last = fuse_mount_remove(fm); + if (last) + virtio_fs_conn_destroy(fm); + } + kill_anon_super(sb); +} + static int virtio_fs_test_super(struct super_block *sb, struct fs_context *fsc) { - struct fuse_conn *fc = fsc->s_fs_info; + struct fuse_mount *fsc_fm = fsc->s_fs_info; + struct fuse_mount *sb_fm = get_fuse_mount_super(sb); - return fc->iq.priv == get_fuse_conn_super(sb)->iq.priv; + return fsc_fm->fc->iq.priv == sb_fm->fc->iq.priv; } static int virtio_fs_set_super(struct super_block *sb, @@ -1393,7 +1402,7 @@ static int virtio_fs_set_super(struct super_block *sb, err = get_anon_bdev(&sb->s_dev); if (!err) - fuse_conn_get(fsc->s_fs_info); + fuse_mount_get(fsc->s_fs_info); return err; } @@ -1403,6 +1412,7 @@ static int virtio_fs_get_tree(struct fs_context *fsc) struct virtio_fs *fs; struct super_block *sb; struct fuse_conn *fc; + struct fuse_mount *fm; int err; if (!fsc->source) @@ -1426,14 +1436,23 @@ static int virtio_fs_get_tree(struct fs_context *fsc) return -ENOMEM; } - fuse_conn_init(fc, get_user_ns(current_user_ns()), &virtio_fs_fiq_ops, - fs); + fm = kzalloc(sizeof(struct fuse_mount), GFP_KERNEL); + if (!fm) { + mutex_lock(&virtio_fs_mutex); + virtio_fs_put(fs); + mutex_unlock(&virtio_fs_mutex); + kfree(fc); + return -ENOMEM; + } + + fuse_conn_init(fc, fm, get_user_ns(current_user_ns()), + &virtio_fs_fiq_ops, fs); fc->release = fuse_free_conn; fc->delete_stale = true; - fsc->s_fs_info = fc; + fsc->s_fs_info = fm; sb = sget_fc(fsc, virtio_fs_test_super, virtio_fs_set_super); - fuse_conn_put(fc); + fuse_mount_put(fm); if (IS_ERR(sb)) return PTR_ERR(sb); diff --git a/fs/fuse/xattr.c b/fs/fuse/xattr.c index 9d3f28b1f6f8..b17cb7224144 100644 --- a/fs/fuse/xattr.c +++ b/fs/fuse/xattr.c @@ -14,12 +14,12 @@ int fuse_setxattr(struct inode *inode, const char *name, const void *value, size_t size, int flags) { - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); struct fuse_setxattr_in inarg; int err; - if (fc->no_setxattr) + if (fm->fc->no_setxattr) return -EOPNOTSUPP; memset(&inarg, 0, sizeof(inarg)); @@ -34,9 +34,9 @@ int fuse_setxattr(struct inode *inode, const char *name, const void *value, args.in_args[1].value = name; args.in_args[2].size = size; args.in_args[2].value = value; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err == -ENOSYS) { - fc->no_setxattr = 1; + fm->fc->no_setxattr = 1; err = -EOPNOTSUPP; } if (!err) { @@ -49,13 +49,13 @@ int fuse_setxattr(struct inode *inode, const char *name, const void *value, ssize_t fuse_getxattr(struct inode *inode, const char *name, void *value, size_t size) { - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); struct fuse_getxattr_in inarg; struct fuse_getxattr_out outarg; ssize_t ret; - if (fc->no_getxattr) + if (fm->fc->no_getxattr) return -EOPNOTSUPP; memset(&inarg, 0, sizeof(inarg)); @@ -77,11 +77,11 @@ ssize_t fuse_getxattr(struct inode *inode, const char *name, void *value, args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; } - ret = fuse_simple_request(fc, &args); + ret = fuse_simple_request(fm, &args); if (!ret && !size) ret = min_t(size_t, outarg.size, XATTR_SIZE_MAX); if (ret == -ENOSYS) { - fc->no_getxattr = 1; + fm->fc->no_getxattr = 1; ret = -EOPNOTSUPP; } return ret; @@ -107,7 +107,7 @@ static int fuse_verify_xattr_list(char *list, size_t size) ssize_t fuse_listxattr(struct dentry *entry, char *list, size_t size) { struct inode *inode = d_inode(entry); - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); struct fuse_getxattr_in inarg; struct fuse_getxattr_out outarg; @@ -116,10 +116,10 @@ ssize_t fuse_listxattr(struct dentry *entry, char *list, size_t size) if (fuse_is_bad(inode)) return -EIO; - if (!fuse_allow_current_process(fc)) + if (!fuse_allow_current_process(fm->fc)) return -EACCES; - if (fc->no_listxattr) + if (fm->fc->no_listxattr) return -EOPNOTSUPP; memset(&inarg, 0, sizeof(inarg)); @@ -139,13 +139,13 @@ ssize_t fuse_listxattr(struct dentry *entry, char *list, size_t size) args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; } - ret = fuse_simple_request(fc, &args); + ret = fuse_simple_request(fm, &args); if (!ret && !size) ret = min_t(size_t, outarg.size, XATTR_LIST_MAX); if (ret > 0 && size) ret = fuse_verify_xattr_list(list, ret); if (ret == -ENOSYS) { - fc->no_listxattr = 1; + fm->fc->no_listxattr = 1; ret = -EOPNOTSUPP; } return ret; @@ -153,11 +153,11 @@ ssize_t fuse_listxattr(struct dentry *entry, char *list, size_t size) int fuse_removexattr(struct inode *inode, const char *name) { - struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); int err; - if (fc->no_removexattr) + if (fm->fc->no_removexattr) return -EOPNOTSUPP; args.opcode = FUSE_REMOVEXATTR; @@ -165,9 +165,9 @@ int fuse_removexattr(struct inode *inode, const char *name) args.in_numargs = 1; args.in_args[0].size = strlen(name) + 1; args.in_args[0].value = name; - err = fuse_simple_request(fc, &args); + err = fuse_simple_request(fm, &args); if (err == -ENOSYS) { - fc->no_removexattr = 1; + fm->fc->no_removexattr = 1; err = -EOPNOTSUPP; } if (!err) { From 0774f61134d76c86692ef6ae9098301175351592 Mon Sep 17 00:00:00 2001 From: Max Reitz Date: Wed, 9 Sep 2020 17:52:17 +0200 Subject: [PATCH 043/148] UPSTREAM: fuse: Allow fuse_fill_super_common() for submounts Submounts have their own superblock, which needs to be initialized. However, they do not have a fuse_fs_context associated with them, and the root node's attributes should be taken from the mountpoint's node. Extend fuse_fill_super_common() to work for submounts by making the @ctx parameter optional, and by adding a @submount_finode parameter. (There is a plain "unsigned" in an existing code block that is being indented by this commit. Extend it to "unsigned int" so checkpatch does not complain.) Change-Id: Ia20a19e84cdc80eb05d8ab8cd7e76da866363e44 Signed-off-by: Max Reitz Signed-off-by: Miklos Szeredi --- fs/fuse/fuse_i.h | 8 ++++ fs/fuse/inode.c | 109 ++++++++++++++++++++++++++++++++++++++--------- 2 files changed, 96 insertions(+), 21 deletions(-) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index fa696ea015aa..422c505e6a03 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1082,6 +1082,14 @@ void fuse_send_init(struct fuse_mount *fm); */ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx); +/* + * Fill in superblock for submounts + * @sb: partially-initialized superblock to fill in + * @parent_fi: The fuse_inode of the parent filesystem where this submount is + * mounted + */ +int fuse_fill_super_submount(struct super_block *sb, + struct fuse_inode *parent_fi); /* * Remove the mount from the connection diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 6537cee436f2..92786c0ae989 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -126,8 +126,11 @@ static void fuse_evict_inode(struct inode *inode) if (FUSE_IS_DAX(inode)) fuse_dax_inode_cleanup(inode); - fuse_queue_forget(fc, fi->forget, fi->nodeid, fi->nlookup); - fi->forget = NULL; + if (fi->nlookup) { + fuse_queue_forget(fc, fi->forget, fi->nodeid, + fi->nlookup); + fi->forget = NULL; + } } if (S_ISREG(inode->i_mode) && !fuse_is_bad(inode)) { WARN_ON(!list_empty(&fi->write_files)); @@ -1242,6 +1245,87 @@ void fuse_dev_free(struct fuse_dev *fud) } EXPORT_SYMBOL_GPL(fuse_dev_free); +static void fuse_fill_attr_from_inode(struct fuse_attr *attr, + const struct fuse_inode *fi) +{ + *attr = (struct fuse_attr){ + .ino = fi->inode.i_ino, + .size = fi->inode.i_size, + .blocks = fi->inode.i_blocks, + .atime = fi->inode.i_atime.tv_sec, + .mtime = fi->inode.i_mtime.tv_sec, + .ctime = fi->inode.i_ctime.tv_sec, + .atimensec = fi->inode.i_atime.tv_nsec, + .mtimensec = fi->inode.i_mtime.tv_nsec, + .ctimensec = fi->inode.i_ctime.tv_nsec, + .mode = fi->inode.i_mode, + .nlink = fi->inode.i_nlink, + .uid = fi->inode.i_uid.val, + .gid = fi->inode.i_gid.val, + .rdev = fi->inode.i_rdev, + .blksize = 1u << fi->inode.i_blkbits, + }; +} + +static void fuse_sb_defaults(struct super_block *sb) +{ + sb->s_magic = FUSE_SUPER_MAGIC; + sb->s_op = &fuse_super_operations; + sb->s_xattr = fuse_xattr_handlers; + sb->s_maxbytes = MAX_LFS_FILESIZE; + sb->s_time_gran = 1; + sb->s_export_op = &fuse_export_operations; + sb->s_iflags |= SB_I_IMA_UNVERIFIABLE_SIGNATURE; + if (sb->s_user_ns != &init_user_ns) + sb->s_iflags |= SB_I_UNTRUSTED_MOUNTER; + sb->s_flags &= ~(SB_NOSEC | SB_I_VERSION); + + /* + * If we are not in the initial user namespace posix + * acls must be translated. + */ + if (sb->s_user_ns != &init_user_ns) + sb->s_xattr = fuse_no_acl_xattr_handlers; +} + +int fuse_fill_super_submount(struct super_block *sb, + struct fuse_inode *parent_fi) +{ + struct fuse_mount *fm = get_fuse_mount_super(sb); + struct super_block *parent_sb = parent_fi->inode.i_sb; + struct fuse_attr root_attr; + struct inode *root; + + fuse_sb_defaults(sb); + fm->sb = sb; + + WARN_ON(sb->s_bdi != &noop_backing_dev_info); + sb->s_bdi = bdi_get(parent_sb->s_bdi); + + sb->s_xattr = parent_sb->s_xattr; + sb->s_time_gran = parent_sb->s_time_gran; + sb->s_blocksize = parent_sb->s_blocksize; + sb->s_blocksize_bits = parent_sb->s_blocksize_bits; + sb->s_subtype = kstrdup(parent_sb->s_subtype, GFP_KERNEL); + if (parent_sb->s_subtype && !sb->s_subtype) + return -ENOMEM; + + fuse_fill_attr_from_inode(&root_attr, parent_fi); + root = fuse_iget(sb, parent_fi->nodeid, 0, &root_attr, 0, 0); + /* + * This inode is just a duplicate, so it is not looked up and + * its nlookup should not be incremented. fuse_iget() does + * that, though, so undo it here. + */ + get_fuse_inode(root)->nlookup--; + sb->s_d_op = &fuse_dentry_operations; + sb->s_root = d_make_root(root); + if (!sb->s_root) + return -ENOMEM; + + return 0; +} + int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) { struct fuse_dev *fud = NULL; @@ -1255,7 +1339,7 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) if (sb->s_flags & SB_MANDLOCK) goto err; - sb->s_flags &= ~(SB_NOSEC | SB_I_VERSION); + fuse_sb_defaults(sb); if (ctx->is_bdev) { #ifdef CONFIG_BLOCK @@ -1270,23 +1354,6 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) sb->s_subtype = ctx->subtype; ctx->subtype = NULL; - sb->s_magic = FUSE_SUPER_MAGIC; - sb->s_op = &fuse_super_operations; - sb->s_xattr = fuse_xattr_handlers; - sb->s_maxbytes = MAX_LFS_FILESIZE; - sb->s_time_gran = 1; - sb->s_export_op = &fuse_export_operations; - sb->s_iflags |= SB_I_IMA_UNVERIFIABLE_SIGNATURE; - if (sb->s_user_ns != &init_user_ns) - sb->s_iflags |= SB_I_UNTRUSTED_MOUNTER; - - /* - * If we are not in the initial user namespace posix - * acls must be translated. - */ - if (sb->s_user_ns != &init_user_ns) - sb->s_xattr = fuse_no_acl_xattr_handlers; - if (IS_ENABLED(CONFIG_FUSE_DAX)) { err = fuse_dax_conn_alloc(fc, ctx->dax_dev); if (err) @@ -1316,7 +1383,7 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) fc->user_id = ctx->user_id; fc->group_id = ctx->group_id; fc->legacy_opts_show = ctx->legacy_opts_show; - fc->max_read = max_t(unsigned, 4096, ctx->max_read); + fc->max_read = max_t(unsigned int, 4096, ctx->max_read); fc->destroy = ctx->destroy; fc->no_control = ctx->no_control; fc->no_force_umount = ctx->no_force_umount; From 699354537fbdc07daff55bd6369ab400abee3b39 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Thu, 24 Sep 2020 08:51:32 +0200 Subject: [PATCH 044/148] BACKPORT: bdi: initialize ->ra_pages and ->io_pages in bdi_init Set up a readahead size by default, as very few users have a good reason to change it. This means code, ecryptfs, and orangefs now set up the values while they were previously missing it, while ubifs, mtd and vboxsf manually set it to 0 to avoid readahead. Change-Id: I2394fea47e8685f26aee9cc52c72018cd2715741 Signed-off-by: Christoph Hellwig Reviewed-by: Jan Kara Acked-by: David Sterba [btrfs] Acked-by: Richard Weinberger [ubifs, mtd] Signed-off-by: Jens Axboe --- block/blk-core.c | 2 -- drivers/mtd/mtdcore.c | 2 ++ fs/9p/vfs_super.c | 6 ++++-- fs/afs/super.c | 1 - fs/btrfs/disk-io.c | 1 - fs/fuse/inode.c | 1 - fs/ubifs/super.c | 2 ++ mm/backing-dev.c | 2 ++ 8 files changed, 10 insertions(+), 7 deletions(-) diff --git a/block/blk-core.c b/block/blk-core.c index 74f7a2a4f295..c565130ddf75 100644 --- a/block/blk-core.c +++ b/block/blk-core.c @@ -504,8 +504,6 @@ struct request_queue *blk_alloc_queue_node(gfp_t gfp_mask, int node_id) if (!q->stats) goto fail_stats; - q->backing_dev_info->ra_pages = VM_READAHEAD_PAGES; - q->backing_dev_info->io_pages = VM_READAHEAD_PAGES; q->backing_dev_info->capabilities = BDI_CAP_CGROUP_WRITEBACK; q->backing_dev_info->name = "block"; q->node = node_id; diff --git a/drivers/mtd/mtdcore.c b/drivers/mtd/mtdcore.c index 83012d74dcd5..bccbde286233 100644 --- a/drivers/mtd/mtdcore.c +++ b/drivers/mtd/mtdcore.c @@ -1948,6 +1948,8 @@ static struct backing_dev_info * __init mtd_bdi_init(char *name) bdi = bdi_alloc(GFP_KERNEL); if (!bdi) return ERR_PTR(-ENOMEM); + bdi->ra_pages = 0; + bdi->io_pages = 0; bdi->name = name; /* diff --git a/fs/9p/vfs_super.c b/fs/9p/vfs_super.c index 46e58fdf9ba5..c226d876383d 100644 --- a/fs/9p/vfs_super.c +++ b/fs/9p/vfs_super.c @@ -80,8 +80,10 @@ v9fs_fill_super(struct super_block *sb, struct v9fs_session_info *v9ses, if (ret) return ret; - if (v9ses->cache) - sb->s_bdi->ra_pages = VM_READAHEAD_PAGES; + if (!v9ses->cache) { + sb->s_bdi->ra_pages = 0; + sb->s_bdi->io_pages = 0; + } sb->s_flags |= SB_ACTIVE | SB_DIRSYNC; if (!v9ses->cache) diff --git a/fs/afs/super.c b/fs/afs/super.c index 554119068ea4..f432713872e3 100644 --- a/fs/afs/super.c +++ b/fs/afs/super.c @@ -449,7 +449,6 @@ static int afs_fill_super(struct super_block *sb, struct afs_fs_context *ctx) ret = super_setup_bdi(sb); if (ret) return ret; - sb->s_bdi->ra_pages = VM_READAHEAD_PAGES; /* allocate the root inode and dentry */ if (as->dyn_root) { diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 660544fc0781..176ba981f055 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -3008,7 +3008,6 @@ int open_ctree(struct super_block *sb, sb->s_bdi->congested_fn = btrfs_congested_fn; sb->s_bdi->congested_data = fs_info; sb->s_bdi->capabilities |= BDI_CAP_CGROUP_WRITEBACK; - sb->s_bdi->ra_pages = VM_READAHEAD_PAGES; sb->s_bdi->ra_pages *= btrfs_super_num_devices(disk_super); sb->s_bdi->ra_pages = max(sb->s_bdi->ra_pages, SZ_4M / PAGE_SIZE); diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 92786c0ae989..83979e03d4eb 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1164,7 +1164,6 @@ static int fuse_bdi_init(struct fuse_conn *fc, struct super_block *sb) if (err) return err; - sb->s_bdi->ra_pages = VM_READAHEAD_PAGES; /* fuse does it's own writeback accounting */ sb->s_bdi->capabilities = BDI_CAP_NO_ACCT_WB | BDI_CAP_STRICTLIMIT; diff --git a/fs/ubifs/super.c b/fs/ubifs/super.c index 0b114f685be1..d67fd6a875bd 100644 --- a/fs/ubifs/super.c +++ b/fs/ubifs/super.c @@ -2189,6 +2189,8 @@ static int ubifs_fill_super(struct super_block *sb, void *data, int silent) c->vi.vol_id); if (err) goto out_close; + sb->s_bdi->ra_pages = 0; + sb->s_bdi->io_pages = 0; sb->s_fs_info = c; sb->s_magic = UBIFS_SUPER_MAGIC; diff --git a/mm/backing-dev.c b/mm/backing-dev.c index 3954ef0cc770..1dcde0685a5f 100644 --- a/mm/backing-dev.c +++ b/mm/backing-dev.c @@ -878,6 +878,8 @@ struct backing_dev_info *bdi_alloc_node(gfp_t gfp_mask, int node_id) kfree(bdi); return NULL; } + bdi->ra_pages = VM_READAHEAD_PAGES; + bdi->io_pages = VM_READAHEAD_PAGES; return bdi; } EXPORT_SYMBOL(bdi_alloc_node); From 8baaccc83c28b16bb73e8867a79de1c18d4f71b8 Mon Sep 17 00:00:00 2001 From: Max Reitz Date: Tue, 21 Apr 2020 14:47:15 +0200 Subject: [PATCH 045/148] UPSTREAM: fuse: implement crossmounts FUSE servers can indicate crossmount points by setting FUSE_ATTR_SUBMOUNT in fuse_attr.flags. The inode will then be marked as S_AUTOMOUNT, and the .d_automount implementation creates a new submount at that location, so that the submount gets a distinct st_dev value. Note that all submounts get a distinct superblock and a distinct st_dev value, so for virtio-fs, even if the same filesystem is mounted more than once on the host, none of its mount points will have the same st_dev. We need distinct superblocks because the superblock points to the root node, but the different host mounts may show different trees (e.g. due to submounts in some of them, but not in others). Right now, this behavior is only enabled when fuse_conn.auto_submounts is set, which is the case only for virtio-fs. Change-Id: I455ddae7b02f28968cf78018b771d0c12d03bacd Signed-off-by: Max Reitz Signed-off-by: Miklos Szeredi --- fs/fuse/dir.c | 78 ++++++++++++++++++++++++++++++++++++++++++++- fs/fuse/fuse_i.h | 3 ++ fs/fuse/inode.c | 26 +++++++++++++-- fs/fuse/virtio_fs.c | 1 + 4 files changed, 105 insertions(+), 3 deletions(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 488cf8f633aa..117edf89e67d 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -10,6 +10,7 @@ #include #include +#include #include #include #include @@ -237,7 +238,8 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) ret = -ENOENT; if (!ret) { fi = get_fuse_inode(inode); - if (outarg.nodeid != get_node_id(inode)) { + if (outarg.nodeid != get_node_id(inode) || + (bool) IS_AUTOMOUNT(inode) != (bool) (outarg.attr.flags & FUSE_ATTR_SUBMOUNT)) { fuse_queue_forget(fm->fc, forget, outarg.nodeid, 1); goto invalid; @@ -336,6 +338,79 @@ default_path: path_get(canonical_path); } +/* + * Create a fuse_mount object with a new superblock (with path->dentry + * as the root), and return that mount so it can be auto-mounted on + * @path. + */ +static struct vfsmount *fuse_dentry_automount(struct path *path) +{ + struct fs_context *fsc; + struct fuse_mount *parent_fm = get_fuse_mount_super(path->mnt->mnt_sb); + struct fuse_conn *fc = parent_fm->fc; + struct fuse_mount *fm; + struct vfsmount *mnt; + struct fuse_inode *mp_fi = get_fuse_inode(d_inode(path->dentry)); + struct super_block *sb; + int err; + + fsc = fs_context_for_submount(path->mnt->mnt_sb->s_type, path->dentry); + if (IS_ERR(fsc)) { + err = PTR_ERR(fsc); + goto out; + } + + err = -ENOMEM; + fm = kzalloc(sizeof(struct fuse_mount), GFP_KERNEL); + if (!fm) + goto out_put_fsc; + + refcount_set(&fm->count, 1); + fsc->s_fs_info = fm; + sb = sget_fc(fsc, NULL, set_anon_super_fc); + if (IS_ERR(sb)) { + err = PTR_ERR(sb); + fuse_mount_put(fm); + goto out_put_fsc; + } + fm->fc = fuse_conn_get(fc); + + /* Initialize superblock, making @mp_fi its root */ + err = fuse_fill_super_submount(sb, mp_fi); + if (err) + goto out_put_sb; + + sb->s_flags |= SB_ACTIVE; + fsc->root = dget(sb->s_root); + /* We are done configuring the superblock, so unlock it */ + up_write(&sb->s_umount); + + down_write(&fc->killsb); + list_add_tail(&fm->fc_entry, &fc->mounts); + up_write(&fc->killsb); + + /* Create the submount */ + mnt = vfs_create_mount(fsc); + if (IS_ERR(mnt)) { + err = PTR_ERR(mnt); + goto out_put_fsc; + } + mntget(mnt); + put_fs_context(fsc); + return mnt; + +out_put_sb: + /* + * Only jump here when fsc->root is NULL and sb is still locked + * (otherwise put_fs_context() will put the superblock) + */ + deactivate_locked_super(sb); +out_put_fsc: + put_fs_context(fsc); +out: + return ERR_PTR(err); +} + const struct dentry_operations fuse_dentry_operations = { .d_revalidate = fuse_dentry_revalidate, .d_delete = fuse_dentry_delete, @@ -343,6 +418,7 @@ const struct dentry_operations fuse_dentry_operations = { .d_init = fuse_dentry_init, .d_release = fuse_dentry_release, #endif + .d_automount = fuse_dentry_automount, .d_canonical_path = fuse_dentry_canonical_path, }; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 422c505e6a03..a51d1b91ef8b 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -760,6 +760,9 @@ struct fuse_conn { /** Do not allow MNT_FORCE umount */ unsigned int no_force_umount:1; + /* Auto-mount submounts announced by the server */ + unsigned int auto_submounts:1; + /** Passthrough mode for read/write IO */ unsigned int passthrough:1; diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 83979e03d4eb..eff17f7f376e 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -315,7 +315,26 @@ struct inode *fuse_iget(struct super_block *sb, u64 nodeid, struct fuse_inode *fi; struct fuse_conn *fc = get_fuse_conn_super(sb); - retry: + /* + * Auto mount points get their node id from the submount root, which is + * not a unique identifier within this filesystem. + * + * To avoid conflicts, do not place submount points into the inode hash + * table. + */ + if (fc->auto_submounts && (attr->flags & FUSE_ATTR_SUBMOUNT) && + S_ISDIR(attr->mode)) { + inode = new_inode(sb); + if (!inode) + return NULL; + + fuse_init_inode(inode, attr); + get_fuse_inode(inode)->nodeid = nodeid; + inode->i_flags |= S_AUTOMOUNT; + goto done; + } + +retry: inode = iget5_locked(sb, nodeid, fuse_inode_eq, fuse_inode_set, &nodeid); if (!inode) return NULL; @@ -336,7 +355,7 @@ struct inode *fuse_iget(struct super_block *sb, u64 nodeid, goto retry; } } - +done: fi = get_fuse_inode(inode); spin_lock(&fi->lock); fi->nlookup++; @@ -1106,6 +1125,9 @@ void fuse_send_init(struct fuse_mount *fm) if (fm->fc->dax) ia->in.flags |= FUSE_MAP_ALIGNMENT; #endif + if (fm->fc->auto_submounts) + ia->in.flags |= FUSE_SUBMOUNTS; + ia->args.opcode = FUSE_INIT; ia->args.in_numargs = 1; ia->args.in_args[0].size = sizeof(ia->in); diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index d3b90052a45b..071917737a2d 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1449,6 +1449,7 @@ static int virtio_fs_get_tree(struct fs_context *fsc) &virtio_fs_fiq_ops, fs); fc->release = fuse_free_conn; fc->delete_stale = true; + fc->auto_submounts = true; fsc->s_fs_info = fm; sb = sget_fc(fsc, virtio_fs_test_super, virtio_fs_set_super); From 5d30dbe4bfec15fa424c74883fef3294e32a5085 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Fri, 9 Oct 2020 12:40:11 +0200 Subject: [PATCH 046/148] UPSTREAM: fuse: connection remove fix Re-add lost removal of fc from fuse_conn_list and the control filesystem. Reported-by: kernel test robot Fixes: fcee216beb9c ("fuse: split fuse_mount off of fuse_conn") Change-Id: Icaaee7958947461b80a10d42a38ff5a3078e268a Signed-off-by: Miklos Szeredi --- fs/fuse/inode.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index eff17f7f376e..bd35be7ba89f 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1581,6 +1581,13 @@ void fuse_conn_destroy(struct fuse_mount *fm) fuse_abort_conn(fc); fuse_wait_aborted(fc); + + if (!list_empty(&fc->entry)) { + mutex_lock(&fuse_mutex); + list_del(&fc->entry); + fuse_ctl_remove_conn(fc); + mutex_unlock(&fuse_mutex); + } } EXPORT_SYMBOL_GPL(fuse_conn_destroy); From 0da478369b6eb19dca4b6810f15f36a6ac0fe615 Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Tue, 6 Oct 2020 14:53:06 -0400 Subject: [PATCH 047/148] UPSTREAM: virtiofs: calculate number of scatter-gather elements accurately virtiofs currently maps various buffers in scatter gather list and it looks at number of pages (ap->pages) and assumes that same number of pages will be used both for input and output (sg_count_fuse_req()), and calculates total number of scatterlist elements accordingly. But looks like this assumption is not valid in all the cases. For example, Cai Qian reported that trinity, triggers warning with virtiofs sometimes. A closer look revealed that if one calls ioctl(fd, 0x5a004000, buf), it will trigger following warning. WARN_ON(out_sgs + in_sgs != total_sgs) In this case, total_sgs = 8, out_sgs=4, in_sgs=3. Number of pages is 2 (ap->pages), but out_sgs are using both the pages but in_sgs are using only one page. In this case, fuse_do_ioctl() sets different size values for input and output. args->in_args[args->in_numargs - 1].size == 6656 args->out_args[args->out_numargs - 1].size == 4096 So current method of calculating how many scatter-gather list elements will be used is not accurate. Make calculations more precise by parsing size and ap->descs. Reported-by: Qian Cai Change-Id: I1ccdbce4b946a016a4a8faea234de0cb470825ac Signed-off-by: Vivek Goyal Link: https://lore.kernel.org/linux-fsdevel/5ea77e9f6cb8c2db43b09fbd4158ab2d8c066a0a.camel@redhat.com/ Reviewed-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi --- fs/fuse/virtio_fs.c | 32 +++++++++++++++++++++++++++----- 1 file changed, 27 insertions(+), 5 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 071917737a2d..ed576fdc6da0 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1033,18 +1033,37 @@ __releases(fiq->lock) spin_unlock(&fiq->lock); } +/* Count number of scatter-gather elements required */ +static unsigned int sg_count_fuse_pages(struct fuse_page_desc *page_descs, + unsigned int num_pages, + unsigned int total_len) +{ + unsigned int i; + unsigned int this_len; + + for (i = 0; i < num_pages && total_len; i++) { + this_len = min(page_descs[i].length, total_len); + total_len -= this_len; + } + + return i; +} + /* Return the number of scatter-gather list elements required */ static unsigned int sg_count_fuse_req(struct fuse_req *req) { struct fuse_args *args = req->args; struct fuse_args_pages *ap = container_of(args, typeof(*ap), args); - unsigned int total_sgs = 1 /* fuse_in_header */; + unsigned int size, total_sgs = 1 /* fuse_in_header */; if (args->in_numargs - args->in_pages) total_sgs += 1; - if (args->in_pages) - total_sgs += ap->num_pages; + if (args->in_pages) { + size = args->in_args[args->in_numargs - 1].size; + total_sgs += sg_count_fuse_pages(ap->descs, ap->num_pages, + size); + } if (!test_bit(FR_ISREPLY, &req->flags)) return total_sgs; @@ -1054,8 +1073,11 @@ static unsigned int sg_count_fuse_req(struct fuse_req *req) if (args->out_numargs - args->out_pages) total_sgs += 1; - if (args->out_pages) - total_sgs += ap->num_pages; + if (args->out_pages) { + size = args->out_args[args->out_numargs - 1].size; + total_sgs += sg_count_fuse_pages(ap->descs, ap->num_pages, + size); + } return total_sgs; } From 507c24a3c3f7a3044a4b19ce0aac3f43dfdde8cb Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Wed, 11 Nov 2020 17:22:31 +0100 Subject: [PATCH 048/148] UPSTREAM: virtiofs fix leak in setup [ Upstream commit 66ab33bf6d4341574f88b511e856a73f6f2a921e ] This can be triggered for example by adding the "-omand" mount option, which will be rejected and virtio_fs_fill_super() will return an error. In such a case the allocations for fuse_conn and fuse_mount will leak due to s_root not yet being set and so ->put_super() not being called. Fixes: a62a8ef9d97d ("virtio-fs: add virtiofs filesystem") Change-Id: I1a25fd72aaf8348e8bd1ca29b59df712d789e3f7 Signed-off-by: Miklos Szeredi Signed-off-by: Sasha Levin --- fs/fuse/virtio_fs.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index ed576fdc6da0..efe86c3cfc4b 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1482,6 +1482,8 @@ static int virtio_fs_get_tree(struct fs_context *fsc) if (!sb->s_root) { err = virtio_fs_fill_super(sb, fsc); if (err) { + fuse_mount_put(fm); + sb->s_fs_info = NULL; deactivate_locked_super(sb); return err; } From 35036fca049ed393b4734c530675f9f23291106a Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Tue, 9 Feb 2021 17:47:54 -0500 Subject: [PATCH 049/148] UPSTREAM: virtiofs: Fail dax mount if device does not support it [ Upstream commit 3f9b9efd82a84f27e95d0414f852caf1fa839e83 ] Right now "mount -t virtiofs -o dax myfs /mnt/virtiofs" succeeds even if filesystem deivce does not have a cache window and hence DAX can't be supported. This gives a false sense to user that they are using DAX with virtiofs but fact of the matter is that they are not. Fix this by returning error if dax can't be supported and user has asked for it. Change-Id: Ibbf05e0c354ae47b6fd897f454d0712b202e0a34 Signed-off-by: Vivek Goyal Reviewed-by: Stefan Hajnoczi Signed-off-by: Miklos Szeredi Signed-off-by: Sasha Levin --- fs/fuse/virtio_fs.c | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index efe86c3cfc4b..e852a497230c 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1339,8 +1339,15 @@ static int virtio_fs_fill_super(struct super_block *sb, struct fs_context *fsc) /* virtiofs allocates and installs its own fuse devices */ ctx->fudptr = NULL; - if (ctx->dax) + if (ctx->dax) { + if (!fs->dax_dev) { + err = -EINVAL; + pr_err("virtio-fs: dax can't be enabled as filesystem" + " device does not support it.\n"); + goto err_free_fuse_devs; + } ctx->dax_dev = fs->dax_dev; + } err = fuse_fill_super_common(sb, ctx); if (err < 0) goto err_free_fuse_devs; From 6d89a44a1c0b3a94abd1dd49d200bac4a3ee9fb2 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Thu, 18 Mar 2021 09:32:48 -0700 Subject: [PATCH 050/148] BACKPORT: ANDROID: mm: page_pinner: use put_user_page at fuse_copy_finish fuse could use get_user_pages_fast by iov_iter_get_pages at fuse_copy_fill so close the false positive by attributing it by put_user_page. Page pinned via pid 670, ts 4554195916 ns PFN 83125 Block 162 type Movable Flags 0xfffffc008001e(referenced|uptodate|dirty|lru|swapbacked) try_grab_compound_head+0x1e8/0x240 internal_get_user_pages_fast+0x66d/0xca0 iov_iter_get_pages+0xd4/0x3a0 fuse_copy_fill+0x197/0x200 fuse_copy_one+0x6e/0xf0 fuse_dev_do_read.constprop.0+0x435/0x7e0 fuse_dev_read+0x5d/0x90 new_sync_read+0x115/0x1a0 vfs_read+0xf4/0x180 ksys_read+0x5f/0xe0 do_syscall_64+0x33/0x80 entry_SYSCALL_64_after_hwframe+0x44/0xae Re-name put_user_page -> put_page Bug: 183414571 Signed-off-by: Minchan Kim Signed-off-by: Minchan Kim Change-Id: Idc80d4a34b546f25e8f6dbc68313d39586e914d9 --- fs/fuse/dev.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index d40e58bd7c16..cea02c9258f1 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -690,6 +690,10 @@ static void fuse_copy_finish(struct fuse_copy_state *cs) flush_dcache_page(cs->pg); set_page_dirty_lock(cs->pg); } + /* + * The page could be GUP page(see iov_iter_get_pages in + * fuse_copy_fill) so use put_page to release it. + */ put_page(cs->pg); } cs->pg = NULL; From b9bba860d2b90eaa4ba4cfa1df865356af4cbafa Mon Sep 17 00:00:00 2001 From: Vivek Goyal Date: Tue, 6 Apr 2021 10:07:06 -0400 Subject: [PATCH 051/148] UPSTREAM: fuse: invalidate attrs when page writeback completes [ Upstream commit 3466958beb31a8e9d3a1441a34228ed088b84f3e ] In fuse when a direct/write-through write happens we invalidate attrs because that might have updated mtime/ctime on server and cached mtime/ctime will be stale. What about page writeback path. Looks like we don't invalidate attrs there. To be consistent, invalidate attrs in writeback path as well. Only exception is when writeback_cache is enabled. In that case we strust local mtime/ctime and there is no need to invalidate attrs. Recently users started experiencing failure of xfstests generic/080, geneirc/215 and generic/614 on virtiofs. This happened only newer "stat" utility and not older one. This patch fixes the issue. So what's the root cause of the issue. Here is detailed explanation. generic/080 test does mmap write to a file, closes the file and then checks if mtime has been updated or not. When file is closed, it leads to flushing of dirty pages (and that should update mtime/ctime on server). But we did not explicitly invalidate attrs after writeback finished. Still generic/080 passed so far and reason being that we invalidated atime in fuse_readpages_end(). This is called in fuse_readahead() path and always seems to trigger before mmaped write. So after mmaped write when lstat() is called, it sees that atleast one of the fields being asked for is invalid (atime) and that results in generating GETATTR to server and mtime/ctime also get updated and test passes. But newer /usr/bin/stat seems to have moved to using statx() syscall now (instead of using lstat()). And statx() allows it to query only ctime or mtime (and not rest of the basic stat fields). That means when querying for mtime, fuse_update_get_attr() sees that mtime is not invalid (only atime is invalid). So it does not generate a new GETATTR and fill stat with cached mtime/ctime. And that means updated mtime is not seen by xfstest and tests start failing. Invalidating attrs after writeback completion should solve this problem in a generic manner. Change-Id: I867c795d35306e17445d5e849676b31b2103c762 Signed-off-by: Vivek Goyal Signed-off-by: Miklos Szeredi Signed-off-by: Sasha Levin --- fs/fuse/file.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 66ccb8ee4889..50482229f571 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1821,8 +1821,17 @@ static void fuse_writepage_end(struct fuse_mount *fm, struct fuse_args *args, container_of(args, typeof(*wpa), ia.ap.args); struct inode *inode = wpa->inode; struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_conn *fc = get_fuse_conn(inode); mapping_set_error(inode->i_mapping, error); + /* + * A writeback finished and this might have updated mtime/ctime on + * server making local mtime/ctime stale. Hence invalidate attrs. + * Do this only if writeback_cache is not enabled. If writeback_cache + * is enabled, we trust local ctime/mtime. + */ + if (!fc->writeback_cache) + fuse_invalidate_attr(inode); spin_lock(&fi->lock); rb_erase(&wpa->writepages_entry, &fi->writepages); while (wpa->next) { From 3d04982cea72c565c458272743fe2fd20d9ce1c2 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Wed, 14 Apr 2021 10:40:58 +0200 Subject: [PATCH 052/148] UPSTREAM: virtiofs: fix userns [ Upstream commit 0a7419c68a45d2d066b996be5087aa2d07ce80eb ] get_user_ns() is done twice (once in virtio_fs_get_tree() and once in fuse_conn_init()), resulting in a reference leak. Also looks better to use fsc->user_ns (which *should* be the current_user_ns() at this point). Change-Id: If41a5cefac82bc65e880ad14ed90f1adca9271ae Signed-off-by: Miklos Szeredi Signed-off-by: Sasha Levin --- fs/fuse/virtio_fs.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index e852a497230c..f8341da273f8 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1474,8 +1474,7 @@ static int virtio_fs_get_tree(struct fs_context *fsc) return -ENOMEM; } - fuse_conn_init(fc, fm, get_user_ns(current_user_ns()), - &virtio_fs_fiq_ops, fs); + fuse_conn_init(fc, fm, fsc->user_ns, &virtio_fs_fiq_ops, fs); fc->release = fuse_free_conn; fc->delete_stale = true; fc->auto_submounts = true; From bc9748cb2a076347b0c4006035a14f2450d1e26f Mon Sep 17 00:00:00 2001 From: Greg Kurz Date: Fri, 4 Jun 2021 18:11:50 +0200 Subject: [PATCH 053/148] UPSTREAM: fuse: Fix crash in fuse_dentry_automount() error path commit d92d88f0568e97c437eeb79d9c9609bd8277406f upstream. If fuse_fill_super_submount() returns an error, the error path triggers a crash: [ 26.206673] BUG: kernel NULL pointer dereference, address: 0000000000000000 [...] [ 26.226362] RIP: 0010:__list_del_entry_valid+0x25/0x90 [...] [ 26.247938] Call Trace: [ 26.248300] fuse_mount_remove+0x2c/0x70 [fuse] [ 26.248892] virtio_kill_sb+0x22/0x160 [virtiofs] [ 26.249487] deactivate_locked_super+0x36/0xa0 [ 26.250077] fuse_dentry_automount+0x178/0x1a0 [fuse] The crash happens because fuse_mount_remove() assumes that the FUSE mount was already added to list under the FUSE connection, but this only done after fuse_fill_super_submount() has returned success. This means that until fuse_fill_super_submount() has returned success, the FUSE mount isn't actually owned by the superblock. We should thus reclaim ownership by clearing sb->s_fs_info, which will skip the call to fuse_mount_remove(), and perform rollback, like virtio_fs_get_tree() already does for the root sb. Fixes: bf109c64040f ("fuse: implement crossmounts") Cc: stable@vger.kernel.org # v5.10+ Change-Id: Ie395b453a703cf3f1dda2c825c7de0fe4bd62d04 Signed-off-by: Greg Kurz Reviewed-by: Max Reitz Signed-off-by: Miklos Szeredi Signed-off-by: Greg Kroah-Hartman --- fs/fuse/dir.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 117edf89e67d..e80961dfc29d 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -377,8 +377,12 @@ static struct vfsmount *fuse_dentry_automount(struct path *path) /* Initialize superblock, making @mp_fi its root */ err = fuse_fill_super_submount(sb, mp_fi); - if (err) + if (err) { + fuse_conn_put(fc); + kfree(fm); + sb->s_fs_info = NULL; goto out_put_sb; + } sb->s_flags |= SB_ACTIVE; fsc->root = dget(sb->s_root); From e673499688c6061f2a2a8bb5086e3d0cf134c172 Mon Sep 17 00:00:00 2001 From: Greg Kurz Date: Fri, 4 Jun 2021 18:11:51 +0200 Subject: [PATCH 054/148] UPSTREAM: fuse: Fix crash if superblock of submount gets killed early commit e3a43f2a95393000778f8f302d48795add2fc4a8 upstream. As soon as fuse_dentry_automount() does up_write(&sb->s_umount), the superblock can theoretically be killed. If this happens before the submount was added to the &fc->mounts list, fuse_mount_remove() later crashes in list_del_init() because it assumes the submount to be already there. Add the submount before dropping sb->s_umount to fix the inconsistency. It is okay to nest fc->killsb under sb->s_umount, we already do this on the ->kill_sb() path. Change-Id: I5fb3786f86b7a7f1492f557284dd05d67bc8dce0 Signed-off-by: Greg Kurz Fixes: bf109c64040f ("fuse: implement crossmounts") Cc: stable@vger.kernel.org # v5.10+ Reviewed-by: Max Reitz Signed-off-by: Miklos Szeredi Signed-off-by: Greg Kroah-Hartman --- fs/fuse/dir.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index e80961dfc29d..aa3c68d98f48 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -384,15 +384,15 @@ static struct vfsmount *fuse_dentry_automount(struct path *path) goto out_put_sb; } + down_write(&fc->killsb); + list_add_tail(&fm->fc_entry, &fc->mounts); + up_write(&fc->killsb); + sb->s_flags |= SB_ACTIVE; fsc->root = dget(sb->s_root); /* We are done configuring the superblock, so unlock it */ up_write(&sb->s_umount); - down_write(&fc->killsb); - list_add_tail(&fm->fc_entry, &fc->mounts); - up_write(&fc->killsb); - /* Create the submount */ mnt = vfs_create_mount(fsc); if (IS_ERR(mnt)) { From 462f12c6dcd2a898d3fd88adb55ba4f38712192c Mon Sep 17 00:00:00 2001 From: Greg Kurz Date: Fri, 4 Jun 2021 18:11:52 +0200 Subject: [PATCH 055/148] UPSTREAM: fuse: Fix infinite loop in sget_fc() commit e4a9ccdd1c03b3dc58214874399d24331ea0a3ab upstream. We don't set the SB_BORN flag on submounts. This is wrong as these superblocks are then considered as partially constructed or dying in the rest of the code and can break some assumptions. One such case is when you have a virtiofs filesystem with submounts and you try to mount it again : virtio_fs_get_tree() tries to obtain a superblock with sget_fc(). The logic in sget_fc() is to loop until it has either found an existing matching superblock with SB_BORN set or to create a brand new one. It is assumed that a superblock without SB_BORN is transient and the loop is restarted. Forgetting to set SB_BORN on submounts hence causes sget_fc() to retry forever. Setting SB_BORN requires special care, i.e. a write barrier for super_cache_count() which can check SB_BORN without taking any lock. We should call vfs_get_tree() to deal with that but this requires to have a proper ->get_tree() implementation for submounts, which is a bigger piece of work. Go for a simple bug fix in the meatime. Fixes: bf109c64040f ("fuse: implement crossmounts") Cc: stable@vger.kernel.org # v5.10+ Change-Id: I709931fa4fc72c0a72707eae03b1de93a630ddf6 Signed-off-by: Greg Kurz Reviewed-by: Max Reitz Signed-off-by: Miklos Szeredi Signed-off-by: Greg Kroah-Hartman --- fs/fuse/dir.c | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index aa3c68d98f48..f805b3c12fd5 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -390,6 +390,17 @@ static struct vfsmount *fuse_dentry_automount(struct path *path) sb->s_flags |= SB_ACTIVE; fsc->root = dget(sb->s_root); + + /* + * FIXME: setting SB_BORN requires a write barrier for + * super_cache_count(). We should actually come + * up with a proper ->get_tree() implementation + * for submounts and call vfs_get_tree() to take + * care of the write barrier. + */ + smp_wmb(); + sb->s_flags |= SB_BORN; + /* We are done configuring the superblock, so unlock it */ up_write(&sb->s_umount); From 831d8c117542086d2e9a67b56a0dc7e1582a5a0d Mon Sep 17 00:00:00 2001 From: Al Viro Date: Mon, 1 Mar 2021 20:37:10 -0500 Subject: [PATCH 056/148] UPSTREAM: new helper: inode_wrong_type() commit 6e3e2c4362e41a2f18e3f7a5ad81bd2f49a47b85 upstream. inode_wrong_type(inode, mode) returns true if setting inode->i_mode to given value would've changed the inode type. We have enough of those checks open-coded to make a helper worthwhile. Change-Id: I346cb7a2d1cec2ca16096962d8adc9d7f848fc94 Signed-off-by: Al Viro Signed-off-by: Amir Goldstein Signed-off-by: Greg Kroah-Hartman --- fs/9p/vfs_inode.c | 4 ++-- fs/9p/vfs_inode_dotl.c | 4 ++-- fs/cifs/inode.c | 5 ++--- fs/fuse/dir.c | 6 +++--- fs/fuse/inode.c | 2 +- fs/fuse/readdir.c | 2 +- fs/nfs/inode.c | 6 +++--- fs/nfsd/nfsproc.c | 2 +- fs/overlayfs/namei.c | 4 ++-- include/linux/fs.h | 5 +++++ 10 files changed, 22 insertions(+), 18 deletions(-) diff --git a/fs/9p/vfs_inode.c b/fs/9p/vfs_inode.c index ffce168296bd..4446e83aea7a 100644 --- a/fs/9p/vfs_inode.c +++ b/fs/9p/vfs_inode.c @@ -455,7 +455,7 @@ static int v9fs_test_inode(struct inode *inode, void *data) umode = p9mode2unixmode(v9ses, st, &rdev); /* don't match inode of different type */ - if ((inode->i_mode & S_IFMT) != (umode & S_IFMT)) + if (inode_wrong_type(inode, umode)) return 0; /* compare qid details */ @@ -1412,7 +1412,7 @@ int v9fs_refresh_inode(struct p9_fid *fid, struct inode *inode) * Don't update inode if the file type is different */ umode = p9mode2unixmode(v9ses, st, &rdev); - if ((inode->i_mode & S_IFMT) != (umode & S_IFMT)) + if (inode_wrong_type(inode, umode)) goto out; /* diff --git a/fs/9p/vfs_inode_dotl.c b/fs/9p/vfs_inode_dotl.c index 9795fe0f1ba2..614f23a9d2bc 100644 --- a/fs/9p/vfs_inode_dotl.c +++ b/fs/9p/vfs_inode_dotl.c @@ -59,7 +59,7 @@ static int v9fs_test_inode_dotl(struct inode *inode, void *data) struct p9_stat_dotl *st = (struct p9_stat_dotl *)data; /* don't match inode of different type */ - if ((inode->i_mode & S_IFMT) != (st->st_mode & S_IFMT)) + if (inode_wrong_type(inode, st->st_mode)) return 0; if (inode->i_generation != st->st_gen) @@ -924,7 +924,7 @@ int v9fs_refresh_inode_dotl(struct p9_fid *fid, struct inode *inode) /* * Don't update inode if the file type is different */ - if ((inode->i_mode & S_IFMT) != (st->st_mode & S_IFMT)) + if (inode_wrong_type(inode, st->st_mode)) goto out; /* diff --git a/fs/cifs/inode.c b/fs/cifs/inode.c index af0980c720c7..932aa65182b3 100644 --- a/fs/cifs/inode.c +++ b/fs/cifs/inode.c @@ -420,8 +420,7 @@ int cifs_get_inode_info_unix(struct inode **pinode, } /* if filetype is different, return error */ - if (unlikely(((*pinode)->i_mode & S_IFMT) != - (fattr.cf_mode & S_IFMT))) { + if (unlikely(inode_wrong_type(*pinode, fattr.cf_mode))) { CIFS_I(*pinode)->time = 0; /* force reval */ rc = -ESTALE; goto cgiiu_exit; @@ -980,7 +979,7 @@ cifs_find_inode(struct inode *inode, void *opaque) return 0; /* don't match inode of different type */ - if ((inode->i_mode & S_IFMT) != (fattr->cf_mode & S_IFMT)) + if (inode_wrong_type(inode, fattr->cf_mode)) return 0; /* if it's not a directory or has no dentries, then flag it */ diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index f805b3c12fd5..0ae9091bfa0b 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -252,7 +252,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) if (ret == -ENOMEM || ret == -EINTR) goto out; if (ret || fuse_invalid_attr(&outarg.attr) || - (outarg.attr.mode ^ inode->i_mode) & S_IFMT) + inode_wrong_type(inode, outarg.attr.mode)) goto invalid; forget_all_cached_acls(inode); @@ -1102,7 +1102,7 @@ static int fuse_do_getattr(struct inode *inode, struct kstat *stat, err = fuse_simple_request(fm, &args); if (!err) { if (fuse_invalid_attr(&outarg.attr) || - (inode->i_mode ^ outarg.attr.mode) & S_IFMT) { + inode_wrong_type(inode, outarg.attr.mode)) { fuse_make_bad(inode); err = -EIO; } else { @@ -1739,7 +1739,7 @@ int fuse_do_setattr(struct dentry *dentry, struct iattr *attr, } if (fuse_invalid_attr(&outarg.attr) || - (inode->i_mode ^ outarg.attr.mode) & S_IFMT) { + inode_wrong_type(inode, outarg.attr.mode)) { fuse_make_bad(inode); err = -EIO; goto error; diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index bd35be7ba89f..31202f88f2d7 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -346,7 +346,7 @@ retry: inode->i_generation = generation; fuse_init_inode(inode, attr); unlock_new_inode(inode); - } else if ((inode->i_mode ^ attr->mode) & S_IFMT) { + } else if (inode_wrong_type(inode, attr->mode)) { /* Inode has changed type, any I/O on the old should fail */ fuse_make_bad(inode); if (inode != d_inode(sb->s_root)) { diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index b0f0585dbdba..6430210643c4 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -202,7 +202,7 @@ retry: inode = d_inode(dentry); if (!inode || get_node_id(inode) != o->nodeid || - ((o->attr.mode ^ inode->i_mode) & S_IFMT)) { + inode_wrong_type(inode, o->attr.mode)) { d_invalidate(dentry); dput(dentry); goto retry; diff --git a/fs/nfs/inode.c b/fs/nfs/inode.c index 4f0d2fc810e4..a4605b17f047 100644 --- a/fs/nfs/inode.c +++ b/fs/nfs/inode.c @@ -310,7 +310,7 @@ nfs_find_actor(struct inode *inode, void *opaque) if (NFS_FILEID(inode) != fattr->fileid) return 0; - if ((S_IFMT & inode->i_mode) != (S_IFMT & fattr->mode)) + if (inode_wrong_type(inode, fattr->mode)) return 0; if (nfs_compare_fh(NFS_FH(inode), fh)) return 0; @@ -1418,7 +1418,7 @@ static int nfs_check_inode_attributes(struct inode *inode, struct nfs_fattr *fat return 0; return -ESTALE; } - if ((fattr->valid & NFS_ATTR_FATTR_TYPE) && (inode->i_mode & S_IFMT) != (fattr->mode & S_IFMT)) + if ((fattr->valid & NFS_ATTR_FATTR_TYPE) && inode_wrong_type(inode, fattr->mode)) return -ESTALE; @@ -1834,7 +1834,7 @@ static int nfs_update_inode(struct inode *inode, struct nfs_fattr *fattr) /* * Make sure the inode's type hasn't changed. */ - if ((fattr->valid & NFS_ATTR_FATTR_TYPE) && (inode->i_mode & S_IFMT) != (fattr->mode & S_IFMT)) { + if ((fattr->valid & NFS_ATTR_FATTR_TYPE) && inode_wrong_type(inode, fattr->mode)) { /* * Big trouble! The inode has become a different object. */ diff --git a/fs/nfsd/nfsproc.c b/fs/nfsd/nfsproc.c index 4aca93e11af7..5f6186b2ca9b 100644 --- a/fs/nfsd/nfsproc.c +++ b/fs/nfsd/nfsproc.c @@ -376,7 +376,7 @@ nfsd_proc_create(struct svc_rqst *rqstp) /* Make sure the type and device matches */ nfserr = nfserr_exist; - if (inode && type != (inode->i_mode & S_IFMT)) + if (inode && inode_wrong_type(inode, type)) goto out_unlock; } diff --git a/fs/overlayfs/namei.c b/fs/overlayfs/namei.c index bcfb0b11b819..f7f259614143 100644 --- a/fs/overlayfs/namei.c +++ b/fs/overlayfs/namei.c @@ -345,7 +345,7 @@ int ovl_check_origin_fh(struct ovl_fs *ofs, struct ovl_fh *fh, bool connected, return PTR_ERR(origin); if (upperdentry && !ovl_is_whiteout(upperdentry) && - ((d_inode(origin)->i_mode ^ d_inode(upperdentry)->i_mode) & S_IFMT)) + inode_wrong_type(d_inode(upperdentry), d_inode(origin)->i_mode)) goto invalid; if (!*stackp) @@ -717,7 +717,7 @@ struct dentry *ovl_lookup_index(struct ovl_fs *ofs, struct dentry *upper, index = ERR_PTR(-ESTALE); goto out; } else if (ovl_dentry_weird(index) || ovl_is_whiteout(index) || - ((inode->i_mode ^ d_inode(origin)->i_mode) & S_IFMT)) { + inode_wrong_type(inode, d_inode(origin)->i_mode)) { /* * Index should always be of the same file type as origin * except for the case of a whiteout index. A whiteout diff --git a/include/linux/fs.h b/include/linux/fs.h index 773492ba2e6e..53be8d36da9c 100644 --- a/include/linux/fs.h +++ b/include/linux/fs.h @@ -2985,6 +2985,11 @@ static inline bool execute_ok(struct inode *inode) return (inode->i_mode & S_IXUGO) || S_ISDIR(inode->i_mode); } +static inline bool inode_wrong_type(const struct inode *inode, umode_t mode) +{ + return (inode->i_mode ^ mode) & S_IFMT; +} + static inline void file_start_write(struct file *file) { if (!S_ISREG(file_inode(file)->i_mode)) From 26f1e21656c27109e791a94f9068a21c1889b315 Mon Sep 17 00:00:00 2001 From: Amir Goldstein Date: Mon, 21 Jun 2021 14:03:53 +0300 Subject: [PATCH 057/148] UPSTREAM: fuse: fix illegal access to inode with reused nodeid commit 15db16837a35d8007cb8563358787412213db25e upstream. Server responds to LOOKUP and other ops (READDIRPLUS/CREATE/MKNOD/...) with ourarg containing nodeid and generation. If a fuse inode is found in inode cache with the same nodeid but different generation, the existing fuse inode should be unhashed and marked "bad" and a new inode with the new generation should be hashed instead. This can happen, for example, with passhrough fuse filesystem that returns the real filesystem ino/generation on lookup and where real inode numbers can get recycled due to real files being unlinked not via the fuse passthrough filesystem. With current code, this situation will not be detected and an old fuse dentry that used to point to an older generation real inode, can be used to access a completely new inode, which should be accessed only via the new dentry. Note that because the FORGET message carries the nodeid w/o generation, the server should wait to get FORGET counts for the nlookup counts of the old and reused inodes combined, before it can free the resources associated to that nodeid. Stable backport notes: * This is not a regression. The bug has been in fuse forever, but only a certain class of low level fuse filesystems can trigger this bug * Because there is no way to check if this fix is applied in runtime, libfuse test_examples.py tests this fix with hardcoded check for kernel version >= 5.14 * After backport to stable kernel(s), the libfuse test can be updated to also check minimal stable kernel version(s) * Depends on "fuse: fix bad inode" which is already applied to stable kernels v5.4.y and v5.10.y * Required backporting helper inode_wrong_type() Change-Id: I5f92158bc8ccdb50627fb035bb28d3ade51820b3 Signed-off-by: Amir Goldstein Signed-off-by: Miklos Szeredi Cc: stable@vger.kernel.org Link: https://lore.kernel.org/linux-fsdevel/CAOQ4uxi8DymG=JO_sAU+wS8akFdzh+PuXwW3Ebgahd2Nwnh7zA@mail.gmail.com/ Signed-off-by: Amir Goldstein Signed-off-by: Greg Kroah-Hartman --- fs/fuse/dir.c | 2 +- fs/fuse/fuse_i.h | 7 +++++++ fs/fuse/inode.c | 4 ++-- fs/fuse/readdir.c | 7 +++++-- 4 files changed, 15 insertions(+), 5 deletions(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 0ae9091bfa0b..c9d0326c045f 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -252,7 +252,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) if (ret == -ENOMEM || ret == -EINTR) goto out; if (ret || fuse_invalid_attr(&outarg.attr) || - inode_wrong_type(inode, outarg.attr.mode)) + fuse_stale_inode(inode, outarg.generation, &outarg.attr)) goto invalid; forget_all_cached_acls(inode); diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index a51d1b91ef8b..d9109bf46aaa 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -885,6 +885,13 @@ static inline u64 fuse_get_attr_version(struct fuse_conn *fc) return atomic64_read(&fc->attr_version); } +static inline bool fuse_stale_inode(const struct inode *inode, int generation, + struct fuse_attr *attr) +{ + return inode->i_generation != generation || + inode_wrong_type(inode, attr->mode); +} + static inline void fuse_make_bad(struct inode *inode) { set_bit(FUSE_I_BAD, &get_fuse_inode(inode)->state); diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 31202f88f2d7..03bee760ec8a 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -346,8 +346,8 @@ retry: inode->i_generation = generation; fuse_init_inode(inode, attr); unlock_new_inode(inode); - } else if (inode_wrong_type(inode, attr->mode)) { - /* Inode has changed type, any I/O on the old should fail */ + } else if (fuse_stale_inode(inode, generation, attr)) { + /* nodeid was reused, any I/O on the old inode should fail */ fuse_make_bad(inode); if (inode != d_inode(sb->s_root)) { remove_inode_hash(inode); diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index 6430210643c4..bcb56cf0a0c3 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -200,9 +200,12 @@ retry: if (!d_in_lookup(dentry)) { struct fuse_inode *fi; inode = d_inode(dentry); + if (inode && get_node_id(inode) != o->nodeid) + inode = NULL; if (!inode || - get_node_id(inode) != o->nodeid || - inode_wrong_type(inode, o->attr.mode)) { + fuse_stale_inode(inode, o->generation, &o->attr)) { + if (inode) + fuse_make_bad(inode); d_invalidate(dentry); dput(dentry); goto retry; From 2e6dc19099eada09d7e70f19935735cf018d3740 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Thu, 2 Dec 2021 13:50:02 -0800 Subject: [PATCH 058/148] BACKPORT: ANDROID: fuse: Move functions in preparation for fuse-bpf Contains squash of this commit: Author: Nathan Chancellor Date: Fri Jan 20 09:40:12 2023 -0700 ANDROID: fuse: Restore upstream type of bitfields in fuse_args Commit 88b7179fcdb59 ("ANDROID: fuse: Move functions in preparation for fuse-bpf") changed the type of these bitfields from the upstream type of 'bool' to 'int', which causes several warnings with recent versions of clang: /builds/linux/fs/fuse/dir.c:168:19: error: implicit truncation from 'int' to a one-bit wide bit-field changes value from 1 to -1 [-Werror,-Wsingle-bi t-bitfield-constant-conversion] args->out_argvar = true; ^ ~~~~ /builds/linux/fs/fuse/dir.c:492:18: error: implicit truncation from 'int' to a one-bit wide bit-field changes value from 1 to -1 [-Werror,-Wsingle-bi t-bitfield-constant-conversion] args.out_argvar = 1; ^ ~ /builds/linux/fs/fuse/dir.c:1649:20: error: implicit truncation from 'int' to a one-bit wide bit-field changes value from 1 to -1 [-Werror,-Wsingle-b it-bitfield-constant-conversion] ap.args.out_pages = true; ^ ~~~~ /builds/linux/fs/fuse/dir.c:1650:21: error: implicit truncation from 'int' to a one-bit wide bit-field changes value from 1 to -1 [-Werror,-Wsingle-b it-bitfield-constant-conversion] ap.args.out_argvar = true; ^ ~~~~ /builds/linux/fs/fuse/dir.c:1651:23: error: implicit truncation from 'int' to a one-bit wide bit-field changes value from 1 to -1 [-Werror,-Wsingle-b it-bitfield-constant-conversion] ap.args.page_zeroing = true; ^ ~~~~ 5 errors generated. When fuse_args was moved back to the internal implementation in commit 9a5023967b4d2 ("ANDROID: fuse-bpf: Use fuse_bpf_args in uapi"), the type was not restored. Do so now to fix the warnings and reduce the delta with upstream. Bug: 265200230 Change-Id: I4d51f331d842a1faff9a937140f0275130e70d73 Signed-off-by: Nathan Chancellor Bug: 202785178 Test: test_fuse passes on linux, feature works on cuttlefish Signed-off-by: Paul Lawrence Signed-off-by: Daniel Rosenberg Change-Id: Ie738893a821d1f5f252c4c6e86274d55a6f09965 --- fs/fuse/dir.c | 81 +------------------------ fs/fuse/fuse_i.h | 120 +++++++++++++++++++++++++++----------- fs/fuse/inode.c | 46 +++++++-------- fs/fuse/readdir.c | 4 +- include/uapi/linux/fuse.h | 44 ++++++++++++++ 5 files changed, 155 insertions(+), 140 deletions(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index c9d0326c045f..82a99eb6125a 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -74,28 +74,6 @@ static void fuse_dentry_settime(struct dentry *dentry, u64 time) __fuse_dentry_settime(dentry, time); } -/* - * FUSE caches dentries and attributes with separate timeout. The - * time in jiffies until the dentry/attributes are valid is stored in - * dentry->d_fsdata and fuse_inode->i_time respectively. - */ - -/* - * Calculate the time in jiffies until a dentry/attributes are valid - */ -static u64 time_to_jiffies(u64 sec, u32 nsec) -{ - if (sec || nsec) { - struct timespec64 ts = { - sec, - min_t(u32, nsec, NSEC_PER_SEC - 1) - }; - - return get_jiffies_64() + timespec64_to_jiffies(&ts); - } else - return 0; -} - /* * Set dentry and possibly attribute timeouts from the lookup/mk* * replies @@ -106,11 +84,6 @@ void fuse_change_entry_timeout(struct dentry *entry, struct fuse_entry_out *o) time_to_jiffies(o->entry_valid, o->entry_valid_nsec)); } -static u64 attr_timeout(struct fuse_attr_out *o) -{ - return time_to_jiffies(o->attr_valid, o->attr_valid_nsec); -} - u64 entry_attr_timeout(struct fuse_entry_out *o) { return time_to_jiffies(o->attr_valid, o->attr_valid_nsec); @@ -1031,7 +1004,7 @@ static int fuse_link(struct dentry *entry, struct inode *newdir, return err; } -static void fuse_fillattr(struct inode *inode, struct fuse_attr *attr, +void fuse_fillattr(struct inode *inode, struct fuse_attr *attr, struct kstat *stat) { unsigned int blkbits; @@ -1492,58 +1465,6 @@ static long fuse_dir_compat_ioctl(struct file *file, unsigned int cmd, FUSE_IOCTL_COMPAT | FUSE_IOCTL_DIR); } -static bool update_mtime(unsigned ivalid, bool trust_local_mtime) -{ - /* Always update if mtime is explicitly set */ - if (ivalid & ATTR_MTIME_SET) - return true; - - /* Or if kernel i_mtime is the official one */ - if (trust_local_mtime) - return true; - - /* If it's an open(O_TRUNC) or an ftruncate(), don't update */ - if ((ivalid & ATTR_SIZE) && (ivalid & (ATTR_OPEN | ATTR_FILE))) - return false; - - /* In all other cases update */ - return true; -} - -static void iattr_to_fattr(struct fuse_conn *fc, struct iattr *iattr, - struct fuse_setattr_in *arg, bool trust_local_cmtime) -{ - unsigned ivalid = iattr->ia_valid; - - if (ivalid & ATTR_MODE) - arg->valid |= FATTR_MODE, arg->mode = iattr->ia_mode; - if (ivalid & ATTR_UID) - arg->valid |= FATTR_UID, arg->uid = from_kuid(fc->user_ns, iattr->ia_uid); - if (ivalid & ATTR_GID) - arg->valid |= FATTR_GID, arg->gid = from_kgid(fc->user_ns, iattr->ia_gid); - if (ivalid & ATTR_SIZE) - arg->valid |= FATTR_SIZE, arg->size = iattr->ia_size; - if (ivalid & ATTR_ATIME) { - arg->valid |= FATTR_ATIME; - arg->atime = iattr->ia_atime.tv_sec; - arg->atimensec = iattr->ia_atime.tv_nsec; - if (!(ivalid & ATTR_ATIME_SET)) - arg->valid |= FATTR_ATIME_NOW; - } - if ((ivalid & ATTR_MTIME) && update_mtime(ivalid, trust_local_cmtime)) { - arg->valid |= FATTR_MTIME; - arg->mtime = iattr->ia_mtime.tv_sec; - arg->mtimensec = iattr->ia_mtime.tv_nsec; - if (!(ivalid & ATTR_MTIME_SET) && !trust_local_cmtime) - arg->valid |= FATTR_MTIME_NOW; - } - if ((ivalid & ATTR_CTIME) && trust_local_cmtime) { - arg->valid |= FATTR_CTIME; - arg->ctime = iattr->ia_ctime.tv_sec; - arg->ctimensec = iattr->ia_ctime.tv_nsec; - } -} - /* * Prevent concurrent writepages on inode * diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index d9109bf46aaa..92a8bff97bae 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -249,47 +249,12 @@ struct fuse_file { bool flock:1; }; -/** One input argument of a request */ -struct fuse_in_arg { - unsigned size; - const void *value; -}; - -/** One output argument of a request */ -struct fuse_arg { - unsigned size; - void *value; -}; - /** FUSE page descriptor */ struct fuse_page_desc { unsigned int length; unsigned int offset; }; -struct fuse_args { - uint64_t nodeid; - uint32_t opcode; - unsigned short in_numargs; - unsigned short out_numargs; - bool force:1; - bool noreply:1; - bool nocreds:1; - bool in_pages:1; - bool out_pages:1; - bool user_pages:1; - bool out_argvar:1; - bool page_zeroing:1; - bool page_replace:1; - bool may_block:1; - struct fuse_in_arg in_args[3]; - struct fuse_arg out_args[2]; - void (*end)(struct fuse_mount *fm, struct fuse_args *args, int error); - - /* Path used for completing d_canonical_path */ - struct path *canonical_path; -}; - struct fuse_args_pages { struct fuse_args args; struct page **pages; @@ -954,6 +919,9 @@ void fuse_read_args_fill(struct fuse_io_args *ia, struct file *file, loff_t pos, size_t count, int opcode); +int fuse_parse_dirfile(char *buf, size_t nbytes, struct file *file, + struct dir_context *ctx); + /** * Send OPEN or OPENDIR request */ @@ -1264,4 +1232,86 @@ ssize_t fuse_passthrough_read_iter(struct kiocb *iocb, struct iov_iter *to); ssize_t fuse_passthrough_write_iter(struct kiocb *iocb, struct iov_iter *from); ssize_t fuse_passthrough_mmap(struct file *file, struct vm_area_struct *vma); +/* + * FUSE caches dentries and attributes with separate timeout. The + * time in jiffies until the dentry/attributes are valid is stored in + * dentry->d_fsdata and fuse_inode->i_time respectively. + */ + +/* + * Calculate the time in jiffies until a dentry/attributes are valid + */ +static inline u64 time_to_jiffies(u64 sec, u32 nsec) +{ + if (sec || nsec) { + struct timespec64 ts = { + sec, + min_t(u32, nsec, NSEC_PER_SEC - 1) + }; + + return get_jiffies_64() + timespec64_to_jiffies(&ts); + } else + return 0; +} + +static inline u64 attr_timeout(struct fuse_attr_out *o) +{ + return time_to_jiffies(o->attr_valid, o->attr_valid_nsec); +} + +static inline bool update_mtime(unsigned ivalid, bool trust_local_mtime) +{ + /* Always update if mtime is explicitly set */ + if (ivalid & ATTR_MTIME_SET) + return true; + + /* Or if kernel i_mtime is the official one */ + if (trust_local_mtime) + return true; + + /* If it's an open(O_TRUNC) or an ftruncate(), don't update */ + if ((ivalid & ATTR_SIZE) && (ivalid & (ATTR_OPEN | ATTR_FILE))) + return false; + + /* In all other cases update */ + return true; +} + +void fuse_fillattr(struct inode *inode, struct fuse_attr *attr, + struct kstat *stat); + +static inline void iattr_to_fattr(struct fuse_conn *fc, struct iattr *iattr, + struct fuse_setattr_in *arg, bool trust_local_cmtime) +{ + unsigned ivalid = iattr->ia_valid; + + if (ivalid & ATTR_MODE) + arg->valid |= FATTR_MODE, arg->mode = iattr->ia_mode; + if (ivalid & ATTR_UID) + arg->valid |= FATTR_UID, arg->uid = from_kuid(fc->user_ns, iattr->ia_uid); + if (ivalid & ATTR_GID) + arg->valid |= FATTR_GID, arg->gid = from_kgid(fc->user_ns, iattr->ia_gid); + if (ivalid & ATTR_SIZE) + arg->valid |= FATTR_SIZE, arg->size = iattr->ia_size; + if (ivalid & ATTR_ATIME) { + arg->valid |= FATTR_ATIME; + arg->atime = iattr->ia_atime.tv_sec; + arg->atimensec = iattr->ia_atime.tv_nsec; + if (!(ivalid & ATTR_ATIME_SET)) + arg->valid |= FATTR_ATIME_NOW; + } + if ((ivalid & ATTR_MTIME) && update_mtime(ivalid, trust_local_cmtime)) { + arg->valid |= FATTR_MTIME; + arg->mtime = iattr->ia_mtime.tv_sec; + arg->mtimensec = iattr->ia_mtime.tv_nsec; + if (!(ivalid & ATTR_MTIME_SET) && !trust_local_cmtime) + arg->valid |= FATTR_MTIME_NOW; + } + if ((ivalid & ATTR_CTIME) && trust_local_cmtime) { + arg->valid |= FATTR_CTIME; + arg->ctime = iattr->ia_ctime.tv_sec; + arg->ctimensec = iattr->ia_ctime.tv_nsec; + } +} + #endif /* _FS_FUSE_I_H */ diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 03bee760ec8a..679af097f760 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -161,6 +161,28 @@ static ino_t fuse_squash_ino(u64 ino64) return ino; } +static void fuse_fill_attr_from_inode(struct fuse_attr *attr, + const struct inode *inode) +{ + *attr = (struct fuse_attr){ + .ino = inode->i_ino, + .size = inode->i_size, + .blocks = inode->i_blocks, + .atime = inode->i_atime.tv_sec, + .mtime = inode->i_mtime.tv_sec, + .ctime = inode->i_ctime.tv_sec, + .atimensec = inode->i_atime.tv_nsec, + .mtimensec = inode->i_mtime.tv_nsec, + .ctimensec = inode->i_ctime.tv_nsec, + .mode = inode->i_mode, + .nlink = inode->i_nlink, + .uid = inode->i_uid.val, + .gid = inode->i_gid.val, + .rdev = inode->i_rdev, + .blksize = 1u << inode->i_blkbits, + }; +} + void fuse_change_attributes_common(struct inode *inode, struct fuse_attr *attr, u64 attr_valid) { @@ -1266,28 +1288,6 @@ void fuse_dev_free(struct fuse_dev *fud) } EXPORT_SYMBOL_GPL(fuse_dev_free); -static void fuse_fill_attr_from_inode(struct fuse_attr *attr, - const struct fuse_inode *fi) -{ - *attr = (struct fuse_attr){ - .ino = fi->inode.i_ino, - .size = fi->inode.i_size, - .blocks = fi->inode.i_blocks, - .atime = fi->inode.i_atime.tv_sec, - .mtime = fi->inode.i_mtime.tv_sec, - .ctime = fi->inode.i_ctime.tv_sec, - .atimensec = fi->inode.i_atime.tv_nsec, - .mtimensec = fi->inode.i_mtime.tv_nsec, - .ctimensec = fi->inode.i_ctime.tv_nsec, - .mode = fi->inode.i_mode, - .nlink = fi->inode.i_nlink, - .uid = fi->inode.i_uid.val, - .gid = fi->inode.i_gid.val, - .rdev = fi->inode.i_rdev, - .blksize = 1u << fi->inode.i_blkbits, - }; -} - static void fuse_sb_defaults(struct super_block *sb) { sb->s_magic = FUSE_SUPER_MAGIC; @@ -1331,7 +1331,7 @@ int fuse_fill_super_submount(struct super_block *sb, if (parent_sb->s_subtype && !sb->s_subtype) return -ENOMEM; - fuse_fill_attr_from_inode(&root_attr, parent_fi); + fuse_fill_attr_from_inode(&root_attr, &parent_fi->inode); root = fuse_iget(sb, parent_fi->nodeid, 0, &root_attr, 0, 0); /* * This inode is just a duplicate, so it is not looked up and diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index bcb56cf0a0c3..9413a45734c0 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -121,7 +121,7 @@ static bool fuse_emit(struct file *file, struct dir_context *ctx, dirent->type); } -static int parse_dirfile(char *buf, size_t nbytes, struct file *file, +int fuse_parse_dirfile(char *buf, size_t nbytes, struct file *file, struct dir_context *ctx) { while (nbytes >= FUSE_NAME_OFFSET) { @@ -368,7 +368,7 @@ static int fuse_readdir_uncached(struct file *file, struct dir_context *ctx) res = parse_dirplusfile(page_address(page), res, file, ctx, attr_version); } else { - res = parse_dirfile(page_address(page), res, file, + res = fuse_parse_dirfile(page_address(page), res, file, ctx); } } diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index bf3c62dd7112..6620e0426355 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -634,6 +634,12 @@ struct fuse_read_in { uint32_t padding; }; +struct fuse_read_out { + uint64_t offset; + uint32_t again; + uint32_t padding; +}; + #define FUSE_COMPAT_WRITE_IN_SIZE 24 struct fuse_write_in { @@ -942,4 +948,42 @@ struct fuse_removemapping_one { #define FUSE_REMOVEMAPPING_MAX_ENTRY \ (PAGE_SIZE / sizeof(struct fuse_removemapping_one)) +struct fuse_mount; + +/** One input argument of a request */ +struct fuse_in_arg { + unsigned size; + const void *value; +}; + +/** One output argument of a request */ +struct fuse_arg { + unsigned size; + void *value; +}; + +struct fuse_args { + uint64_t nodeid; + uint32_t opcode; + uint32_t error_in; + unsigned short in_numargs; + unsigned short out_numargs; + bool force:1; + bool noreply:1; + bool nocreds:1; + bool in_pages:1; + bool out_pages:1; + bool user_pages:1; + bool out_argvar:1; + bool page_zeroing:1; + bool page_replace:1; + bool may_block:1; + struct fuse_in_arg in_args[3]; + struct fuse_arg out_args[2]; + void (*end)(struct fuse_mount *fm, struct fuse_args *args, int error); + + /* Path used for completing d_canonical_path */ + struct path *canonical_path; +}; + #endif /* _LINUX_FUSE_H */ From b3fbcd3ca5800b12f8580ea4709652f92dad1176 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Thu, 2 Dec 2021 14:38:56 -0800 Subject: [PATCH 059/148] BACKPORT: ANDROID: fuse-bpf v1 Bug: 202785178 Test: test_fuse passes on linux, feature works on cuttlefish Signed-off-by: Paul Lawrence Signed-off-by: Daniel Rosenberg Change-Id: I987684b799b07391ccde350e98fde7976f5601aa --- fs/fuse/Kconfig | 8 + fs/fuse/Makefile | 1 + fs/fuse/backing.c | 1924 +++++++++++++++++++++++++++++++++++++ fs/fuse/dev.c | 6 + fs/fuse/dir.c | 382 +++++++- fs/fuse/file.c | 107 ++- fs/fuse/fuse_i.h | 519 +++++++++- fs/fuse/inode.c | 179 +++- fs/fuse/readdir.c | 15 + fs/fuse/xattr.c | 33 + include/linux/bpf_types.h | 3 + include/uapi/linux/bpf.h | 1 + include/uapi/linux/fuse.h | 25 +- kernel/bpf/Makefile | 3 + kernel/bpf/bpf_fuse.c | 72 ++ kernel/bpf/btf.c | 1 + 16 files changed, 3211 insertions(+), 68 deletions(-) create mode 100644 fs/fuse/backing.c create mode 100644 kernel/bpf/bpf_fuse.c diff --git a/fs/fuse/Kconfig b/fs/fuse/Kconfig index ac32f7c956a8..ab3428492b61 100644 --- a/fs/fuse/Kconfig +++ b/fs/fuse/Kconfig @@ -52,3 +52,11 @@ config FUSE_DAX If you want to allow mounting a Virtio Filesystem with the "dax" option, answer Y. + +config FUSE_BPF + bool "Adds BPF to fuse" + depends on FUSE_FS + depends on BPF + help + Extends FUSE by adding BPF to prefilter calls and potentially pass to a + backing file system diff --git a/fs/fuse/Makefile b/fs/fuse/Makefile index 8729e1e2033b..7bd922ff8604 100644 --- a/fs/fuse/Makefile +++ b/fs/fuse/Makefile @@ -9,5 +9,6 @@ obj-$(CONFIG_VIRTIO_FS) += virtiofs.o fuse-y := dev.o dir.o file.o inode.o control.o xattr.o acl.o readdir.o passthrough.o fuse-$(CONFIG_FUSE_DAX) += dax.o +fuse-$(CONFIG_FUSE_BPF) += backing.o virtiofs-y := virtio_fs.o diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c new file mode 100644 index 000000000000..f30fe8ef5c05 --- /dev/null +++ b/fs/fuse/backing.c @@ -0,0 +1,1924 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * FUSE-BPF: Filesystem in Userspace with BPF + * Copyright (c) 2021 Google LLC + */ + +#include "fuse_i.h" + +#include +#include +#include +#include + +#include "../internal.h" + +/* Reimplement these functions since fget_task is not exported */ +static struct file *fuse__fget_files(struct files_struct *files, + unsigned int fd, fmode_t mask, unsigned int refs) +{ + struct file *file; + + rcu_read_lock(); +loop: + file = fcheck_files(files, fd); + if (file) { + /* File object ref couldn't be taken. + * dup2() atomicity guarantee is the reason + * we loop to catch the new file (or NULL pointer) + */ + if (file->f_mode & mask) + file = NULL; + else if (!get_file_rcu_many(file, refs)) + goto loop; + } + rcu_read_unlock(); + return file; +} + +static struct file *fuse_fget_task(struct task_struct *task, unsigned int fd) +{ + struct file *file = NULL; + + task_lock(task); + if (task->files) + file = fuse__fget_files(task->files, fd, 0, 1); + task_unlock(task); + + return file; +} + +struct file *fuse_fget(struct fuse_conn *fc, unsigned int fd) +{ + return fuse_fget_task(fc->task, fd); +} + +struct bpf_prog *fuse_get_bpf_prog(struct fuse_conn *fc, unsigned int fd) +{ + struct file *bpf_file = fuse_fget(fc, fd); + struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); + + if (!bpf_file) + goto out; + /** + * Two ways of getting a bpf prog from another task's fd, since + * bpf_prog_get_type_dev only works with an fd + * + * 1) Duplicate a little of the needed code. Requires access to + * bpf_prog_fops for validation, which is not exported for modules + * 2) Insert the bpf_file object into a fd from the current task + * Stupidly complex, but I think OK, as security checks are not run + * during the existence of the handle + * + * Best would be to upstream 1) into kernel/bpf/syscall.c and export it + * for use here. Failing that, we have to use 2, since fuse must be + * compilable as a module. + */ +#if 0 + if (bpf_file->f_op != &bpf_prog_fops) + goto out; + + bpf_prog = bpf_file->private_data; + if (bpf_prog->type == BPF_PROG_TYPE_FUSE) + bpf_prog_inc(bpf_prog); + else + bpf_prog = ERR_PTR(-EINVAL); + +#else + { + int task_fd = get_unused_fd_flags(bpf_file->f_flags); + + if (task_fd < 0) + goto out; + fd_install(task_fd, bpf_file); + + bpf_prog = bpf_prog_get_type_dev(task_fd, BPF_PROG_TYPE_FUSE, + false); + __close_fd(current->files, task_fd); + + /* TODO I think this file is probably being leaked */ + bpf_file = NULL; + } +#endif + +out: + if (bpf_file) + fput(bpf_file); + return bpf_prog; +} + +int fuse_open_initialize(struct fuse_args *fa, struct fuse_open_io *foio, + struct inode *inode, struct file *file, bool isdir) +{ + foio->foi = (struct fuse_open_in) { + .flags = file->f_flags & ~(O_CREAT | O_EXCL | O_NOCTTY), + }; + + foio->foo = (struct fuse_open_out) {0}; + + *fa = (struct fuse_args) { + .nodeid = get_fuse_inode(inode)->nodeid, + .opcode = isdir ? FUSE_OPENDIR : FUSE_OPEN, + .in_numargs = 1, + .out_numargs = 1, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(foio->foi), + .value = &foio->foi, + }, + .out_args[0] = (struct fuse_arg) { + .size = sizeof(foio->foo), + .value = &foio->foo, + }, + }; + + return 0; +} + +int fuse_open_backing(struct fuse_args *fa, + struct inode *inode, struct file *file, bool isdir) +{ + struct fuse_mount *fm = get_fuse_mount(inode); + const struct fuse_open_in *foi = fa->in_args[0].value; + struct fuse_file *ff; + struct fuse_dentry *fd = get_fuse_dentry(file->f_path.dentry); + struct file *backing_file; + + ff = fuse_file_alloc(fm); + if (!ff) + return -ENOMEM; + file->private_data = ff; + + backing_file = dentry_open(&fd->backing_path, + foi->flags, + current_cred()); + if (IS_ERR(backing_file)) { + fuse_file_free(ff); + file->private_data = NULL; + return PTR_ERR(backing_file); + } + ff->backing_file = backing_file; + + return 0; +} + +void *fuse_open_finalize(struct fuse_args *fa, + struct inode *inode, struct file *file, bool isdir) +{ + struct fuse_file *ff = file->private_data; + struct fuse_open_out *foo = fa->out_args[0].value; + + ff->fh = foo->fh; + return 0; +} + +int fuse_create_open_initialize( + struct fuse_args *fa, struct fuse_create_open_io *fcoio, + struct inode *dir, struct dentry *entry, + struct file *file, unsigned int flags, umode_t mode) +{ + fcoio->fci = (struct fuse_create_in) { + .flags = file->f_flags & ~(O_CREAT | O_EXCL | O_NOCTTY), + .mode = mode, + }; + + fcoio->feo = (struct fuse_entry_out) {0}; + fcoio->foo = (struct fuse_open_out) {0}; + + *fa = (struct fuse_args) { + .nodeid = get_node_id(dir), + .opcode = FUSE_CREATE, + .in_numargs = 2, + .out_numargs = 2, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(fcoio->fci), + .value = &fcoio->fci, + }, + .in_args[1] = (struct fuse_in_arg) { + .size = entry->d_name.len + 1, + .value = entry->d_name.name, + }, + .out_args[0] = (struct fuse_arg) { + .size = sizeof(fcoio->feo), + .value = &fcoio->feo, + }, + .out_args[1] = (struct fuse_arg) { + .size = sizeof(fcoio->foo), + .value = &fcoio->foo, + }, + }; + + return 0; +} + +static int fuse_open_file_backing(struct inode *inode, struct file *file) +{ + struct fuse_mount *fm = get_fuse_mount(inode); + struct dentry *entry = file->f_path.dentry; + struct fuse_dentry *fuse_dentry = get_fuse_dentry(entry); + struct fuse_file *fuse_file; + struct file *backing_file; + + fuse_file = fuse_file_alloc(fm); + if (!fuse_file) + return -ENOMEM; + file->private_data = fuse_file; + + backing_file = dentry_open(&fuse_dentry->backing_path, file->f_flags, + current_cred()); + if (IS_ERR(backing_file)) { + fuse_file_free(fuse_file); + file->private_data = NULL; + return PTR_ERR(backing_file); + } + fuse_file->backing_file = backing_file; + + return 0; +} + +int fuse_create_open_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, + struct file *file, unsigned int flags, umode_t mode) +{ + struct fuse_inode *dir_fuse_inode = get_fuse_inode(dir); + struct fuse_dentry *dir_fuse_dentry = get_fuse_dentry(entry->d_parent); + struct dentry *backing_dentry = NULL; + struct inode *inode = NULL; + struct dentry *newent; + int err = 0; + const struct fuse_create_in *fci = fa->in_args[0].value; + + if (!dir_fuse_inode || !dir_fuse_dentry) + return -EIO; + + inode_lock_nested(dir_fuse_inode->backing_inode, I_MUTEX_PARENT); + backing_dentry = lookup_one_len(fa->in_args[1].value, + dir_fuse_dentry->backing_path.dentry, + strlen(fa->in_args[1].value)); + inode_unlock(dir_fuse_inode->backing_inode); + + if (IS_ERR(backing_dentry)) + return PTR_ERR(backing_dentry); + + if (d_really_is_positive(backing_dentry)) { + err = -EIO; + goto out; + } + + err = vfs_create(dir_fuse_inode->backing_inode, backing_dentry, + fci->mode, true); + if (err) + goto out; + + if (get_fuse_dentry(entry)->backing_path.dentry) + path_put(&get_fuse_dentry(entry)->backing_path); + get_fuse_dentry(entry)->backing_path = (struct path) { + .mnt = dir_fuse_dentry->backing_path.mnt, + .dentry = backing_dentry, + }; + path_get(&get_fuse_dentry(entry)->backing_path); + + inode = fuse_iget_backing(dir->i_sb, + get_fuse_dentry(entry)->backing_path.dentry->d_inode); + if (IS_ERR(inode)) { + err = PTR_ERR(inode); + goto out; + } + + if (get_fuse_inode(inode)->bpf) + bpf_prog_put(get_fuse_inode(inode)->bpf); + get_fuse_inode(inode)->bpf = dir_fuse_inode->bpf; + if (get_fuse_inode(inode)->bpf) + bpf_prog_inc(dir_fuse_inode->bpf); + + newent = d_splice_alias(inode, entry); + if (IS_ERR(newent)) { + err = PTR_ERR(newent); + goto out; + } + + entry = newent ? newent : entry; + err = finish_open(file, entry, fuse_open_file_backing); + +out: + dput(backing_dentry); + return err; +} + +void *fuse_create_open_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, + struct file *file, unsigned int flags, umode_t mode) +{ + struct fuse_file *ff = file->private_data; + struct fuse_inode *fi = get_fuse_inode(file->f_inode); + struct fuse_entry_out *feo = fa->out_args[0].value; + struct fuse_open_out *foo = fa->out_args[1].value; + + fi->nodeid = feo->nodeid; + ff->fh = foo->fh; + return 0; +} + +int fuse_release_initialize(struct fuse_args *fa, struct fuse_release_in *fri, + struct inode *inode, struct file *file) +{ + struct fuse_file *fuse_file = file->private_data; + + /* Always put backing file whatever bpf/userspace says */ + fput(fuse_file->backing_file); + + *fri = (struct fuse_release_in) { + .fh = ((struct fuse_file *)(file->private_data))->fh, + }; + + *fa = (struct fuse_args) { + .nodeid = get_fuse_inode(inode)->nodeid, + .opcode = FUSE_RELEASE, + .in_numargs = 1, + .in_args[0].size = sizeof(*fri), + .in_args[0].value = fri, + }; + + return 0; +} + +int fuse_releasedir_initialize(struct fuse_args *fa, + struct fuse_release_in *fri, + struct inode *inode, struct file *file) +{ + struct fuse_file *fuse_file = file->private_data; + + /* Always put backing file whatever bpf/userspace says */ + fput(fuse_file->backing_file); + + *fri = (struct fuse_release_in) { + .fh = ((struct fuse_file *)(file->private_data))->fh, + }; + + *fa = (struct fuse_args) { + .nodeid = get_fuse_inode(inode)->nodeid, + .opcode = FUSE_RELEASEDIR, + .in_numargs = 1, + .in_args[0].size = sizeof(*fri), + .in_args[0].value = fri, + }; + + return 0; +} + +int fuse_release_backing(struct fuse_args *fa, + struct inode *inode, struct file *file) +{ + return 0; +} + +void *fuse_release_finalize(struct fuse_args *fa, + struct inode *inode, struct file *file) +{ + fuse_file_free(file->private_data); + return NULL; +} + +int fuse_flush_initialize(struct fuse_args *fa, struct fuse_flush_in *ffi, + struct file *file, fl_owner_t id) +{ + struct fuse_file *fuse_file = file->private_data; + + *ffi = (struct fuse_flush_in) { + .fh = fuse_file->fh, + }; + + *fa = (struct fuse_args) { + .nodeid = get_node_id(file->f_inode), + .opcode = FUSE_FLUSH, + .in_numargs = 1, + .in_args[0].size = sizeof(*ffi), + .in_args[0].value = ffi, + .force = true, + }; + + return 0; +} + +int fuse_flush_backing(struct fuse_args *fa, struct file *file, fl_owner_t id) +{ + struct fuse_file *fuse_file = file->private_data; + struct file *backing_file = fuse_file->backing_file; + + if (backing_file->f_op->flush) + return backing_file->f_op->flush(backing_file, id); + return 0; +} + +void *fuse_flush_finalize(struct fuse_args *fa, struct file *file, fl_owner_t id) +{ + return NULL; +} + +int fuse_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, + struct file *file, loff_t start, loff_t end, int datasync) +{ + struct fuse_file *fuse_file = file->private_data; + + *ffi = (struct fuse_fsync_in) { + .fh = fuse_file->fh, + .fsync_flags = datasync ? FUSE_FSYNC_FDATASYNC : 0, + }; + + *fa = (struct fuse_args) { + .nodeid = get_fuse_inode(file->f_inode)->nodeid, + .opcode = FUSE_FSYNC, + .in_numargs = 1, + .in_args[0].size = sizeof(*ffi), + .in_args[0].value = ffi, + .force = true, + }; + + return 0; +} + +int fuse_fsync_backing(struct fuse_args *fa, + struct file *file, loff_t start, loff_t end, int datasync) +{ + struct fuse_file *fuse_file = file->private_data; + struct file *backing_file = fuse_file->backing_file; + const struct fuse_fsync_in *ffi = fa->in_args[0].value; + int new_datasync = (ffi->fsync_flags & FUSE_FSYNC_FDATASYNC) ? 1 : 0; + + return vfs_fsync(backing_file, new_datasync); +} + +void *fuse_fsync_finalize(struct fuse_args *fa, + struct file *file, loff_t start, loff_t end, int datasync) +{ + return NULL; +} + +int fuse_dir_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, + struct file *file, loff_t start, loff_t end, int datasync) +{ + struct fuse_file *fuse_file = file->private_data; + + *ffi = (struct fuse_fsync_in) { + .fh = fuse_file->fh, + .fsync_flags = datasync ? FUSE_FSYNC_FDATASYNC : 0, + }; + + *fa = (struct fuse_args) { + .nodeid = get_fuse_inode(file->f_inode)->nodeid, + .opcode = FUSE_FSYNCDIR, + .in_numargs = 1, + .in_args[0].size = sizeof(*ffi), + .in_args[0].value = ffi, + .force = true, + }; + + return 0; +} + +int fuse_getxattr_initialize(struct fuse_args *fa, + struct fuse_getxattr_io *fgio, + struct dentry *dentry, const char *name, void *value, + size_t size) +{ + *fgio = (struct fuse_getxattr_io) { + .fgi.size = size, + }; + + *fa = (struct fuse_args) { + .nodeid = get_fuse_inode(dentry->d_inode)->nodeid, + .opcode = FUSE_GETXATTR, + .in_numargs = 2, + .out_numargs = 1, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(fgio->fgi), + .value = &fgio->fgi, + }, + .in_args[1] = (struct fuse_in_arg) { + .size = strlen(name) + 1, + .value = name, + }, + .out_argvar = size ? true : false, + .out_args[0].size = size ? size : sizeof(fgio->fgo), + .out_args[0].value = size ? value : &fgio->fgo, + }; + + return 0; +} + +int fuse_getxattr_backing(struct fuse_args *fa, + struct dentry *dentry, const char *name, void *value, + size_t size) +{ + ssize_t ret = vfs_getxattr(get_fuse_dentry(dentry)->backing_path.dentry, + fa->in_args[1].value, value, size); + + if (fa->out_argvar) + fa->out_args[0].size = ret; + else + ((struct fuse_getxattr_out *)fa->out_args[0].value)->size = ret; + + return 0; +} + +void *fuse_getxattr_finalize(struct fuse_args *fa, + struct dentry *dentry, const char *name, void *value, + size_t size) +{ + struct fuse_getxattr_out *fgo; + + if (fa->out_argvar) + return ERR_PTR(fa->out_args[0].size); + + fgo = fa->out_args[0].value; + + return ERR_PTR(fgo->size); + +} + +int fuse_listxattr_initialize(struct fuse_args *fa, + struct fuse_getxattr_io *fgio, + struct dentry *dentry, char *list, size_t size) +{ + *fgio = (struct fuse_getxattr_io){ + .fgi.size = size, + }; + + *fa = (struct fuse_args){ + .nodeid = get_fuse_inode(dentry->d_inode)->nodeid, + .opcode = FUSE_LISTXATTR, + .in_numargs = 1, + .out_numargs = 1, + .in_args[0] = + (struct fuse_in_arg){ + .size = sizeof(fgio->fgi), + .value = &fgio->fgi, + }, + .out_argvar = size ? true : false, + .out_args[0].size = size ? size : sizeof(fgio->fgo), + .out_args[0].value = size ? (void *)list : &fgio->fgo, + }; + + return 0; +} + +int fuse_listxattr_backing(struct fuse_args *fa, struct dentry *dentry, + char *list, size_t size) +{ + ssize_t ret = + vfs_listxattr(get_fuse_dentry(dentry)->backing_path.dentry, + list, size); + + if (fa->out_argvar) + fa->out_args[0].size = ret; + else + ((struct fuse_getxattr_out *)fa->out_args[0].value)->size = ret; + + return 0; +} + +void *fuse_listxattr_finalize(struct fuse_args *fa, struct dentry *dentry, + char *list, size_t size) +{ + struct fuse_getxattr_out *fgo; + + if (fa->out_argvar) + return ERR_PTR(fa->out_args[0].size); + + fgo = fa->out_args[0].value; + + return ERR_PTR(fgo->size); +} + +int fuse_setxattr_initialize(struct fuse_args *fa, + struct fuse_setxattr_in *fsxi, + struct dentry *dentry, const char *name, + const void *value, size_t size, int flags) +{ + *fsxi = (struct fuse_setxattr_in) { + .size = size, + .flags = flags, + }; + + *fa = (struct fuse_args) { + .nodeid = get_fuse_inode(dentry->d_inode)->nodeid, + .opcode = FUSE_SETXATTR, + .in_numargs = 3, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(*fsxi), + .value = fsxi, + }, + .in_args[1] = (struct fuse_in_arg) { + .size = strlen(name) + 1, + .value = name, + }, + .in_args[2] = (struct fuse_in_arg) { + .size = size, + .value = value, + }, + }; + + return 0; +} + +int fuse_setxattr_backing(struct fuse_args *fa, struct dentry *dentry, + const char *name, const void *value, size_t size, + int flags) +{ + return vfs_setxattr(get_fuse_dentry(dentry)->backing_path.dentry, name, + value, size, flags); +} + +void *fuse_setxattr_finalize(struct fuse_args *fa, struct dentry *dentry, + const char *name, const void *value, size_t size, + int flags) +{ + return NULL; +} + +int fuse_file_read_iter_initialize( + struct fuse_args *fa, struct fuse_read_in *fri, + struct kiocb *iocb, struct iov_iter *to) +{ + struct file *file = iocb->ki_filp; + struct fuse_file *ff = file->private_data; + + *fri = (struct fuse_read_in) { + .fh = ff->fh, + .offset = iocb->ki_pos, + .size = to->count, + }; + + /* TODO we can't assume 'to' is a kvec */ + /* TODO we also can't assume the vector has only one component */ + *fa = (struct fuse_args) { + .opcode = FUSE_READ, + .nodeid = ff->nodeid, + .in_numargs = 1, + .in_args[0].size = sizeof(*fri), + .in_args[0].value = fri, + .out_numargs = 1, + .out_args[0].size = fri->size, + .out_args[0].value = to->kvec->iov_base, + /* + * TODO Design this properly. + * Possible approach: do not pass buf to bpf + * If going to userland, do a deep copy + * For extra credit, do that to/from the vector, rather than + * making an extra copy in the kernel + */ + }; + + return 0; +} + +int fuse_file_read_iter_backing(struct fuse_args *fa, + struct kiocb *iocb, struct iov_iter *to) +{ + struct file *file = iocb->ki_filp; + struct fuse_file *ff = file->private_data; + ssize_t result; + + /* TODO This just plain ignores any change to fuse_read_in */ + result = vfs_iter_read(ff->backing_file, to, &iocb->ki_pos, 0); + + if (result < 0) + return result; + + /* TODO Need to point value at the buffer for post-modification */ + fa->out_args[0].size = result; + return result; +} + +void *fuse_file_read_iter_finalize(struct fuse_args *fa, + struct kiocb *iocb, struct iov_iter *to) +{ + return ERR_PTR(fa->out_args[0].size); +} + +int fuse_file_write_iter_initialize( + struct fuse_args *fa, struct fuse_file_write_iter_io *fwio, + struct kiocb *iocb, struct iov_iter *from) +{ + struct file *file = iocb->ki_filp; + struct fuse_file *ff = file->private_data; + + *fwio = (struct fuse_file_write_iter_io) { + .fwi.fh = ff->fh, + .fwi.offset = iocb->ki_pos, + .fwi.size = from->count, + }; + + /* TODO we can't assume 'from' is a kvec */ + *fa = (struct fuse_args) { + .opcode = FUSE_WRITE, + .nodeid = ff->nodeid, + .in_numargs = 2, + .in_args[0].size = sizeof(fwio->fwi), + .in_args[0].value = &fwio->fwi, + .in_args[1].size = fwio->fwi.size, + .in_args[1].value = from->kvec->iov_base, + .out_numargs = 1, + .out_args[0].size = sizeof(fwio->fwo), + .out_args[0].value = &fwio->fwo, + }; + + return 0; +} + +int fuse_file_write_iter_backing(struct fuse_args *fa, + struct kiocb *iocb, struct iov_iter *from) +{ + struct file *file = iocb->ki_filp; + struct fuse_file *ff = file->private_data; + struct fuse_write_out *fwo = fa->out_args[0].value; + + /* TODO This just plain ignores any change to fuse_write_in */ + fwo->size = vfs_iter_write(ff->backing_file, from, &iocb->ki_pos, 0); + + if (fwo->size < 0) + return fwo->size; + return 0; +} + +void *fuse_file_write_iter_finalize(struct fuse_args *fa, + struct kiocb *iocb, struct iov_iter *from) +{ + struct fuse_write_out *fwo = fa->out_args[0].value; + + return ERR_PTR(fwo->size); +} + +int fuse_file_fallocate_initialize(struct fuse_args *fa, + struct fuse_fallocate_in *ffi, + struct file *file, int mode, loff_t offset, loff_t length) +{ + struct fuse_file *ff = file->private_data; + + *ffi = (struct fuse_fallocate_in) { + .fh = ff->fh, + .offset = offset, + .length = length, + .mode = mode + }; + + *fa = (struct fuse_args) { + .opcode = FUSE_FALLOCATE, + .nodeid = ff->nodeid, + .in_numargs = 1, + .in_args[0].size = sizeof(*ffi), + .in_args[0].value = ffi, + }; + + return 0; +} + +int fuse_file_fallocate_backing(struct fuse_args *fa, + struct file *file, int mode, loff_t offset, loff_t length) +{ + const struct fuse_fallocate_in *ffi = fa->in_args[0].value; + struct fuse_file *ff = file->private_data; + + return vfs_fallocate(ff->backing_file, ffi->mode, ffi->offset, + ffi->length); +} + +void *fuse_file_fallocate_finalize(struct fuse_args *fa, + struct file *file, int mode, loff_t offset, loff_t length) +{ + return NULL; +} + +/******************************************************************************* + * Directory operations after here * + ******************************************************************************/ + +int fuse_lookup_initialize(struct fuse_args *fa, struct fuse_lookup_io *fli, + struct inode *dir, struct dentry *entry, unsigned int flags) +{ + *fa = (struct fuse_args) { + .nodeid = get_fuse_inode(dir)->nodeid, + .opcode = FUSE_LOOKUP, + .in_numargs = 1, + .out_numargs = 2, + .out_argvar = 1, + .in_args[0] = (struct fuse_in_arg) { + .size = entry->d_name.len + 1, + .value = entry->d_name.name, + }, + .out_args[0] = (struct fuse_arg) { + .size = sizeof(fli->feo), + .value = &fli->feo, + }, + .out_args[1] = (struct fuse_arg) { + .size = sizeof(fli->febo), + .value = &fli->febo, + }, + }; + + return 0; +} + +int fuse_lookup_backing(struct fuse_args *fa, struct inode *dir, + struct dentry *entry, unsigned int flags) +{ + struct fuse_dentry *fuse_entry = get_fuse_dentry(entry); + struct fuse_dentry *dir_fuse_entry = get_fuse_dentry(entry->d_parent); + struct dentry *dir_backing_entry = dir_fuse_entry->backing_path.dentry; + struct inode *dir_backing_inode = dir_backing_entry->d_inode; + struct dentry *backing_entry; + + /* TODO this will not handle lookups over mount points */ + inode_lock_nested(dir_backing_inode, I_MUTEX_PARENT); + backing_entry = lookup_one_len(entry->d_name.name, dir_backing_entry, + strlen(entry->d_name.name)); + inode_unlock(dir_backing_inode); + + if (IS_ERR(backing_entry)) + return PTR_ERR(backing_entry); + + fuse_entry->backing_path = (struct path) { + .dentry = backing_entry, + .mnt = dir_fuse_entry->backing_path.mnt, + }; + + mntget(fuse_entry->backing_path.mnt); + return 0; +} + +struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, + struct dentry *entry, unsigned int flags) +{ + struct fuse_dentry *fd; + struct dentry *bd; + struct inode *inode, *backing_inode; + struct fuse_entry_out *feo = fa->out_args[0].value; + struct fuse_entry_bpf_out *febo = fa->out_args[1].value; + + fd = get_fuse_dentry(entry); + if (!fd) + return ERR_PTR(-EIO); + bd = fd->backing_path.dentry; + if (!bd) + return ERR_PTR(-ENOENT); + backing_inode = bd->d_inode; + if (!backing_inode) + return 0; + + inode = fuse_iget_backing(dir->i_sb, backing_inode); + + if (IS_ERR(inode)) + return ERR_PTR(PTR_ERR(inode)); + + /* TODO Make sure this handles invalid handles */ + /* TODO Do we need the same code in revalidate */ + if (get_fuse_inode(inode)->bpf) { + bpf_prog_put(get_fuse_inode(inode)->bpf); + get_fuse_inode(inode)->bpf = NULL; + } + + switch (febo->bpf_action) { + case FUSE_ACTION_KEEP: + get_fuse_inode(inode)->bpf = get_fuse_inode(dir)->bpf; + if (get_fuse_inode(inode)->bpf) + bpf_prog_inc(get_fuse_inode(inode)->bpf); + break; + + case FUSE_ACTION_REMOVE: + get_fuse_inode(inode)->bpf = NULL; + break; + + case FUSE_ACTION_REPLACE: { + struct fuse_conn *fc = get_fuse_mount(dir)->fc; + struct bpf_prog *bpf_prog = fuse_get_bpf_prog(fc, febo->bpf_fd); + + if (IS_ERR(bpf_prog)) + return ERR_PTR(PTR_ERR(bpf_prog)); + + get_fuse_inode(inode)->bpf = bpf_prog; + break; + } + + default: + return ERR_PTR(-EIO); + } + + switch (febo->backing_action) { + case FUSE_ACTION_KEEP: + /* backing inode/path are added in fuse_lookup_backing */ + break; + + case FUSE_ACTION_REMOVE: + iput(get_fuse_inode(inode)->backing_inode); + get_fuse_inode(inode)->backing_inode = NULL; + path_put_init(&get_fuse_dentry(entry)->backing_path); + break; + + case FUSE_ACTION_REPLACE: { + struct fuse_conn *fc; + struct file *backing_file; + + fc = get_fuse_mount(dir)->fc; + backing_file = fuse_fget(fc, febo->backing_fd); + if (!backing_file) + return ERR_PTR(-EIO); + + iput(get_fuse_inode(inode)->backing_inode); + get_fuse_inode(inode)->backing_inode = + backing_file->f_inode; + ihold(get_fuse_inode(inode)->backing_inode); + + path_put(&get_fuse_dentry(entry)->backing_path); + get_fuse_dentry(entry)->backing_path = backing_file->f_path; + path_get(&get_fuse_dentry(entry)->backing_path); + + fput(backing_file); + break; + } + + default: + return ERR_PTR(-EIO); + } + + get_fuse_inode(inode)->nodeid = feo->nodeid; + + return d_splice_alias(inode, entry); +} + +int fuse_revalidate_backing(struct fuse_args *fa, struct inode *dir, + struct dentry *entry, unsigned int flags) +{ + struct fuse_dentry *fuse_dentry = get_fuse_dentry(entry); + struct dentry *backing_entry = fuse_dentry->backing_path.dentry; + + if (unlikely(backing_entry->d_flags & DCACHE_OP_REVALIDATE)) + return backing_entry->d_op->d_revalidate(backing_entry, flags); + return 1; +} + +void *fuse_revalidate_finalize(struct fuse_args *fa, struct inode *dir, + struct dentry *entry, unsigned int flags) +{ + return 0; +} + +int fuse_mknod_initialize( + struct fuse_args *fa, struct fuse_mknod_in *fmi, + struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) +{ + *fmi = (struct fuse_mknod_in) { + .mode = mode, + .rdev = new_encode_dev(rdev), + .umask = current_umask(), + }; + *fa = (struct fuse_args) { + .nodeid = get_node_id(dir), + .opcode = FUSE_MKNOD, + .in_numargs = 2, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(*fmi), + .value = fmi, + }, + .in_args[1] = (struct fuse_in_arg) { + .size = entry->d_name.len + 1, + .value = entry->d_name.name, + }, + }; + + return 0; +} + +int fuse_mknod_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) +{ + int err = 0; + const struct fuse_mknod_in *fmi = fa->in_args[0].value; + struct inode *backing_inode = get_fuse_inode(dir)->backing_inode; + struct path backing_path = {}; + struct inode *inode = NULL; + + //TODO Actually deal with changing the backing entry in mknod + get_fuse_backing_path(entry, &backing_path); + if (!backing_path.dentry) + return -EBADF; + + inode_lock_nested(backing_inode, I_MUTEX_PARENT); + err = vfs_mknod(backing_inode, backing_path.dentry, + fmi->mode & ~fmi->umask, new_decode_dev(fmi->rdev)); + inode_unlock(backing_inode); + if (err) + goto out; + if (d_really_is_negative(backing_path.dentry) || + unlikely(d_unhashed(backing_path.dentry))) { + err = -EINVAL; + /** + * TODO: overlayfs responds to this situation with a + * lookupOneLen. Should we do that too? + */ + goto out; + } + inode = fuse_iget_backing(dir->i_sb, backing_inode); + if (IS_ERR(inode)) { + err = PTR_ERR(inode); + goto out; + } + d_instantiate(entry, inode); +out: + path_put(&backing_path); + return err; +} + +void *fuse_mknod_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) +{ + return NULL; +} + +int fuse_mkdir_initialize( + struct fuse_args *fa, struct fuse_mkdir_in *fmi, + struct inode *dir, struct dentry *entry, umode_t mode) +{ + *fmi = (struct fuse_mkdir_in) { + .mode = mode, + .umask = current_umask(), + }; + *fa = (struct fuse_args) { + .nodeid = get_node_id(dir), + .opcode = FUSE_MKDIR, + .in_numargs = 2, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(*fmi), + .value = fmi, + }, + .in_args[1] = (struct fuse_in_arg) { + .size = entry->d_name.len + 1, + .value = entry->d_name.name, + }, + }; + + return 0; +} + +int fuse_mkdir_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, umode_t mode) +{ + int err = 0; + const struct fuse_mkdir_in *fmi = fa->in_args[0].value; + struct inode *backing_inode = get_fuse_inode(dir)->backing_inode; + struct path backing_path = {}; + struct inode *inode = NULL; + + //TODO Actually deal with changing the backing entry in mkdir + get_fuse_backing_path(entry, &backing_path); + if (!backing_path.dentry) + return -EBADF; + + inode_lock_nested(backing_inode, I_MUTEX_PARENT); + err = vfs_mkdir(backing_inode, backing_path.dentry, fmi->mode & ~fmi->umask); + inode_unlock(backing_inode); + if (err) + goto out; + if (d_really_is_negative(backing_path.dentry) || + unlikely(d_unhashed(backing_path.dentry))) { + err = -EINVAL; + /** + * TODO: overlayfs responds to this situation with a + * lookupOneLen. Should we do that too? + */ + goto out; + } + inode = fuse_iget_backing(dir->i_sb, backing_inode); + if (IS_ERR(inode)) { + err = PTR_ERR(inode); + goto out; + } + d_instantiate(entry, inode); +out: + path_put(&backing_path); + return err; +} + +void *fuse_mkdir_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, umode_t mode) +{ + return NULL; +} + +int fuse_rmdir_initialize( + struct fuse_args *fa, struct fuse_dummy_io *dummy, + struct inode *dir, struct dentry *entry) +{ + *fa = (struct fuse_args) { + .nodeid = get_node_id(dir), + .opcode = FUSE_RMDIR, + .in_numargs = 1, + .in_args[0] = (struct fuse_in_arg) { + .size = entry->d_name.len + 1, + .value = entry->d_name.name, + }, + }; + + return 0; +} + +int fuse_rmdir_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry) +{ + int err = 0; + struct path backing_path = {}; + struct dentry *backing_parent_dentry; + struct inode *backing_inode; + + /* TODO Actually deal with changing the backing entry in rmdir */ + get_fuse_backing_path(entry, &backing_path); + if (!backing_path.dentry) + return -EBADF; + + /* TODO Not sure if we should reverify like overlayfs, or get inode from d_parent */ + backing_parent_dentry = dget_parent(backing_path.dentry); + backing_inode = d_inode(backing_parent_dentry); + + inode_lock_nested(backing_inode, I_MUTEX_PARENT); + err = vfs_rmdir(backing_inode, backing_path.dentry); + inode_unlock(backing_inode); + + dput(backing_parent_dentry); + if (!err) + d_drop(entry); + path_put(&backing_path); + return err; +} + +void *fuse_rmdir_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry) +{ + return NULL; +} + +static int fuse_rename_backing_common( + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent, + unsigned int flags) +{ + int err = 0; + struct path old_backing_path; + struct path new_backing_path; + struct dentry *old_backing_dir_dentry; + struct dentry *old_backing_dentry; + struct dentry *new_backing_dir_dentry; + struct dentry *new_backing_dentry; + struct dentry *trap = NULL; + struct inode *target_inode; + + //TODO Actually deal with changing anything that isn't a flag + get_fuse_backing_path(oldent, &old_backing_path); + if (!old_backing_path.dentry) + return -EBADF; + get_fuse_backing_path(newent, &new_backing_path); + if (!new_backing_path.dentry) { + err = -EBADF; + goto put_old_path; + } + if (new_backing_path.mnt != old_backing_path.mnt) { + err = -EXDEV; + goto put_new_path; + } + old_backing_dentry = old_backing_path.dentry; + new_backing_dentry = new_backing_path.dentry; + old_backing_dir_dentry = dget_parent(old_backing_dentry); + new_backing_dir_dentry = dget_parent(new_backing_dentry); + target_inode = d_inode(newent); + + trap = lock_rename(old_backing_dir_dentry, new_backing_dir_dentry); + if (trap == old_backing_dentry) { + err = -EINVAL; + goto put_parents; + } + if (trap == new_backing_dentry) { + err = -ENOTEMPTY; + goto put_parents; + } + err = vfs_rename(d_inode(old_backing_dir_dentry), old_backing_dentry, + d_inode(new_backing_dir_dentry), new_backing_dentry, + NULL, flags); + if (err) + goto unlock; + if (target_inode) + fsstack_copy_attr_all(target_inode, + get_fuse_inode(target_inode)->backing_inode); + fsstack_copy_attr_all(newdir, d_inode(new_backing_dir_dentry)); +unlock: + unlock_rename(old_backing_dir_dentry, new_backing_dir_dentry); +put_parents: + dput(new_backing_dir_dentry); + dput(old_backing_dir_dentry); +put_new_path: + path_put(&new_backing_path); +put_old_path: + path_put(&old_backing_path); + return err; +} + +int fuse_rename2_initialize(struct fuse_args *fa, struct fuse_rename2_in *fri, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent, + unsigned int flags) +{ + *fri = (struct fuse_rename2_in) { + .newdir = get_node_id(newdir), + .flags = flags, + }; + *fa = (struct fuse_args) { + .nodeid = get_node_id(olddir), + .opcode = FUSE_RENAME2, + .in_numargs = 3, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(*fri), + .value = fri, + }, + .in_args[1] = (struct fuse_in_arg) { + .size = oldent->d_name.len + 1, + .value = oldent->d_name.name, + }, + .in_args[2] = (struct fuse_in_arg) { + .size = newent->d_name.len + 1, + .value = newent->d_name.name, + }, + }; + + return 0; +} + +int fuse_rename2_backing(struct fuse_args *fa, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent, + unsigned int flags) +{ + const struct fuse_rename2_in *fri = fa->in_args[0].value; + + /* TODO: deal with changing dirs/ents */ + return fuse_rename_backing_common(olddir, oldent, newdir, newent, fri->flags); +} + +void *fuse_rename2_finalize(struct fuse_args *fa, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent, + unsigned int flags) +{ + return NULL; +} + +int fuse_rename_initialize(struct fuse_args *fa, struct fuse_rename_in *fri, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent) +{ + *fri = (struct fuse_rename_in) { + .newdir = get_node_id(newdir), + }; + *fa = (struct fuse_args) { + .nodeid = get_node_id(olddir), + .opcode = FUSE_RENAME, + .in_numargs = 3, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(*fri), + .value = fri, + }, + .in_args[1] = (struct fuse_in_arg) { + .size = oldent->d_name.len + 1, + .value = oldent->d_name.name, + }, + .in_args[2] = (struct fuse_in_arg) { + .size = newent->d_name.len + 1, + .value = newent->d_name.name, + }, + }; + + return 0; +} + +int fuse_rename_backing(struct fuse_args *fa, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent) +{ + /* TODO: deal with changing dirs/ents */ + return fuse_rename_backing_common(olddir, oldent, newdir, newent, 0); +} + +void *fuse_rename_finalize(struct fuse_args *fa, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent) +{ + return NULL; +} + +int fuse_unlink_initialize( + struct fuse_args *fa, struct fuse_dummy_io *dummy, + struct inode *dir, struct dentry *entry) +{ + *fa = (struct fuse_args) { + .nodeid = get_node_id(dir), + .opcode = FUSE_UNLINK, + .in_numargs = 1, + .in_args[0] = (struct fuse_in_arg) { + .size = entry->d_name.len + 1, + .value = entry->d_name.name, + }, + }; + + return 0; +} + +int fuse_unlink_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry) +{ + int err = 0; + struct path backing_path = {}; + struct dentry *backing_parent_dentry; + struct inode *backing_inode; + + /* TODO Actually deal with changing the backing entry in unlink */ + get_fuse_backing_path(entry, &backing_path); + if (!backing_path.dentry) + return -EBADF; + + /* TODO Not sure if we should reverify like overlayfs, or get inode from d_parent */ + backing_parent_dentry = dget_parent(backing_path.dentry); + backing_inode = d_inode(backing_parent_dentry); + + inode_lock_nested(backing_inode, I_MUTEX_PARENT); + err = vfs_unlink(backing_inode, backing_path.dentry, NULL); + inode_unlock(backing_inode); + + dput(backing_parent_dentry); + if (!err) + d_drop(entry); + path_put(&backing_path); + return err; +} + +void *fuse_unlink_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry) +{ + return NULL; +} + +int fuse_link_initialize(struct fuse_args *fa, struct fuse_link_in *fli, + struct dentry *entry, struct inode *dir, + struct dentry *newent) +{ + struct inode *src_inode = entry->d_inode; + + *fli = (struct fuse_link_in){ + .oldnodeid = get_node_id(src_inode), + }; + + fa->opcode = FUSE_LINK; + fa->in_numargs = 2; + fa->in_args[0].size = sizeof(*fli); + fa->in_args[0].value = fli; + fa->in_args[1].size = newent->d_name.len + 1; + fa->in_args[1].value = newent->d_name.name; + + return 0; +} + +int fuse_link_backing(struct fuse_args *fa, struct dentry *entry, + struct inode *dir, struct dentry *newent) +{ + int err = 0; + struct path backing_old_path = {}; + struct path backing_new_path = {}; + struct dentry *backing_dir_dentry; + struct inode *fuse_new_inode = NULL; + struct inode *backing_dir_inode = get_fuse_inode(dir)->backing_inode; + + get_fuse_backing_path(entry, &backing_old_path); + if (!backing_old_path.dentry) + return -EBADF; + + get_fuse_backing_path(newent, &backing_new_path); + if (!backing_new_path.dentry) { + err = -EBADF; + goto err_dst_path; + } + + backing_dir_dentry = dget_parent(backing_new_path.dentry); + backing_dir_inode = d_inode(backing_dir_dentry); + + inode_lock_nested(backing_dir_inode, I_MUTEX_PARENT); + err = vfs_link(backing_old_path.dentry, backing_dir_inode, backing_new_path.dentry, NULL); + inode_unlock(backing_dir_inode); + if (err) + goto out; + + if (d_really_is_negative(backing_new_path.dentry) || + unlikely(d_unhashed(backing_new_path.dentry))) { + err = -EINVAL; + /** + * TODO: overlayfs responds to this situation with a + * lookupOneLen. Should we do that too? + */ + goto out; + } + + fuse_new_inode = fuse_iget_backing(dir->i_sb, backing_dir_inode); + if (IS_ERR(fuse_new_inode)) { + err = PTR_ERR(fuse_new_inode); + goto out; + } + d_instantiate(newent, fuse_new_inode); + +out: + dput(backing_dir_dentry); + path_put(&backing_new_path); +err_dst_path: + path_put(&backing_old_path); + return err; +} + +void *fuse_link_finalize(struct fuse_args *fa, struct dentry *entry, + struct inode *dir, struct dentry *newent) +{ + return NULL; +} + +int fuse_getattr_initialize(struct fuse_args *fa, struct fuse_getattr_io *fgio, + const struct dentry *entry, struct kstat *stat, + u32 request_mask, unsigned int flags) +{ + fgio->fgi = (struct fuse_getattr_in) { + .getattr_flags = flags, + .fh = -1, /* TODO is this OK? */ + }; + + fgio->fao = (struct fuse_attr_out) {0}; + + *fa = (struct fuse_args) { + .nodeid = get_node_id(entry->d_inode), + .opcode = FUSE_GETATTR, + .in_numargs = 1, + .out_numargs = 1, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(fgio->fgi), + .value = &fgio->fgi, + }, + .out_args[0] = (struct fuse_arg) { + .size = sizeof(fgio->fao), + .value = &fgio->fao, + }, + }; + + return 0; +} + +static void fuse_stat_to_attr(struct fuse_conn *fc, struct inode *inode, + struct kstat *stat, struct fuse_attr *attr) +{ + unsigned int blkbits; + + /* see the comment in fuse_change_attributes() */ + if (fc->writeback_cache && S_ISREG(inode->i_mode)) { + stat->size = i_size_read(inode); + stat->mtime.tv_sec = inode->i_mtime.tv_sec; + stat->mtime.tv_nsec = inode->i_mtime.tv_nsec; + stat->ctime.tv_sec = inode->i_ctime.tv_sec; + stat->ctime.tv_nsec = inode->i_ctime.tv_nsec; + } + + attr->ino = stat->ino; + attr->mode = (inode->i_mode & S_IFMT) | (stat->mode & 07777); + attr->nlink = stat->nlink; + attr->uid = from_kuid(fc->user_ns, stat->uid); + attr->gid = from_kgid(fc->user_ns, stat->gid); + attr->atime = stat->atime.tv_sec; + attr->atimensec = stat->atime.tv_nsec; + attr->mtime = stat->mtime.tv_sec; + attr->mtimensec = stat->mtime.tv_nsec; + attr->ctime = stat->ctime.tv_sec; + attr->ctimensec = stat->ctime.tv_nsec; + attr->size = stat->size; + attr->blocks = stat->blocks; + + if (stat->blksize != 0) + blkbits = ilog2(stat->blksize); + else + blkbits = inode->i_sb->s_blocksize_bits; + + attr->blksize = 1 << blkbits; +} + +int fuse_getattr_backing(struct fuse_args *fa, + const struct dentry *entry, struct kstat *stat, + u32 request_mask, unsigned int flags) +{ + struct path *backing_path = + &get_fuse_dentry(entry)->backing_path; + struct inode *backing_inode = backing_path->dentry->d_inode; + struct fuse_attr_out *fao = fa->out_args[0].value; + struct kstat tmp; + int err; + + if (!stat) + stat = &tmp; + + err = vfs_getattr(backing_path, stat, request_mask, flags); + + if (!err) + fuse_stat_to_attr(get_fuse_conn(entry->d_inode), + backing_inode, stat, &fao->attr); + + return err; +} + +void *fuse_getattr_finalize(struct fuse_args *fa, + const struct dentry *entry, struct kstat *stat, + u32 request_mask, unsigned int flags) +{ + struct fuse_attr_out *outarg = fa->out_args[0].value; + struct inode *inode = entry->d_inode; + u64 attr_version = fuse_get_attr_version(get_fuse_mount(inode)->fc); + int err = 0; + + /* TODO: Ensure this doesn't happen if we had an error getting attrs in + * backing. + */ + err = finalize_attr(inode, outarg, attr_version, stat); + return ERR_PTR(err); +} + +static void fattr_to_iattr(const struct fuse_setattr_in *arg, + struct iattr *iattr) +{ + unsigned int ivalid = arg->valid; + + if (ivalid & ATTR_MODE) + iattr->ia_valid |= FATTR_MODE, iattr->ia_mode = arg->mode; + if (ivalid & ATTR_UID) { + iattr->ia_valid |= FATTR_UID; + iattr->ia_uid = KUIDT_INIT(arg->uid); + } + if (ivalid & ATTR_GID) { + iattr->ia_valid |= FATTR_GID; + iattr->ia_gid = KGIDT_INIT(arg->gid); + } + if (ivalid & ATTR_SIZE) + iattr->ia_valid |= FATTR_SIZE, iattr->ia_size = arg->size; + if (ivalid & ATTR_ATIME) { + iattr->ia_valid |= FATTR_ATIME; + iattr->ia_atime.tv_sec = arg->atime; + iattr->ia_atime.tv_nsec = arg->atimensec; + if (!(ivalid & ATTR_ATIME_SET)) + iattr->ia_valid |= FATTR_ATIME_NOW; + } + if (ivalid & ATTR_MTIME) { + iattr->ia_valid |= FATTR_MTIME; + iattr->ia_mtime.tv_sec = arg->mtime; + iattr->ia_mtime.tv_nsec = arg->mtimensec; + } + if (ivalid & ATTR_CTIME) { + iattr->ia_valid |= FATTR_CTIME; + iattr->ia_ctime.tv_sec = arg->ctime; + iattr->ia_ctime.tv_nsec = arg->ctimensec; + } +} + +int fuse_setattr_initialize(struct fuse_args *fa, struct fuse_setattr_io *fsio, + struct dentry *dentry, struct iattr *attr, struct file *file) +{ + struct fuse_conn *fc = get_fuse_conn(dentry->d_inode); + + *fsio = (struct fuse_setattr_io) {0}; + iattr_to_fattr(fc, attr, &fsio->fsi, true); + + *fa = (struct fuse_args) { + .opcode = FUSE_SETATTR, + .nodeid = get_node_id(dentry->d_inode), + .in_numargs = 1, + .in_args[0].size = sizeof(fsio->fsi), + .in_args[0].value = &fsio->fsi, + .out_numargs = 1, + .out_args[0].size = sizeof(fsio->fao), + .out_args[0].value = &fsio->fao, + }; + + return 0; +} + +int fuse_setattr_backing(struct fuse_args *fa, + struct dentry *dentry, struct iattr *attr, struct file *file) +{ + const struct fuse_setattr_in *fsi = fa->in_args[0].value; + struct iattr new_attr = {0}; + struct path *backing_path = &get_fuse_dentry(dentry)->backing_path; + int res; + + fattr_to_iattr(fsi, &new_attr); + inode_lock(d_inode(backing_path->dentry)); + res = notify_change(backing_path->dentry, &new_attr, NULL); + inode_unlock(d_inode(backing_path->dentry)); + return res; +} + +void *fuse_setattr_finalize(struct fuse_args *fa, + struct dentry *dentry, struct iattr *attr, struct file *file) +{ + return NULL; +} + +int fuse_get_link_initialize(struct fuse_args *fa, struct fuse_dummy_io *unused, + struct inode *inode, struct dentry *dentry, + struct delayed_call *callback, const char **out) +{ + /* + * TODO + * If we want to handle changing these things, we'll need to copy + * the lower fs's data into our own buffer, and provide our own callback + * to free that buffer. + * + * Pre could change the name we're looking at + * postfilter can change the name we return + * + * We ought to only make that buffer if it's been requested, so leaving + * this unimplemented for the moment + */ + *fa = (struct fuse_args) { + .opcode = FUSE_READLINK, + .nodeid = get_node_id(inode), + .in_numargs = 1, + .in_args[0] = (struct fuse_in_arg) { + .size = dentry->d_name.len + 1, + .value = dentry->d_name.name, + }, + /* + * .out_argvar = 1, + * .out_numargs = 1, + * .out_args[0].size = , + * .out_args[0].value = , + */ + }; + + return 0; +} + +int fuse_get_link_backing(struct fuse_args *fa, + struct inode *inode, struct dentry *dentry, + struct delayed_call *callback, const char **out) +{ + struct path backing_path; + + if (!dentry) { + *out = ERR_PTR(-ECHILD); + return PTR_ERR(*out); + } + + get_fuse_backing_path(dentry, &backing_path); + if (!backing_path.dentry) { + *out = ERR_PTR(-ECHILD); + return PTR_ERR(*out); + } + + /* + * TODO: If we want to do our own thing, copy the data and then call the + * callback + */ + *out = vfs_get_link(backing_path.dentry, callback); + + path_put(&backing_path); + return 0; +} + +void *fuse_get_link_finalize(struct fuse_args *fa, + struct inode *inode, struct dentry *dentry, + struct delayed_call *callback, const char **out) +{ + return NULL; +} + +int fuse_symlink_initialize( + struct fuse_args *fa, struct fuse_dummy_io *unused, + struct inode *dir, struct dentry *entry, const char *link, int len) +{ + *fa = (struct fuse_args) { + .nodeid = get_node_id(dir), + .opcode = FUSE_SYMLINK, + .in_numargs = 2, + .in_args[0] = (struct fuse_in_arg) { + .size = entry->d_name.len + 1, + .value = entry->d_name.name, + }, + .in_args[1] = (struct fuse_in_arg) { + .size = len, + .value = link, + }, + }; + + return 0; +} + +int fuse_symlink_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, const char *link, int len) +{ + int err = 0; + struct inode *backing_inode = get_fuse_inode(dir)->backing_inode; + struct path backing_path = {}; + struct inode *inode = NULL; + + //TODO Actually deal with changing the backing entry in symlink + get_fuse_backing_path(entry, &backing_path); + if (!backing_path.dentry) + return -EBADF; + + inode_lock_nested(backing_inode, I_MUTEX_PARENT); + err = vfs_symlink(backing_inode, backing_path.dentry, link); + inode_unlock(backing_inode); + if (err) + goto out; + if (d_really_is_negative(backing_path.dentry) || + unlikely(d_unhashed(backing_path.dentry))) { + err = -EINVAL; + /** + * TODO: overlayfs responds to this situation with a + * lookupOneLen. Should we do that too? + */ + goto out; + } + inode = fuse_iget_backing(dir->i_sb, backing_inode); + if (IS_ERR(inode)) { + err = PTR_ERR(inode); + goto out; + } + d_instantiate(entry, inode); +out: + path_put(&backing_path); + return err; +} + +void *fuse_symlink_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, const char *link, int len) +{ + return NULL; +} + +int fuse_readdir_initialize(struct fuse_args *fa, struct fuse_read_io *frio, + struct file *file, struct dir_context *ctx, + bool *force_again, bool *allow_force) +{ + struct fuse_file *ff = file->private_data; + u8 *page = (u8 *)__get_free_page(GFP_KERNEL); + + if (!page) + return -ENOMEM; + + *fa = (struct fuse_args) { + .nodeid = ff->nodeid, + .opcode = FUSE_READDIR, + .in_numargs = 1, + .out_argvar = true, + .out_numargs = 2, + .in_args[0] = (struct fuse_in_arg) { + .size = sizeof(frio->fri), + .value = &frio->fri, + }, + .out_args[0] = (struct fuse_arg) { + .size = sizeof(frio->fro), + .value = &frio->fro, + }, + .out_args[1] = (struct fuse_arg) { + .size = PAGE_SIZE, + .value = page, + }, + }; + + frio->fri = (struct fuse_read_in) { + .fh = ff->fh, + .offset = ctx->pos, + .size = PAGE_SIZE, + }; + frio->fro = (struct fuse_read_out) { + .again = 0, + .offset = 0, + }; + *force_again = false; + *allow_force = true; + return 0; +} + +struct extfuse_ctx { + struct dir_context ctx; + u8 *addr; + size_t offset; +}; + +static int filldir(struct dir_context *ctx, const char *name, int namelen, + loff_t offset, u64 ino, unsigned int d_type) +{ + struct extfuse_ctx *ec = container_of(ctx, struct extfuse_ctx, ctx); + struct fuse_dirent *fd = (struct fuse_dirent *) (ec->addr + ec->offset); + + if (ec->offset + sizeof(struct fuse_dirent) + namelen > PAGE_SIZE) + return -ENOMEM; + + *fd = (struct fuse_dirent) { + .ino = ino, + .off = offset, + .namelen = namelen, + .type = d_type, + }; + + strcpy(fd->name, name); + ec->offset += FUSE_DIRENT_SIZE(fd); + + return 0; +} + +int fuse_readdir_backing(struct fuse_args *fa, + struct file *file, struct dir_context *ctx, + bool *force_again, bool *allow_force) +{ + struct fuse_file *ff = file->private_data; + struct file *backing_dir = ff->backing_file; + struct fuse_read_out *fro = fa->out_args[0].value; + struct extfuse_ctx ec; + int err; + + ec = (struct extfuse_ctx) { + .ctx.actor = filldir, + .ctx.pos = ctx->pos, + .addr = fa->out_args[1].value, + }; + + if (!ec.addr) + return -ENOMEM; + + err = iterate_dir(backing_dir, &ec.ctx); + if (ec.offset == 0) + *allow_force = false; + fa->out_args[1].size = ec.offset; + + fro->offset = ec.ctx.pos; + fro->again = false; + return err; +} + +void *fuse_readdir_finalize(struct fuse_args *fa, + struct file *file, struct dir_context *ctx, + bool *force_again, bool *allow_force) +{ + int err = 0; + struct fuse_file *ff = file->private_data; + struct file *backing_dir = ff->backing_file; + struct fuse_read_out *fro = fa->out_args[0].value; + + err = fuse_parse_dirfile(fa->out_args[1].value, + fa->out_args[1].size, file, ctx); + *force_again = !!fro->again; + if (*force_again && !*allow_force) + err = -EINVAL; + backing_dir->f_pos = fro->offset; + + free_page((unsigned long) fa->out_args[1].value); + return ERR_PTR(err); +} + +int fuse_access_initialize(struct fuse_args *fa, struct fuse_access_in *fai, + struct inode *inode, int mask) +{ + *fai = (struct fuse_access_in) { + .mask = mask, + }; + + *fa = (struct fuse_args) { + .opcode = FUSE_ACCESS, + .nodeid = get_node_id(inode), + .in_numargs = 1, + .in_args[0].size = sizeof(*fai), + .in_args[0].value = fai, + }; + + return 0; +} + +int fuse_access_backing(struct fuse_args *fa, struct inode *inode, int mask) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + const struct fuse_access_in *fai = fa->in_args[0].value; + + return inode_permission(/* For mainline: init_user_ns,*/ + fi->backing_inode, fai->mask); +} + +void *fuse_access_finalize(struct fuse_args *fa, struct inode *inode, int mask) +{ + return NULL; +} diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index cea02c9258f1..2dc094bfe3e4 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -243,6 +243,11 @@ void fuse_queue_forget(struct fuse_conn *fc, struct fuse_forget_link *forget, { struct fuse_iqueue *fiq = &fc->iq; + if (nodeid == 0) { + kfree(forget); + return; + } + forget->forget_one.nodeid = nodeid; forget->forget_one.nlookup = nlookup; @@ -486,6 +491,7 @@ static void fuse_args_to_req(struct fuse_req *req, struct fuse_args *args) { req->in.h.opcode = args->opcode; req->in.h.nodeid = args->nodeid; + req->in.h.error_in = args->error_in; req->args = args; if (args->end) __set_bit(FR_ASYNC, &req->flags); diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 82a99eb6125a..df8381ff0738 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -8,8 +8,10 @@ #include "fuse_i.h" +#include #include #include +#include #include #include #include @@ -18,6 +20,8 @@ #include #include +#include "../internal.h" + static void fuse_advise_use_readdirplus(struct inode *dir) { struct fuse_inode *fi = get_fuse_inode(dir); @@ -25,7 +29,7 @@ static void fuse_advise_use_readdirplus(struct inode *dir) set_bit(FUSE_I_ADVISE_RDPLUS, &fi->state); } -#if BITS_PER_LONG >= 64 +#if BITS_PER_LONG >= 64 && !defined(CONFIG_FUSE_BPF) static inline void __fuse_dentry_settime(struct dentry *entry, u64 time) { entry->d_fsdata = (void *) time; @@ -37,19 +41,15 @@ static inline u64 fuse_dentry_time(const struct dentry *entry) } #else -union fuse_dentry { - u64 time; - struct rcu_head rcu; -}; static inline void __fuse_dentry_settime(struct dentry *dentry, u64 time) { - ((union fuse_dentry *) dentry->d_fsdata)->time = time; + ((struct fuse_dentry *) dentry->d_fsdata)->time = time; } static inline u64 fuse_dentry_time(const struct dentry *entry) { - return ((union fuse_dentry *) entry->d_fsdata)->time; + return ((struct fuse_dentry *) entry->d_fsdata)->time; } #endif @@ -74,6 +74,18 @@ static void fuse_dentry_settime(struct dentry *dentry, u64 time) __fuse_dentry_settime(dentry, time); } +void fuse_init_dentry_root(struct dentry *root, struct file *backing_dir) +{ +#ifdef CONFIG_FUSE_BPF + struct fuse_dentry *fuse_dentry = root->d_fsdata; + + if (backing_dir) { + fuse_dentry->backing_path = backing_dir->f_path; + path_get(&fuse_dentry->backing_path); + } +#endif +} + /* * Set dentry and possibly attribute timeouts from the lookup/mk* * replies @@ -144,7 +156,8 @@ static void fuse_invalidate_entry(struct dentry *entry) static void fuse_lookup_init(struct fuse_conn *fc, struct fuse_args *args, u64 nodeid, const struct qstr *name, - struct fuse_entry_out *outarg) + struct fuse_entry_out *outarg, + struct fuse_entry_bpf_out *bpf_outarg) { memset(outarg, 0, sizeof(struct fuse_entry_out)); args->opcode = FUSE_LOOKUP; @@ -152,9 +165,12 @@ static void fuse_lookup_init(struct fuse_conn *fc, struct fuse_args *args, args->in_numargs = 1; args->in_args[0].size = name->len + 1; args->in_args[0].value = name->name; - args->out_numargs = 1; + args->out_argvar = true; + args->out_numargs = 2; args->out_args[0].size = sizeof(struct fuse_entry_out); args->out_args[0].value = outarg; + args->out_args[1].size = sizeof(struct fuse_entry_bpf_out); + args->out_args[1].value = bpf_outarg; } /* @@ -180,6 +196,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) else if (time_before64(fuse_dentry_time(entry), get_jiffies_64()) || (flags & LOOKUP_REVAL)) { struct fuse_entry_out outarg; + struct fuse_entry_bpf_out bpf_outarg; FUSE_ARGS(args); struct fuse_forget_link *forget; u64 attr_version; @@ -191,7 +208,20 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) ret = -ECHILD; if (flags & LOOKUP_RCU) goto out; +#ifdef CONFIG_FUSE_BPF + { + struct fuse_err_ret fer; + fer = fuse_bpf_backing(entry->d_parent->d_inode, + struct fuse_lookup_io, + fuse_lookup_initialize, + fuse_revalidate_backing, + fuse_revalidate_finalize, + d_inode(entry->d_parent), entry, flags); + if (fer.ret) + return PTR_ERR(fer.result); + } +#endif fm = get_fuse_mount(inode); forget = fuse_alloc_forget(); @@ -202,8 +232,9 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) attr_version = fuse_get_attr_version(fm->fc); parent = dget_parent(entry); + fuse_lookup_init(fm->fc, &args, get_node_id(d_inode(parent)), - &entry->d_name, &outarg); + &entry->d_name, &outarg, &bpf_outarg); ret = fuse_simple_request(fm, &args); dput(parent); /* Zero nodeid is same as -ENOENT */ @@ -253,17 +284,20 @@ invalid: goto out; } -#if BITS_PER_LONG < 64 +#if BITS_PER_LONG < 64 || defined(CONFIG_FUSE_BPF) static int fuse_dentry_init(struct dentry *dentry) { - dentry->d_fsdata = kzalloc(sizeof(union fuse_dentry), + dentry->d_fsdata = kzalloc(sizeof(struct fuse_dentry), GFP_KERNEL_ACCOUNT | __GFP_RECLAIMABLE); return dentry->d_fsdata ? 0 : -ENOMEM; } static void fuse_dentry_release(struct dentry *dentry) { - union fuse_dentry *fd = dentry->d_fsdata; + struct fuse_dentry *fd = dentry->d_fsdata; + + if (fd && fd->backing_path.dentry) + path_put(&fd->backing_path); kfree_rcu(fd, rcu); } @@ -402,7 +436,7 @@ out: const struct dentry_operations fuse_dentry_operations = { .d_revalidate = fuse_dentry_revalidate, .d_delete = fuse_dentry_delete, -#if BITS_PER_LONG < 64 +#if BITS_PER_LONG < 64 || defined(CONFIG_FUSE_BPF) .d_init = fuse_dentry_init, .d_release = fuse_dentry_release, #endif @@ -411,7 +445,7 @@ const struct dentry_operations fuse_dentry_operations = { }; const struct dentry_operations fuse_root_dentry_operations = { -#if BITS_PER_LONG < 64 +#if BITS_PER_LONG < 64 || defined(CONFIG_FUSE_BPF) .d_init = fuse_dentry_init, .d_release = fuse_dentry_release, #endif @@ -430,7 +464,10 @@ bool fuse_invalid_attr(struct fuse_attr *attr) } int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name, - struct fuse_entry_out *outarg, struct inode **inode) + struct fuse_entry_out *outarg, + struct fuse_entry_bpf_out *bpf_outarg, + struct dentry *entry, + struct inode **inode) { struct fuse_mount *fm = get_fuse_mount_super(sb); FUSE_ARGS(args); @@ -451,23 +488,82 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name attr_version = fuse_get_attr_version(fm->fc); - fuse_lookup_init(fm->fc, &args, nodeid, name, outarg); + fuse_lookup_init(fm->fc, &args, nodeid, name, outarg, bpf_outarg); err = fuse_simple_request(fm, &args); - /* Zero nodeid is same as -ENOENT, but with valid timeout */ - if (err || !outarg->nodeid) - goto out_put_forget; - err = -EIO; - if (!outarg->nodeid) - goto out_put_forget; - if (fuse_invalid_attr(&outarg->attr)) - goto out_put_forget; +#ifdef CONFIG_FUSE_BPF + if (err == sizeof(*bpf_outarg)) { + /* TODO Make sure this handles invalid handles */ + /* TODO Do we need the same code in revalidate */ + struct file *backing_file; + struct inode *backing_inode; + + err = -ENOENT; + if (!entry) + goto out_queue_forget; + + err = -EINVAL; + if (bpf_outarg->backing_action != FUSE_ACTION_REPLACE) + goto out_queue_forget; + + backing_file = fuse_fget(fm->fc, bpf_outarg->backing_fd); + if (!backing_file) + goto out_queue_forget; + + /* TODO userspace doesn't really know when the right time to + * close the passed fd is. This because after replying to the + * driver request, so assume that after a lookup with bpf_args, + * the daemon passes the fd ownership to the kernel, which also + * takes care of closing it at the right time. + */ + __close_fd(fm->fc->task->files, bpf_outarg->backing_fd); + + backing_inode = backing_file->f_inode; + *inode = fuse_iget_backing(sb, backing_inode); + if (!*inode) + goto bpf_outarg_out; + + if (bpf_outarg->bpf_action == FUSE_ACTION_REPLACE) { + struct bpf_prog *bpf_prog = fuse_get_bpf_prog(fm->fc, + bpf_outarg->bpf_fd); + + if (IS_ERR(bpf_prog)) { + iput(*inode); + *inode = NULL; + err = PTR_ERR(bpf_prog); + goto bpf_outarg_out; + } + get_fuse_inode(*inode)->bpf = bpf_prog; + } + + get_fuse_dentry(entry)->backing_path = backing_file->f_path; + path_get(&get_fuse_dentry(entry)->backing_path); + +bpf_outarg_out: + fput(backing_file); + } else +#endif + { + /* Zero nodeid is same as -ENOENT, but with valid timeout */ + if (err || !outarg->nodeid) + goto out_put_forget; + + err = -EIO; + if (!outarg->nodeid) + goto out_put_forget; + if (fuse_invalid_attr(&outarg->attr)) + goto out_put_forget; + + *inode = fuse_iget(sb, outarg->nodeid, outarg->generation, + &outarg->attr, entry_attr_timeout(outarg), + attr_version); + } - *inode = fuse_iget(sb, outarg->nodeid, outarg->generation, - &outarg->attr, entry_attr_timeout(outarg), - attr_version); err = -ENOMEM; - if (!*inode) { +#ifdef CONFIG_FUSE_BPF +out_queue_forget: +#endif + if (!*inode && outarg->nodeid) { fuse_queue_forget(fm->fc, forget, outarg->nodeid, 1); goto out; } @@ -484,17 +580,29 @@ static struct dentry *fuse_lookup(struct inode *dir, struct dentry *entry, { int err; struct fuse_entry_out outarg; + struct fuse_entry_bpf_out bpf_outarg = {0}; struct inode *inode; struct dentry *newent; bool outarg_valid = true; bool locked; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(dir, struct fuse_lookup_io, + fuse_lookup_initialize, fuse_lookup_backing, + fuse_lookup_finalize, + dir, entry, flags); + if (fer.ret) + return fer.result; +#endif + if (fuse_is_bad(dir)) return ERR_PTR(-EIO); locked = fuse_lock_inode(dir); err = fuse_lookup_name(dir->i_sb, get_node_id(dir), &entry->d_name, - &outarg, &inode); + &outarg, &bpf_outarg, entry, &inode); fuse_unlock_inode(dir, locked); if (err == -ENOENT) { outarg_valid = false; @@ -553,6 +661,20 @@ static int fuse_create_open(struct inode *dir, struct dentry *entry, /* Userspace expects S_IFREG in create mode */ BUG_ON((mode & S_IFMT) != S_IFREG); +#ifdef CONFIG_FUSE_BPF + { + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(dir, struct fuse_create_open_io, + fuse_create_open_initialize, + fuse_create_open_backing, + fuse_create_open_finalize, + dir, entry, file, flags, mode); + if (fer.ret) + return PTR_ERR(fer.result); + } +#endif + forget = fuse_alloc_forget(); err = -ENOMEM; if (!forget) @@ -745,6 +867,17 @@ static int fuse_mknod(struct inode *dir, struct dentry *entry, umode_t mode, struct fuse_mount *fm = get_fuse_mount(dir); FUSE_ARGS(args); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(dir, struct fuse_mknod_in, + fuse_mknod_initialize, fuse_mknod_backing, + fuse_mknod_finalize, + dir, entry, mode, rdev); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (!fm->fc->dont_mask) mode &= ~current_umask(); @@ -773,6 +906,17 @@ static int fuse_mkdir(struct inode *dir, struct dentry *entry, umode_t mode) struct fuse_mount *fm = get_fuse_mount(dir); FUSE_ARGS(args); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(dir, struct fuse_mkdir_in, + fuse_mkdir_initialize, fuse_mkdir_backing, + fuse_mkdir_finalize, + dir, entry, mode); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (!fm->fc->dont_mask) mode &= ~current_umask(); @@ -795,6 +939,17 @@ static int fuse_symlink(struct inode *dir, struct dentry *entry, unsigned len = strlen(link) + 1; FUSE_ARGS(args); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(dir, struct fuse_dummy_io, + fuse_symlink_initialize, fuse_symlink_backing, + fuse_symlink_finalize, + dir, entry, link, len); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + args.opcode = FUSE_SYMLINK; args.in_numargs = 2; args.in_args[0].size = entry->d_name.len + 1; @@ -821,6 +976,20 @@ static int fuse_unlink(struct inode *dir, struct dentry *entry) if (fuse_is_bad(dir)) return -EIO; +#ifdef CONFIG_FUSE_BPF + { + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(dir, struct fuse_dummy_io, + fuse_unlink_initialize, + fuse_unlink_backing, + fuse_unlink_finalize, + dir, entry); + if (fer.ret) + return PTR_ERR(fer.result); + } +#endif + args.opcode = FUSE_UNLINK; args.nodeid = get_node_id(dir); args.in_numargs = 1; @@ -860,6 +1029,20 @@ static int fuse_rmdir(struct inode *dir, struct dentry *entry) if (fuse_is_bad(dir)) return -EIO; +#ifdef CONFIG_FUSE_BPF + { + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(dir, struct fuse_dummy_io, + fuse_rmdir_initialize, + fuse_rmdir_backing, + fuse_rmdir_finalize, + dir, entry); + if (fer.ret) + return PTR_ERR(fer.result); + } +#endif + args.opcode = FUSE_RMDIR; args.nodeid = get_node_id(dir); args.in_numargs = 1; @@ -945,6 +1128,18 @@ static int fuse_rename2(struct inode *olddir, struct dentry *oldent, return -EINVAL; if (flags) { +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(olddir, struct fuse_rename2_in, + fuse_rename2_initialize, fuse_rename2_backing, + fuse_rename2_finalize, + olddir, oldent, newdir, newent, flags); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + + /* TODO: how should this go with bpfs involved? */ if (fc->no_rename2 || fc->minor < 23) return -EINVAL; @@ -956,6 +1151,17 @@ static int fuse_rename2(struct inode *olddir, struct dentry *oldent, err = -EINVAL; } } else { +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(olddir, struct fuse_rename_in, + fuse_rename_initialize, fuse_rename_backing, + fuse_rename_finalize, + olddir, oldent, newdir, newent); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + err = fuse_rename_common(olddir, oldent, newdir, newent, 0, FUSE_RENAME, sizeof(struct fuse_rename_in)); @@ -973,6 +1179,16 @@ static int fuse_link(struct dentry *entry, struct inode *newdir, struct fuse_mount *fm = get_fuse_mount(inode); FUSE_ARGS(args); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_link_in, fuse_link_initialize, + fuse_link_backing, fuse_link_finalize, entry, + newdir, newent); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + memset(&inarg, 0, sizeof(inarg)); inarg.oldnodeid = get_node_id(inode); args.opcode = FUSE_LINK; @@ -1073,23 +1289,13 @@ static int fuse_do_getattr(struct inode *inode, struct kstat *stat, args.out_args[0].size = sizeof(outarg); args.out_args[0].value = &outarg; err = fuse_simple_request(fm, &args); - if (!err) { - if (fuse_invalid_attr(&outarg.attr) || - inode_wrong_type(inode, outarg.attr.mode)) { - fuse_make_bad(inode); - err = -EIO; - } else { - fuse_change_attributes(inode, &outarg.attr, - attr_timeout(&outarg), - attr_version); - if (stat) - fuse_fillattr(inode, &outarg.attr, stat); - } - } + if (!err) + err = finalize_attr(inode, &outarg, attr_version, stat); return err; } static int fuse_update_get_attr(struct inode *inode, struct file *file, + const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -1097,6 +1303,17 @@ static int fuse_update_get_attr(struct inode *inode, struct file *file, int err = 0; bool sync; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_getattr_io, + fuse_getattr_initialize, fuse_getattr_backing, + fuse_getattr_finalize, + path->dentry, stat, request_mask, flags); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (flags & AT_STATX_FORCE_SYNC) sync = true; else if (flags & AT_STATX_DONT_SYNC) @@ -1121,7 +1338,7 @@ static int fuse_update_get_attr(struct inode *inode, struct file *file, int fuse_update_attributes(struct inode *inode, struct file *file) { /* Do *not* need to get atime for internal purposes */ - return fuse_update_get_attr(inode, file, NULL, + return fuse_update_get_attr(inode, file, &file->f_path, NULL, STATX_BASIC_STATS & ~STATX_ATIME, 0); } @@ -1230,6 +1447,16 @@ static int fuse_access(struct inode *inode, int mask) struct fuse_access_in inarg; int err; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_access_in, + fuse_access_initialize, fuse_access_backing, + fuse_access_finalize, inode, mask); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + BUG_ON(mask & MAY_NOT_BLOCK); if (fm->fc->no_access) @@ -1256,6 +1483,7 @@ static int fuse_perm_getattr(struct inode *inode, int mask) return -ECHILD; forget_all_cached_acls(inode); + /* TODO: BPF stuff here? But we have no dentry for path for vfs_getattr */ return fuse_do_getattr(inode, NULL, NULL); } @@ -1379,6 +1607,21 @@ static const char *fuse_get_link(struct dentry *dentry, struct inode *inode, if (fuse_is_bad(inode)) goto out_err; +#ifdef CONFIG_FUSE_BPF + { + struct fuse_err_ret fer; + const char *out = NULL; + + fer = fuse_bpf_backing(inode, struct fuse_dummy_io, + fuse_get_link_initialize, + fuse_get_link_backing, + fuse_get_link_finalize, + inode, dentry, callback, &out); + if (fer.ret) + return fer.result ?: out; + } +#endif + if (fc->cache_symlinks) return page_get_link_raw(dentry, inode, callback); @@ -1412,8 +1655,18 @@ static int fuse_dir_open(struct inode *inode, struct file *file) static int fuse_dir_release(struct inode *inode, struct file *file) { - fuse_release_common(file, true); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + fer = fuse_bpf_backing(inode, struct fuse_release_in, + fuse_releasedir_initialize, fuse_release_backing, + fuse_release_finalize, + inode, file); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + + fuse_release_common(file, true); return 0; } @@ -1427,6 +1680,19 @@ static int fuse_dir_fsync(struct file *file, loff_t start, loff_t end, if (fuse_is_bad(inode)) return -EIO; +#ifdef CONFIG_FUSE_BFP + { + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_fsync_in, + fuse_dir_fsync_initialize, fuse_fsync_backing, + fuse_fsync_finalize, + file, start, end, datasync); + if (fer.ret) + return PTR_ERR(fer.result); + } +#endif + if (fc->no_fsyncdir) return 0; @@ -1578,6 +1844,16 @@ int fuse_do_setattr(struct dentry *dentry, struct iattr *attr, bool trust_local_cmtime = is_wb && S_ISREG(inode->i_mode); bool fault_blocked = false; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_setattr_io, + fuse_setattr_initialize, fuse_setattr_backing, + fuse_setattr_finalize, dentry, attr, file); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (!fc->default_permissions) attr->ia_valid |= ATTR_FORCE; @@ -1741,11 +2017,22 @@ static int fuse_setattr(struct dentry *entry, struct iattr *attr) * This should be done on write(), truncate() and chown(). */ if (!fc->handle_killpriv) { +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + /* * ia_mode calculation may have used stale i_mode. * Refresh and recalculate. */ - ret = fuse_do_getattr(inode, NULL, file); + fer = fuse_bpf_backing(inode, struct fuse_getattr_io, + fuse_getattr_initialize, fuse_getattr_backing, + fuse_getattr_finalize, + entry, NULL, 0, 0); + if (fer.ret) + ret = PTR_ERR(fer.result); + else +#endif + ret = fuse_do_getattr(inode, NULL, file); if (ret) return ret; @@ -1801,7 +2088,8 @@ static int fuse_getattr(const struct path *path, struct kstat *stat, return -EACCES; } - return fuse_update_get_attr(inode, NULL, stat, request_mask, flags); + return fuse_update_get_attr(inode, NULL, path, stat, request_mask, + flags); } static const struct inode_operations fuse_dir_inode_operations = { diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 50482229f571..b5ff5d34ac65 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -8,6 +8,7 @@ #include "fuse_i.h" +#include #include #include #include @@ -137,7 +138,11 @@ int fuse_do_open(struct fuse_mount *fm, u64 nodeid, struct file *file, struct fuse_file *ff; int opcode = isdir ? FUSE_OPENDIR : FUSE_OPEN; - ff = fuse_file_alloc(fm); + if (file->private_data) { + ff = file->private_data; + file->private_data = NULL; + } else + ff = fuse_file_alloc(fm); if (!ff) return -ENOMEM; @@ -236,6 +241,20 @@ int fuse_open_common(struct inode *inode, struct file *file, bool isdir) if (err) return err; +#ifdef CONFIG_FUSE_BPF + { + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_open_io, + fuse_open_initialize, + fuse_open_backing, + fuse_open_finalize, + inode, file, isdir); + if (fer.ret) + return PTR_ERR(fer.result); + } +#endif + if (is_wb_truncate || dax_truncate) { inode_lock(inode); fuse_set_nowrite(inode); @@ -334,6 +353,17 @@ static int fuse_release(struct inode *inode, struct file *file) { struct fuse_conn *fc = get_fuse_conn(inode); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_release_in, + fuse_release_initialize, fuse_release_backing, + fuse_release_finalize, + inode, file); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + /* see fuse_vma_close() for !writeback_cache case */ if (fc->writeback_cache) write_inode_now(inode, 1); @@ -471,6 +501,17 @@ static int fuse_flush(struct file *file, fl_owner_t id) FUSE_ARGS(args); int err; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(file->f_inode, struct fuse_flush_in, + fuse_flush_initialize, fuse_flush_backing, + fuse_flush_finalize, + file, id); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (fuse_is_bad(inode)) return -EIO; @@ -543,6 +584,17 @@ static int fuse_fsync(struct file *file, loff_t start, loff_t end, struct fuse_conn *fc = get_fuse_conn(inode); int err; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_fsync_in, + fuse_fsync_initialize, fuse_fsync_backing, + fuse_fsync_finalize, + file, start, end, datasync); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (fuse_is_bad(inode)) return -EIO; @@ -1631,6 +1683,20 @@ static ssize_t fuse_file_read_iter(struct kiocb *iocb, struct iov_iter *to) if (FUSE_IS_DAX(inode)) return fuse_dax_read_iter(iocb, to); +#ifdef CONFIG_FUSE_BPF + { + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_read_in, + fuse_file_read_iter_initialize, + fuse_file_read_iter_backing, + fuse_file_read_iter_finalize, + iocb, to); + if (fer.ret) + return PTR_ERR(fer.result); + } +#endif + if (ff->passthrough.filp) return fuse_passthrough_read_iter(iocb, to); else if (!(ff->open_flags & FOPEN_DIRECT_IO)) @@ -1651,6 +1717,20 @@ static ssize_t fuse_file_write_iter(struct kiocb *iocb, struct iov_iter *from) if (FUSE_IS_DAX(inode)) return fuse_dax_write_iter(iocb, from); +#ifdef CONFIG_FUSE_BPF + { + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_file_write_iter_io, + fuse_file_write_iter_initialize, + fuse_file_write_iter_backing, + fuse_file_write_iter_finalize, + iocb, from); + if (fer.ret) + return PTR_ERR(fer.result); + } +#endif + if (ff->passthrough.filp) return fuse_passthrough_write_iter(iocb, from); else if (!(ff->open_flags & FOPEN_DIRECT_IO)) @@ -1906,6 +1986,19 @@ int fuse_write_inode(struct inode *inode, struct writeback_control *wbc) struct fuse_file *ff; int err; + /** + * TODO - fully understand why this is necessary + * + * With fuse-bpf, fsstress fails if rename is enabled without this + * + * We are getting writes here on directory inodes, which do not have an + * initialized file list so crash. + * + * The question is why we are getting those writes + */ + if (!S_ISREG(inode->i_mode)) + return 0; + ff = __fuse_write_file_get(fc, fi); err = fuse_flush_times(inode, ff); if (ff) @@ -3327,6 +3420,18 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, (!(mode & FALLOC_FL_KEEP_SIZE) || (mode & FALLOC_FL_PUNCH_HOLE)); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_fallocate_in, + fuse_file_fallocate_initialize, + fuse_file_fallocate_backing, + fuse_file_fallocate_finalize, + file, mode, offset, length); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (mode & ~(FALLOC_FL_KEEP_SIZE | FALLOC_FL_PUNCH_HOLE)) return -EOPNOTSUPP; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 92a8bff97bae..eb8e6c78c128 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -13,6 +13,8 @@ # define pr_fmt(fmt) "fuse: " fmt #endif +#include +#include #include #include #include @@ -63,11 +65,57 @@ struct fuse_forget_link { struct fuse_forget_link *next; }; +/** FUSE specific dentry data */ +#if BITS_PER_LONG < 64 || defined(CONFIG_FUSE_BPF) +struct fuse_dentry { + union { + u64 time; + struct rcu_head rcu; + }; + struct path backing_path; +}; + +static inline struct fuse_dentry *get_fuse_dentry(const struct dentry *entry) +{ + return entry->d_fsdata; +} +#endif + +#ifdef CONFIG_FUSE_BPF +static inline void get_fuse_backing_path(const struct dentry *d, + struct path *path) +{ + struct fuse_dentry *di = get_fuse_dentry(d); + + if (!di) { + *path = (struct path) {}; + return; + } + + *path = di->backing_path; + path_get(path); +} +#endif + /** FUSE inode */ struct fuse_inode { /** Inode data */ struct inode inode; +#ifdef CONFIG_FUSE_BPF + /** + * Backing inode, if this inode is from a backing file system. + * If this is set, nodeid is 0. + */ + struct inode *backing_inode; + + /** + * bpf_prog, run on all operations to determine whether to pass through + * or handle in place + */ + struct bpf_prog *bpf; +#endif + /** Unique ID, which identifies the inode between userspace * and kernel */ u64 nodeid; @@ -239,6 +287,14 @@ struct fuse_file { /** Container for data related to the passthrough functionality */ struct fuse_passthrough passthrough; +#ifdef CONFIG_FUSE_BPF + /** + * TODO: Reconcile with passthrough file + * backing file when in bpf mode + */ + struct file *backing_file; +#endif + /** RB node to be linked on fuse_conn->polled_files */ struct rb_node polled_node; @@ -488,6 +544,8 @@ struct fuse_fs_context { unsigned int max_read; unsigned int blksize; const char *subtype; + struct bpf_prog *root_bpf; + struct file *root_dir; /* DAX device, may be NULL */ struct dax_device *dax_dev; @@ -780,6 +838,9 @@ struct fuse_conn { /** Protects passthrough_req */ spinlock_t passthrough_req_lock; + + /** task_struct for fd lookups in fuse-bpf */ + struct task_struct *task; }; /* @@ -876,12 +937,16 @@ extern const struct dentry_operations fuse_root_dentry_operations; /** * Get a filled in inode */ +struct inode *fuse_iget_backing(struct super_block *sb, + struct inode *backing_inode); struct inode *fuse_iget(struct super_block *sb, u64 nodeid, int generation, struct fuse_attr *attr, u64 attr_valid, u64 attr_version); int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name, - struct fuse_entry_out *outarg, struct inode **inode); + struct fuse_entry_out *outarg, + struct fuse_entry_bpf_out *bpf_outarg, + struct dentry *entry, struct inode **inode); /** * Send FORGET command @@ -1018,6 +1083,7 @@ void fuse_invalidate_entry_cache(struct dentry *entry); void fuse_invalidate_atime(struct inode *inode); u64 entry_attr_timeout(struct fuse_entry_out *o); +void fuse_init_dentry_root(struct dentry *root, struct file *backing_dir); void fuse_change_entry_timeout(struct dentry *entry, struct fuse_entry_out *o); /** @@ -1232,6 +1298,293 @@ ssize_t fuse_passthrough_read_iter(struct kiocb *iocb, struct iov_iter *to); ssize_t fuse_passthrough_write_iter(struct kiocb *iocb, struct iov_iter *from); ssize_t fuse_passthrough_mmap(struct file *file, struct vm_area_struct *vma); +/* backing.c */ + +struct file *fuse_fget(struct fuse_conn *fc, unsigned int fd); +struct bpf_prog *fuse_get_bpf_prog(struct fuse_conn *fc, unsigned int fd); + +/* + * Dummy io passed to fuse_bpf_backing when io operation needs no scratch space + */ +struct fuse_dummy_io { + int unused; +}; + +struct fuse_open_io { + struct fuse_open_in foi; + struct fuse_open_out foo; +}; + +int fuse_open_initialize(struct fuse_args *fa, struct fuse_open_io *foi, + struct inode *inode, struct file *file, bool isdir); +int fuse_open_backing(struct fuse_args *fa, + struct inode *inode, struct file *file, bool isdir); +void *fuse_open_finalize(struct fuse_args *fa, + struct inode *inode, struct file *file, bool isdir); + +struct fuse_create_open_io { + struct fuse_create_in fci; + struct fuse_entry_out feo; + struct fuse_open_out foo; +}; + +int fuse_create_open_initialize( + struct fuse_args *fa, struct fuse_create_open_io *fcoi, + struct inode *dir, struct dentry *entry, + struct file *file, unsigned int flags, umode_t mode); +int fuse_create_open_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, + struct file *file, unsigned int flags, umode_t mode); +void *fuse_create_open_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, + struct file *file, unsigned int flags, umode_t mode); + +int fuse_mknod_initialize( + struct fuse_args *fa, struct fuse_mknod_in *fmi, + struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev); +int fuse_mknod_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev); +void *fuse_mknod_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev); + +int fuse_mkdir_initialize( + struct fuse_args *fa, struct fuse_mkdir_in *fmi, + struct inode *dir, struct dentry *entry, umode_t mode); +int fuse_mkdir_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, umode_t mode); +void *fuse_mkdir_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, umode_t mode); + +int fuse_rmdir_initialize( + struct fuse_args *fa, struct fuse_dummy_io *fmi, + struct inode *dir, struct dentry *entry); +int fuse_rmdir_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry); +void *fuse_rmdir_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry); + +int fuse_rename2_initialize(struct fuse_args *fa, struct fuse_rename2_in *fri, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent, + unsigned int flags); +int fuse_rename2_backing(struct fuse_args *fa, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent, + unsigned int flags); +void *fuse_rename2_finalize(struct fuse_args *fa, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent, + unsigned int flags); + +int fuse_rename_initialize(struct fuse_args *fa, struct fuse_rename_in *fri, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent); +int fuse_rename_backing(struct fuse_args *fa, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent); +void *fuse_rename_finalize(struct fuse_args *fa, + struct inode *olddir, struct dentry *oldent, + struct inode *newdir, struct dentry *newent); + +int fuse_unlink_initialize( + struct fuse_args *fa, struct fuse_dummy_io *fmi, + struct inode *dir, struct dentry *entry); +int fuse_unlink_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry); +void *fuse_unlink_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry); + +int fuse_link_initialize(struct fuse_args *fa, struct fuse_link_in *fli, + struct dentry *entry, struct inode *dir, + struct dentry *newent); +int fuse_link_backing(struct fuse_args *fa, struct dentry *entry, + struct inode *dir, struct dentry *newent); +void *fuse_link_finalize(struct fuse_args *fa, struct dentry *entry, + struct inode *dir, struct dentry *newent); + +int fuse_release_initialize(struct fuse_args *fa, struct fuse_release_in *fri, + struct inode *inode, struct file *file); +int fuse_releasedir_initialize(struct fuse_args *fa, + struct fuse_release_in *fri, + struct inode *inode, struct file *file); +int fuse_release_backing(struct fuse_args *fa, + struct inode *inode, struct file *file); +void *fuse_release_finalize(struct fuse_args *fa, + struct inode *inode, struct file *file); + +int fuse_flush_initialize(struct fuse_args *fa, struct fuse_flush_in *ffi, + struct file *file, fl_owner_t id); +int fuse_flush_backing(struct fuse_args *fa, struct file *file, fl_owner_t id); +void *fuse_flush_finalize(struct fuse_args *fa, + struct file *file, fl_owner_t id); + +int fuse_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, + struct file *file, loff_t start, loff_t end, int datasync); +int fuse_fsync_backing(struct fuse_args *fa, + struct file *file, loff_t start, loff_t end, int datasync); +void *fuse_fsync_finalize(struct fuse_args *fa, + struct file *file, loff_t start, loff_t end, int datasync); +int fuse_dir_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, + struct file *file, loff_t start, loff_t end, int datasync); + +struct fuse_getxattr_io { + struct fuse_getxattr_in fgi; + struct fuse_getxattr_out fgo; +}; + +int fuse_getxattr_initialize( + struct fuse_args *fa, struct fuse_getxattr_io *fgio, + struct dentry *dentry, const char *name, void *value, + size_t size); +int fuse_getxattr_backing( + struct fuse_args *fa, + struct dentry *dentry, const char *name, void *value, + size_t size); +void *fuse_getxattr_finalize( + struct fuse_args *fa, + struct dentry *dentry, const char *name, void *value, + size_t size); + +int fuse_listxattr_initialize(struct fuse_args *fa, + struct fuse_getxattr_io *fgio, + struct dentry *dentry, char *list, size_t size); +int fuse_listxattr_backing(struct fuse_args *fa, struct dentry *dentry, + char *list, size_t size); +void *fuse_listxattr_finalize(struct fuse_args *fa, struct dentry *dentry, + char *list, size_t size); + +int fuse_setxattr_initialize(struct fuse_args *fa, + struct fuse_setxattr_in *fsxi, + struct dentry *dentry, const char *name, + const void *value, size_t size, int flags); +int fuse_setxattr_backing(struct fuse_args *fa, struct dentry *dentry, + const char *name, const void *value, size_t size, + int flags); +void *fuse_setxattr_finalize(struct fuse_args *fa, struct dentry *dentry, + const char *name, const void *value, size_t size, + int flags); + +int fuse_file_read_iter_initialize( + struct fuse_args *fa, struct fuse_read_in *fri, + struct kiocb *iocb, struct iov_iter *to); +int fuse_file_read_iter_backing(struct fuse_args *fa, + struct kiocb *iocb, struct iov_iter *to); +void *fuse_file_read_iter_finalize(struct fuse_args *fa, + struct kiocb *iocb, struct iov_iter *to); + +struct fuse_file_write_iter_io { + struct fuse_write_in fwi; + struct fuse_write_out fwo; +}; + +int fuse_file_write_iter_initialize( + struct fuse_args *fa, struct fuse_file_write_iter_io *fwio, + struct kiocb *iocb, struct iov_iter *from); +int fuse_file_write_iter_backing(struct fuse_args *fa, + struct kiocb *iocb, struct iov_iter *from); +void *fuse_file_write_iter_finalize(struct fuse_args *fa, + struct kiocb *iocb, struct iov_iter *from); + +int fuse_file_fallocate_initialize(struct fuse_args *fa, + struct fuse_fallocate_in *ffi, + struct file *file, int mode, loff_t offset, loff_t length); +int fuse_file_fallocate_backing(struct fuse_args *fa, + struct file *file, int mode, loff_t offset, loff_t length); +void *fuse_file_fallocate_finalize(struct fuse_args *fa, + struct file *file, int mode, loff_t offset, loff_t length); + +struct fuse_lookup_io { + struct fuse_entry_out feo; + struct fuse_entry_bpf_out febo; +}; + +int fuse_lookup_initialize(struct fuse_args *fa, struct fuse_lookup_io *feo, + struct inode *dir, struct dentry *entry, unsigned int flags); +int fuse_lookup_backing(struct fuse_args *fa, struct inode *dir, + struct dentry *entry, unsigned int flags); +struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, + struct dentry *entry, unsigned int flags); +int fuse_revalidate_backing(struct fuse_args *fa, struct inode *dir, + struct dentry *entry, unsigned int flags); +void *fuse_revalidate_finalize(struct fuse_args *fa, struct inode *dir, + struct dentry *entry, unsigned int flags); + +struct fuse_getattr_io { + struct fuse_getattr_in fgi; + struct fuse_attr_out fao; +}; +int fuse_getattr_initialize(struct fuse_args *fa, struct fuse_getattr_io *fgio, + const struct dentry *entry, struct kstat *stat, + u32 request_mask, unsigned int flags); +int fuse_getattr_backing(struct fuse_args *fa, + const struct dentry *entry, struct kstat *stat, + u32 request_mask, unsigned int flags); +void *fuse_getattr_finalize(struct fuse_args *fa, + const struct dentry *entry, struct kstat *stat, + u32 request_mask, unsigned int flags); + +struct fuse_setattr_io { + struct fuse_setattr_in fsi; + struct fuse_attr_out fao; +}; + +int fuse_setattr_initialize(struct fuse_args *fa, struct fuse_setattr_io *fsi, + struct dentry *dentry, struct iattr *attr, struct file *file); +int fuse_setattr_backing(struct fuse_args *fa, + struct dentry *dentry, struct iattr *attr, struct file *file); +void *fuse_setattr_finalize(struct fuse_args *fa, + struct dentry *dentry, struct iattr *attr, struct file *file); + +int fuse_get_link_initialize(struct fuse_args *fa, struct fuse_dummy_io *dummy, + struct inode *inode, struct dentry *dentry, + struct delayed_call *callback, const char **out); +int fuse_get_link_backing(struct fuse_args *fa, + struct inode *inode, struct dentry *dentry, + struct delayed_call *callback, const char **out); +void *fuse_get_link_finalize(struct fuse_args *fa, + struct inode *inode, struct dentry *dentry, + struct delayed_call *callback, const char **out); + +int fuse_symlink_initialize( + struct fuse_args *fa, struct fuse_dummy_io *unused, + struct inode *dir, struct dentry *entry, const char *link, int len); +int fuse_symlink_backing( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, const char *link, int len); +void *fuse_symlink_finalize( + struct fuse_args *fa, + struct inode *dir, struct dentry *entry, const char *link, int len); + +struct fuse_read_io { + struct fuse_read_in fri; + struct fuse_read_out fro; +}; + +int fuse_readdir_initialize(struct fuse_args *fa, struct fuse_read_io *frio, + struct file *file, struct dir_context *ctx, + bool *force_again, bool *allow_force); +int fuse_readdir_backing(struct fuse_args *fa, + struct file *file, struct dir_context *ctx, + bool *force_again, bool *allow_force); +void *fuse_readdir_finalize(struct fuse_args *fa, + struct file *file, struct dir_context *ctx, + bool *force_again, bool *allow_force); + +int fuse_access_initialize(struct fuse_args *fa, struct fuse_access_in *fai, + struct inode *inode, int mask); +int fuse_access_backing(struct fuse_args *fa, struct inode *inode, int mask); +void *fuse_access_finalize(struct fuse_args *fa, struct inode *inode, int mask); + /* * FUSE caches dentries and attributes with separate timeout. The * time in jiffies until the dentry/attributes are valid is stored in @@ -1314,4 +1667,168 @@ static inline void iattr_to_fattr(struct fuse_conn *fc, struct iattr *iattr, } } +static inline int finalize_attr(struct inode *inode, struct fuse_attr_out *outarg, + u64 attr_version, struct kstat *stat) +{ + int err = 0; + + if (fuse_invalid_attr(&outarg->attr) || + ((inode->i_mode ^ outarg->attr.mode) & S_IFMT)) { + fuse_make_bad(inode); + err = -EIO; + } else { + fuse_change_attributes(inode, &outarg->attr, + attr_timeout(outarg), + attr_version); + if (stat) + fuse_fillattr(inode, &outarg->attr, stat); + } + return err; +} + +#ifdef CONFIG_FUSE_BPF +struct fuse_err_ret { + void *result; + bool ret; +}; + +/* + * expression statement to wrap the backing filter logic + * struct inode *inode: inode with bpf and backing inode + * typedef io: (typically complex) type whose components fuse_args can point to. + * An instance of this type is created locally and passed to initialize + * void initialize(struct fuse_args *fa, io *in_out, args...): function that sets + * up fa and io based on args + * int backing(struct fuse_args *fa, args...): function that actually performs + * the backing io operation + * void *finalize(struct fuse_args *, args...): function that performs any final + * work needed to commit the backing io + */ +#define fuse_bpf_backing(inode, io, initialize, backing, finalize, \ + args...) \ +({ \ + struct fuse_err_ret fer = {0}; \ + int ext_flags; \ + struct fuse_inode *fuse_inode = get_fuse_inode(inode); \ + struct fuse_mount *fm = get_fuse_mount(inode); \ + io feo = {0}; \ + struct fuse_args fa = {0}, fa_backup = {0}; \ + bool locked; \ + ssize_t res; \ + void *err; \ + int i; \ + bool initialized = false; \ + \ + do { \ + if (!fuse_inode || !fuse_inode->backing_inode) \ + break; \ + \ + err = ERR_PTR(initialize(&fa, &feo, args)); \ + if (err) { \ + fer = (struct fuse_err_ret) { \ + err, \ + true, \ + }; \ + break; \ + } \ + initialized = true; \ + \ + fa_backup = fa; \ + fa.opcode |= FUSE_PREFILTER; \ + for (i = 0; i < fa.in_numargs; ++i) \ + fa.out_args[i] = (struct fuse_arg) { \ + .size = fa.in_args[i].size, \ + .value = (void *)fa.in_args[i].value, \ + }; \ + fa.out_numargs = fa.in_numargs; \ + \ + ext_flags = fuse_inode->bpf ? \ + BPF_PROG_RUN(fuse_inode->bpf, &fa) : \ + FUSE_BPF_BACKING; \ + if (ext_flags < 0) { \ + fer = (struct fuse_err_ret) { \ + ERR_PTR(ext_flags), \ + true, \ + }; \ + break; \ + } \ + \ + if (ext_flags & FUSE_BPF_USER_FILTER) { \ + locked = fuse_lock_inode(inode); \ + res = fuse_simple_request(fm, &fa); \ + fuse_unlock_inode(inode, locked); \ + if (res < 0) { \ + fer = (struct fuse_err_ret) { \ + ERR_PTR(res), \ + true, \ + }; \ + break; \ + } \ + } \ + \ + if (!(ext_flags & FUSE_BPF_BACKING)) \ + break; \ + \ + fa.opcode &= ~FUSE_PREFILTER; \ + for (i = 0; i < fa.in_numargs; ++i) \ + fa.in_args[i] = (struct fuse_in_arg) { \ + .size = fa.out_args[i].size, \ + .value = fa.out_args[i].value, \ + }; \ + for (i = 0; i < fa.out_numargs; ++i) \ + fa.out_args[i] = (struct fuse_arg) { \ + .size = fa_backup.out_args[i].size, \ + .value = fa_backup.out_args[i].value, \ + }; \ + fa.out_numargs = fa_backup.out_numargs; \ + \ + fer = (struct fuse_err_ret) { \ + ERR_PTR(backing(&fa, args)), \ + true, \ + }; \ + if (!(ext_flags & FUSE_BPF_POST_FILTER)) \ + break; \ + \ + fa.opcode |= FUSE_POSTFILTER; \ + if (IS_ERR(fer.result)) \ + fa.error_in = PTR_ERR(fer.result); \ + for (i = 0; i < fa.out_numargs; ++i) \ + fa.in_args[fa.in_numargs++] = \ + (struct fuse_in_arg) { \ + .size = fa.out_args[i].size, \ + .value = fa.out_args[i].value, \ + }; \ + ext_flags = BPF_PROG_RUN(fuse_inode->bpf, &fa); \ + if (ext_flags < 0) { \ + fer = (struct fuse_err_ret) { \ + ERR_PTR(ext_flags), \ + true, \ + }; \ + break; \ + } \ + if (!(ext_flags & FUSE_BPF_USER_FILTER)) \ + break; \ + \ + fa.out_args[0].size = fa_backup.out_args[0].size; \ + fa.out_args[1].size = fa_backup.out_args[1].size; \ + fa.out_numargs = fa_backup.out_numargs; \ + locked = fuse_lock_inode(inode); \ + res = fuse_simple_request(fm, &fa); \ + fuse_unlock_inode(inode, locked); \ + if (res < 0) { \ + fer.result = ERR_PTR(res); \ + break; \ + } \ + } while (false); \ + \ + if (initialized && fer.ret) { \ + err = finalize(&fa, args); \ + if (err) \ + fer.result = err; \ + } \ + \ + fer; \ +}) +#endif /* CONFIG_FUSE_BPF */ + #endif /* _FS_FUSE_I_H */ diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 679af097f760..4e1003775aa0 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -79,6 +79,10 @@ static struct inode *fuse_alloc_inode(struct super_block *sb) fi->i_time = 0; fi->inval_mask = 0; +#ifdef CONFIG_FUSE_BPF + fi->backing_inode = NULL; + fi->bpf = NULL; +#endif fi->nodeid = 0; fi->nlookup = 0; fi->attr_version = 0; @@ -119,6 +123,12 @@ static void fuse_evict_inode(struct inode *inode) { struct fuse_inode *fi = get_fuse_inode(inode); +#ifdef CONFIG_FUSE_BPF + iput(fi->backing_inode); + if (fi->bpf) + bpf_prog_put(fi->bpf); + fi->bpf = NULL; +#endif truncate_inode_pages_final(&inode->i_data); clear_inode(inode); if (inode->i_sb->s_flags & SB_ACTIVE) { @@ -313,22 +323,74 @@ static void fuse_init_inode(struct inode *inode, struct fuse_attr *attr) BUG(); } +struct fuse_inode_identifier { + u64 nodeid; + struct inode *backing_inode; +}; + static int fuse_inode_eq(struct inode *inode, void *_nodeidp) { - u64 nodeid = *(u64 *) _nodeidp; - if (get_node_id(inode) == nodeid) - return 1; - else - return 0; + struct fuse_inode_identifier *fii = + (struct fuse_inode_identifier *) _nodeidp; + struct fuse_inode *fi = get_fuse_inode(inode); + + return fii->nodeid == fi->nodeid +#ifdef CONFIG_FUSE_BPF + && fii->backing_inode == fi->backing_inode +#endif + ; } static int fuse_inode_set(struct inode *inode, void *_nodeidp) { - u64 nodeid = *(u64 *) _nodeidp; - get_fuse_inode(inode)->nodeid = nodeid; + struct fuse_inode_identifier *fii = + (struct fuse_inode_identifier *) _nodeidp; + struct fuse_inode *fi = get_fuse_inode(inode); + + fi->nodeid = fii->nodeid; +#ifdef CONFIG_FUSE_BPF + fi->backing_inode = fii->backing_inode; + if (fi->backing_inode) + ihold(fi->backing_inode); +#endif + return 0; } +struct inode *fuse_iget_backing(struct super_block *sb, + struct inode *backing_inode) +{ + struct inode *inode; + struct fuse_inode *fi; + struct fuse_conn *fc = get_fuse_conn_super(sb); + struct fuse_inode_identifier fii = { + .backing_inode = backing_inode, + }; + struct fuse_attr attr; + + fuse_fill_attr_from_inode(&attr, backing_inode); + inode = iget5_locked(sb, (unsigned long) backing_inode, fuse_inode_eq, + fuse_inode_set, &fii); + if (!inode) + return NULL; + + if ((inode->i_state & I_NEW)) { + inode->i_flags |= S_NOATIME; + if (!fc->writeback_cache) + inode->i_flags |= S_NOCMTIME; + fuse_init_common(inode); + unlock_new_inode(inode); + } + + fi = get_fuse_inode(inode); + fuse_init_inode(inode, &attr); + spin_lock(&fi->lock); + fi->nlookup++; + spin_unlock(&fi->lock); + + return inode; +} + struct inode *fuse_iget(struct super_block *sb, u64 nodeid, int generation, struct fuse_attr *attr, u64 attr_valid, u64 attr_version) @@ -336,6 +398,9 @@ struct inode *fuse_iget(struct super_block *sb, u64 nodeid, struct inode *inode; struct fuse_inode *fi; struct fuse_conn *fc = get_fuse_conn_super(sb); + struct fuse_inode_identifier fii = { + .nodeid = nodeid, + }; /* * Auto mount points get their node id from the submount root, which is @@ -357,7 +422,7 @@ struct inode *fuse_iget(struct super_block *sb, u64 nodeid, } retry: - inode = iget5_locked(sb, nodeid, fuse_inode_eq, fuse_inode_set, &nodeid); + inode = iget5_locked(sb, nodeid, fuse_inode_eq, fuse_inode_set, &fii); if (!inode) return NULL; @@ -392,13 +457,16 @@ struct inode *fuse_ilookup(struct fuse_conn *fc, u64 nodeid, { struct fuse_mount *fm_iter; struct inode *inode; + struct fuse_inode_identifier fii = { + .nodeid = nodeid, + }; WARN_ON(!rwsem_is_locked(&fc->killsb)); list_for_each_entry(fm_iter, &fc->mounts, fc_entry) { if (!fm_iter->sb) continue; - inode = ilookup5(fm_iter->sb, nodeid, fuse_inode_eq, &nodeid); + inode = ilookup5(fm_iter->sb, nodeid, fuse_inode_eq, &fii); if (inode) { if (fm) *fm = fm_iter; @@ -537,6 +605,8 @@ enum { OPT_ALLOW_OTHER, OPT_MAX_READ, OPT_BLKSIZE, + OPT_ROOT_BPF, + OPT_ROOT_DIR, OPT_ERR }; @@ -551,6 +621,8 @@ static const struct fs_parameter_spec fuse_param_specs[] = { fsparam_u32 ("max_read", OPT_MAX_READ), fsparam_u32 ("blksize", OPT_BLKSIZE), fsparam_string ("subtype", OPT_SUBTYPE), + fsparam_u32 ("root_bpf", OPT_ROOT_BPF), + fsparam_u32 ("root_dir", OPT_ROOT_DIR), {} }; @@ -639,6 +711,21 @@ static int fuse_parse_param(struct fs_context *fc, struct fs_parameter *param) ctx->blksize = result.uint_32; break; + case OPT_ROOT_BPF: + ctx->root_bpf = bpf_prog_get_type_dev(result.uint_32, + BPF_PROG_TYPE_FUSE, false); + if (IS_ERR(ctx->root_bpf)) { + ctx->root_bpf = NULL; + return invalf(fc, "Unable to open bpf program"); + } + break; + + case OPT_ROOT_DIR: + ctx->root_dir = fget(result.uint_32); + if (!ctx->root_dir) + return invalf(fc, "Unable to open root directory"); + break; + default: return -EINVAL; } @@ -651,6 +738,10 @@ static void fuse_free_fc(struct fs_context *fc) struct fuse_fs_context *ctx = fc->fs_private; if (ctx) { + if (ctx->root_dir) + fput(ctx->root_dir); + if (ctx->root_bpf) + bpf_prog_put(ctx->root_bpf); kfree(ctx->subtype); kfree(ctx); } @@ -787,15 +878,34 @@ struct fuse_mount *fuse_mount_get(struct fuse_mount *fm) } EXPORT_SYMBOL_GPL(fuse_mount_get); -static struct inode *fuse_get_root_inode(struct super_block *sb, unsigned mode) +static struct inode *fuse_get_root_inode(struct super_block *sb, + unsigned int mode, + struct bpf_prog *root_bpf, + struct file *backing_fd) { struct fuse_attr attr; - memset(&attr, 0, sizeof(attr)); + struct inode *inode; + memset(&attr, 0, sizeof(attr)); attr.mode = mode; attr.ino = FUSE_ROOT_ID; attr.nlink = 1; - return fuse_iget(sb, 1, 0, &attr, 0, 0); + inode = fuse_iget(sb, 1, 0, &attr, 0, 0); + if (!inode) + return NULL; + +#ifdef CONFIG_FUSE_BPF + get_fuse_inode(inode)->bpf = root_bpf; + if (root_bpf) + bpf_prog_inc(root_bpf); + + if (backing_fd) { + get_fuse_inode(inode)->backing_inode = backing_fd->f_inode; + ihold(backing_fd->f_inode); + } +#endif + + return inode; } struct fuse_inode_handle { @@ -810,20 +920,24 @@ static struct dentry *fuse_get_dentry(struct super_block *sb, struct inode *inode; struct dentry *entry; int err = -ESTALE; + struct fuse_inode_identifier fii = { + .nodeid = handle->nodeid, + }; if (handle->nodeid == 0) goto out_err; - inode = ilookup5(sb, handle->nodeid, fuse_inode_eq, &handle->nodeid); + inode = ilookup5(sb, handle->nodeid, fuse_inode_eq, &fii); if (!inode) { struct fuse_entry_out outarg; + struct fuse_entry_bpf_out bpf_outarg; const struct qstr name = QSTR_INIT(".", 1); if (!fc->export_support) goto out_err; err = fuse_lookup_name(sb, handle->nodeid, &name, &outarg, - &inode); + &bpf_outarg, NULL, &inode); if (err && err != -ENOENT) goto out_err; if (err || !inode) { @@ -917,6 +1031,7 @@ static struct dentry *fuse_get_parent(struct dentry *child) struct inode *inode; struct dentry *parent; struct fuse_entry_out outarg; + struct fuse_entry_bpf_out bpf_outarg; const struct qstr name = QSTR_INIT("..", 2); int err; @@ -924,7 +1039,7 @@ static struct dentry *fuse_get_parent(struct dentry *child) return ERR_PTR(-ESTALE); err = fuse_lookup_name(child_inode->i_sb, get_node_id(child_inode), - &name, &outarg, &inode); + &name, &outarg, &bpf_outarg, NULL, &inode); if (err) { if (err == -ENOENT) return ERR_PTR(-ESTALE); @@ -1110,6 +1225,7 @@ static void process_init_reply(struct fuse_mount *fm, struct fuse_args *args, fc->minor = arg->minor; fc->max_write = arg->minor < 5 ? 4096 : arg->max_write; fc->max_write = max_t(unsigned, 4096, fc->max_write); + fc->task = get_task_struct(current); fc->conn_init = 1; } kfree(ia); @@ -1186,6 +1302,8 @@ void fuse_free_conn(struct fuse_conn *fc) idr_for_each(&fc->passthrough_req, free_fuse_passthrough, NULL); idr_destroy(&fc->passthrough_req); kfree_rcu(fc, rcu); + if (fc->task) + put_task_struct(fc->task); } EXPORT_SYMBOL_GPL(fuse_free_conn); @@ -1410,11 +1528,13 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) fc->no_force_umount = ctx->no_force_umount; err = -ENOMEM; - root = fuse_get_root_inode(sb, ctx->rootmode); + root = fuse_get_root_inode(sb, ctx->rootmode, ctx->root_bpf, + ctx->root_dir); sb->s_d_op = &fuse_root_dentry_operations; root_dentry = d_make_root(root); if (!root_dentry) goto err_dev_free; + fuse_init_dentry_root(root_dentry, ctx->root_dir); /* Root dentry doesn't have .d_revalidate */ sb->s_d_op = &fuse_dentry_operations; @@ -1710,6 +1830,26 @@ static void fuse_fs_cleanup(void) static struct kobject *fuse_kobj; +/* TODO Remove this once BPF_PROG_TYPE_FUSE is upstreamed */ +static ssize_t bpf_prog_type_fuse_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buff) +{ + return sysfs_emit(buff, "%d\n", BPF_PROG_TYPE_FUSE); +} + +static struct kobj_attribute bpf_prog_type_fuse_attr = + __ATTR_RO(bpf_prog_type_fuse); + +static struct attribute *bpf_attributes[] = { + &bpf_prog_type_fuse_attr.attr, + NULL, +}; + +static const struct attribute_group bpf_attr_group = { + .attrs = bpf_attributes, +}; +/* TODO remove to here */ + static int fuse_sysfs_init(void) { int err; @@ -1724,8 +1864,15 @@ static int fuse_sysfs_init(void) if (err) goto out_fuse_unregister; + /* TODO Remove when BPF_PROG_TYPE_FUSE is upstreamed */ + err = sysfs_create_group(fuse_kobj, &bpf_attr_group); + if (err) + goto out_fuse_remove_mount_point; + return 0; + out_fuse_remove_mount_point: + sysfs_remove_mount_point(fuse_kobj, "connections"); out_fuse_unregister: kobject_put(fuse_kobj); out_err: diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index 9413a45734c0..1ca60c98cf9f 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -579,6 +579,21 @@ int fuse_readdir(struct file *file, struct dir_context *ctx) struct inode *inode = file_inode(file); int err; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + bool force_again, allow_force; +again: + fer = fuse_bpf_backing(inode, struct fuse_read_io, + fuse_readdir_initialize, fuse_readdir_backing, + fuse_readdir_finalize, + file, ctx, &force_again, &allow_force); + if (force_again && !IS_ERR(fer.result)) + goto again; + + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (fuse_is_bad(inode)) return -EIO; diff --git a/fs/fuse/xattr.c b/fs/fuse/xattr.c index b17cb7224144..3f09fa38d412 100644 --- a/fs/fuse/xattr.c +++ b/fs/fuse/xattr.c @@ -113,6 +113,17 @@ ssize_t fuse_listxattr(struct dentry *entry, char *list, size_t size) struct fuse_getxattr_out outarg; ssize_t ret; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_getxattr_io, + fuse_listxattr_initialize, + fuse_listxattr_backing, fuse_listxattr_finalize, + entry, list, size); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (fuse_is_bad(inode)) return -EIO; @@ -181,6 +192,17 @@ static int fuse_xattr_get(const struct xattr_handler *handler, struct dentry *dentry, struct inode *inode, const char *name, void *value, size_t size, int flags) { +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_getxattr_io, + fuse_getxattr_initialize, fuse_getxattr_backing, + fuse_getxattr_finalize, + dentry, name, value, size); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (fuse_is_bad(inode)) return -EIO; @@ -192,6 +214,17 @@ static int fuse_xattr_set(const struct xattr_handler *handler, const char *name, const void *value, size_t size, int flags) { +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_setxattr_in, + fuse_setxattr_initialize, fuse_setxattr_backing, + fuse_setxattr_finalize, dentry, name, value, + size, flags); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (fuse_is_bad(inode)) return -EIO; diff --git a/include/linux/bpf_types.h b/include/linux/bpf_types.h index 2bb5801b5887..eaddb8d8365b 100644 --- a/include/linux/bpf_types.h +++ b/include/linux/bpf_types.h @@ -77,6 +77,9 @@ BPF_PROG_TYPE(BPF_PROG_TYPE_LSM, lsm, void *, void *) #endif /* CONFIG_BPF_LSM */ #endif +#ifdef CONFIG_FUSE_BPF +BPF_PROG_TYPE(BPF_PROG_TYPE_FUSE, fuse, struct fuse_args, struct fuse_args) +#endif BPF_MAP_TYPE(BPF_MAP_TYPE_ARRAY, array_map_ops) BPF_MAP_TYPE(BPF_MAP_TYPE_PERCPU_ARRAY, percpu_array_map_ops) diff --git a/include/uapi/linux/bpf.h b/include/uapi/linux/bpf.h index a33c6997adcd..f1307ab8c21b 100644 --- a/include/uapi/linux/bpf.h +++ b/include/uapi/linux/bpf.h @@ -199,6 +199,7 @@ enum bpf_prog_type { BPF_PROG_TYPE_EXT, BPF_PROG_TYPE_LSM, BPF_PROG_TYPE_SK_LOOKUP, + BPF_PROG_TYPE_FUSE, }; enum bpf_attach_type { diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index 6620e0426355..dfb20abbf536 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -516,6 +516,17 @@ struct fuse_entry_out { struct fuse_attr attr; }; +#define FUSE_ACTION_KEEP 0 +#define FUSE_ACTION_REMOVE 1 +#define FUSE_ACTION_REPLACE 2 + +struct fuse_entry_bpf_out { + uint64_t backing_action; + uint64_t backing_fd; + uint64_t bpf_action; + uint64_t bpf_fd; +}; + struct fuse_forget_in { uint64_t nlookup; }; @@ -813,7 +824,7 @@ struct fuse_in_header { uint32_t uid; uint32_t gid; uint32_t pid; - uint32_t padding; + uint32_t error_in; }; struct fuse_out_header { @@ -978,12 +989,20 @@ struct fuse_args { bool page_zeroing:1; bool page_replace:1; bool may_block:1; - struct fuse_in_arg in_args[3]; - struct fuse_arg out_args[2]; + struct fuse_in_arg in_args[5]; + struct fuse_arg out_args[3]; void (*end)(struct fuse_mount *fm, struct fuse_args *args, int error); /* Path used for completing d_canonical_path */ struct path *canonical_path; }; +#define FUSE_BPF_USER_FILTER 1 +#define FUSE_BPF_BACKING 2 +#define FUSE_BPF_POST_FILTER 4 + +#define FUSE_OPCODE_FILTER 0x0ffff +#define FUSE_PREFILTER 0x10000 +#define FUSE_POSTFILTER 0x20000 + #endif /* _LINUX_FUSE_H */ diff --git a/kernel/bpf/Makefile b/kernel/bpf/Makefile index c1b9f71ee6aa..bfdc63dc74db 100644 --- a/kernel/bpf/Makefile +++ b/kernel/bpf/Makefile @@ -36,3 +36,6 @@ obj-$(CONFIG_BPF_SYSCALL) += bpf_struct_ops.o obj-${CONFIG_BPF_LSM} += bpf_lsm.o endif obj-$(CONFIG_BPF_PRELOAD) += preload/ +ifeq ($(CONFIG_FUSE_BPF),y) +obj-$(CONFIG_BPF_SYSCALL) += bpf_fuse.o +endif diff --git a/kernel/bpf/bpf_fuse.c b/kernel/bpf/bpf_fuse.c new file mode 100644 index 000000000000..2a4db5ce06b4 --- /dev/null +++ b/kernel/bpf/bpf_fuse.c @@ -0,0 +1,72 @@ +// SPDX-License-Identifier: GPL-2.0 +// Copyright (c) 2021 Google LLC + +#include +#include + +static const struct bpf_func_proto * +fuse_prog_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog) +{ + switch (func_id) { + case BPF_FUNC_trace_printk: + return bpf_get_trace_printk_proto(); + + case BPF_FUNC_get_current_uid_gid: + return &bpf_get_current_uid_gid_proto; + + case BPF_FUNC_get_current_pid_tgid: + return &bpf_get_current_pid_tgid_proto; + + case BPF_FUNC_map_lookup_elem: + return &bpf_map_lookup_elem_proto; + + case BPF_FUNC_map_update_elem: + return &bpf_map_update_elem_proto; + + default: + pr_debug("Invalid fuse bpf func %d\n", func_id); + return NULL; + } +} + +static bool fuse_prog_is_valid_access(int off, int size, + enum bpf_access_type type, + const struct bpf_prog *prog, + struct bpf_insn_access_aux *info) +{ + int i; + + if (off < 0 || off > offsetofend(struct fuse_args, out_args)) + return false; + + /* TODO This is garbage. Do it properly */ + for (i = 0; i < 5; i++) { + if (off == offsetof(struct fuse_args, in_args[i].value)) { + info->reg_type = PTR_TO_RDONLY_BUF; + info->ctx_field_size = 256; + if (type != BPF_READ) + return false; + return true; + } + } + for (i = 0; i < 3; i++) { + if (off == offsetof(struct fuse_args, out_args[i].value)) { + info->reg_type = PTR_TO_RDWR_BUF; + info->ctx_field_size = 256; + return true; + } + } + if (type != BPF_READ) + return false; + + return true; +} + +const struct bpf_verifier_ops fuse_verifier_ops = { + .get_func_proto = fuse_prog_func_proto, + .is_valid_access = fuse_prog_is_valid_access, +}; + +const struct bpf_prog_ops fuse_prog_ops = { +}; + diff --git a/kernel/bpf/btf.c b/kernel/bpf/btf.c index 61759369d7de..face6bda1c93 100644 --- a/kernel/bpf/btf.c +++ b/kernel/bpf/btf.c @@ -3,6 +3,7 @@ #include #include +#include #include #include #include From 595da2719e1511c76c40418c07d8d720005af598 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Thu, 16 Dec 2021 18:16:18 +0000 Subject: [PATCH 060/148] UPSTREAM: ANDROID: fuse-bpf: Partial fix for mmap'd files This will simply pass all mapping operations to the backing file if it exists. This is sufficient for our needs in Android13, but must be extended in the future. Signed-off-by: Paul Lawrence Bug: 202785178 Test: fuse_test passes Change-Id: I3c51ef62415633ff0db039f25bfed1adf14e1e80 --- fs/fuse/backing.c | 38 ++++++++++++++++++++++++++++++++++++++ fs/fuse/file.c | 6 ++++++ fs/fuse/fuse_i.h | 2 ++ 3 files changed, 46 insertions(+) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index f30fe8ef5c05..bc7258d78878 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -750,6 +750,44 @@ void *fuse_file_write_iter_finalize(struct fuse_args *fa, return ERR_PTR(fwo->size); } +ssize_t fuse_backing_mmap(struct file *file, struct vm_area_struct *vma) +{ + int ret; + struct fuse_file *ff = file->private_data; + struct inode *fuse_inode = file_inode(file); + struct file *backing_file = ff->backing_file; + struct inode *backing_inode = file_inode(backing_file); + + if (!backing_file->f_op->mmap) + return -ENODEV; + + if (WARN_ON(file != vma->vm_file)) + return -EIO; + + vma->vm_file = get_file(backing_file); + + ret = call_mmap(vma->vm_file, vma); + + if (ret) + fput(backing_file); + else + fput(file); + + if (file->f_flags & O_NOATIME) + return ret; + + if ((!timespec64_equal(&fuse_inode->i_mtime, + &backing_inode->i_mtime) || + !timespec64_equal(&fuse_inode->i_ctime, + &backing_inode->i_ctime))) { + fuse_inode->i_mtime = backing_inode->i_mtime; + fuse_inode->i_ctime = backing_inode->i_ctime; + } + touch_atime(&file->f_path); + + return ret; +} + int fuse_file_fallocate_initialize(struct fuse_args *fa, struct fuse_fallocate_in *ffi, struct file *file, int mode, loff_t offset, loff_t length) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index b5ff5d34ac65..c3346b090954 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -2525,6 +2525,12 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma) if (FUSE_IS_DAX(file_inode(file))) return fuse_dax_mmap(file, vma); +#ifdef CONFIG_FUSE_BPF + /* TODO - this is simply passthrough, not a proper BPF filter */ + if (ff->backing_file) + return fuse_backing_mmap(file, vma); +#endif + if (ff->passthrough.filp) return fuse_passthrough_mmap(file, vma); diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index eb8e6c78c128..7adb8f4a6224 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1495,6 +1495,8 @@ int fuse_file_write_iter_backing(struct fuse_args *fa, void *fuse_file_write_iter_finalize(struct fuse_args *fa, struct kiocb *iocb, struct iov_iter *from); +ssize_t fuse_backing_mmap(struct file *file, struct vm_area_struct *vma); + int fuse_file_fallocate_initialize(struct fuse_args *fa, struct fuse_fallocate_in *ffi, struct file *file, int mode, loff_t offset, loff_t length); From f0e6d9571154448849c1522488b6dcb633b2c076 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Mon, 20 Dec 2021 16:40:24 +0000 Subject: [PATCH 061/148] UPSTREAM: ANDROID: fuse-bpf: Close backing-fd on both paths Bug: 202785178 Test: fuse_test passes Signed-off-by: Paul Lawrence Change-Id: Ie99e196cd67ab6d84a733cd3e027080ad2a4588b --- fs/fuse/backing.c | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index bc7258d78878..b43e55d98800 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -959,6 +959,7 @@ struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, fc = get_fuse_mount(dir)->fc; backing_file = fuse_fget(fc, febo->backing_fd); + __close_fd(fc->task->files, febo->backing_fd); if (!backing_file) return ERR_PTR(-EIO); From 950ab972348e5627b401b39e0fda2f330f487596 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Tue, 21 Dec 2021 00:13:39 +0000 Subject: [PATCH 062/148] UPSTREAM: ANDROID: fuse-bpf: Close backing fds in fuse_dentry_revalidate Bug: 202785178 Test: fuse_test runs, no leak on Android Signed-off-by: Paul Lawrence Change-Id: If3c8dbe680f21c646e98f66140b842869c2c5abf --- fs/fuse/dir.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index df8381ff0738..b1da68b09456 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -237,6 +237,15 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) &entry->d_name, &outarg, &bpf_outarg); ret = fuse_simple_request(fm, &args); dput(parent); + + /* + * TODO This doesn't seem sufficient, though we don't plan to + * change the backing file ever, so not sure what is correct + * here yet, especially as we can't return an error to user + */ + if (bpf_outarg.backing_action == FUSE_ACTION_REPLACE) + __close_fd(fm->fc->task->files, bpf_outarg.backing_fd); + /* Zero nodeid is same as -ENOENT */ if (!ret && !outarg.nodeid) ret = -ENOENT; From 5f188dae207ae43144146cfece2bd83e08083585 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Fri, 7 Jan 2022 14:15:57 -0800 Subject: [PATCH 063/148] UPSTREAM: ANDROID: fuse-bpf: Check if mkdir dentry setup Uses lookup_one_len if given dentry is still negative/unlocked after mkdir is called Bug: 202785178 Test: atest android.scopedstorage.cts.device.ScopedStorageDeviceTest#testCreateAndDeleteEmptyDir Change-Id: Id5c7dfd303c242d6966fab82d96712d289676857 Signed-off-by: Daniel Rosenberg --- fs/fuse/backing.c | 17 ++++++++++------- 1 file changed, 10 insertions(+), 7 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index b43e55d98800..d4610ff7a922 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1110,6 +1110,7 @@ int fuse_mkdir_backing( struct inode *backing_inode = get_fuse_inode(dir)->backing_inode; struct path backing_path = {}; struct inode *inode = NULL; + struct dentry *d; //TODO Actually deal with changing the backing entry in mkdir get_fuse_backing_path(entry, &backing_path); @@ -1118,17 +1119,18 @@ int fuse_mkdir_backing( inode_lock_nested(backing_inode, I_MUTEX_PARENT); err = vfs_mkdir(backing_inode, backing_path.dentry, fmi->mode & ~fmi->umask); - inode_unlock(backing_inode); if (err) goto out; if (d_really_is_negative(backing_path.dentry) || unlikely(d_unhashed(backing_path.dentry))) { - err = -EINVAL; - /** - * TODO: overlayfs responds to this situation with a - * lookupOneLen. Should we do that too? - */ - goto out; + d = lookup_one_len(entry->d_name.name, backing_path.dentry->d_parent, + entry->d_name.len); + if (IS_ERR(d)) { + err = PTR_ERR(d); + goto out; + } + dput(backing_path.dentry); + backing_path.dentry = d; } inode = fuse_iget_backing(dir->i_sb, backing_inode); if (IS_ERR(inode)) { @@ -1137,6 +1139,7 @@ int fuse_mkdir_backing( } d_instantiate(entry, inode); out: + inode_unlock(backing_inode); path_put(&backing_path); return err; } From f41257ce1bbfc1dde1653cc55d9ece9b0e3e38b9 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Fri, 7 Jan 2022 20:36:53 +0000 Subject: [PATCH 064/148] UPSTREAM: ANDROID: fuse-bpf: Fix perms on readdir Add checks for both fuse accesses and backing fs accesses Bug: 202785178 Test: fuse_test passes, also atest ScopedStorageDeviceTest passes Change-Id: Ida7d90e14ca36588a8cc19453e0d40b4f6f41aa9 Signed-off-by: Paul Lawrence --- fs/fuse/backing.c | 24 ++++++++++++++++++++++++ fs/fuse/dir.c | 4 +++- 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index d4610ff7a922..d201e59a7086 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -140,6 +140,8 @@ int fuse_open_backing(struct fuse_args *fa, struct fuse_mount *fm = get_fuse_mount(inode); const struct fuse_open_in *foi = fa->in_args[0].value; struct fuse_file *ff; + int retval; + int mask; struct fuse_dentry *fd = get_fuse_dentry(file->f_path.dentry); struct file *backing_file; @@ -148,9 +150,31 @@ int fuse_open_backing(struct fuse_args *fa, return -ENOMEM; file->private_data = ff; + switch (foi->flags & O_ACCMODE) { + case O_RDONLY: + mask = MAY_READ; + break; + + case O_WRONLY: + mask = MAY_WRITE; + break; + + case O_RDWR: + mask = MAY_READ | MAY_WRITE; + break; + + default: + return -EINVAL; + } + + retval = inode_permission(get_fuse_inode(inode)->backing_inode, mask); + if (retval) + return retval; + backing_file = dentry_open(&fd->backing_path, foi->flags, current_cred()); + if (IS_ERR(backing_file)) { fuse_file_free(ff); file->private_data = NULL; diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index b1da68b09456..b2aa101ab2ef 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -1514,6 +1514,7 @@ static int fuse_permission(struct inode *inode, int mask) struct fuse_conn *fc = get_fuse_conn(inode); bool refreshed = false; int err = 0; + struct fuse_inode *fi = get_fuse_inode(inode); if (fuse_is_bad(inode)) return -EIO; @@ -1526,7 +1527,6 @@ static int fuse_permission(struct inode *inode, int mask) */ if (fc->default_permissions || ((mask & MAY_EXEC) && S_ISREG(inode->i_mode))) { - struct fuse_inode *fi = get_fuse_inode(inode); u32 perm_mask = STATX_MODE | STATX_UID | STATX_GID; if (perm_mask & READ_ONCE(fi->inval_mask) || @@ -1566,6 +1566,8 @@ static int fuse_permission(struct inode *inode, int mask) if (!err && !(inode->i_mode & S_IXUGO)) return -EACCES; } + } else if (!(mask & MAY_NOT_BLOCK) && fi->backing_inode) { + err = fuse_access(inode, mask); } return err; } From 711aab4fc879a49341c735c8d02a31b92cf7c459 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Tue, 11 Jan 2022 21:31:18 +0000 Subject: [PATCH 065/148] UPSTREAM: ANDROID: fuse-bpf: Make compile with CONFIG_FUSE but no CONFIG_FUSE_BPF Fixes: ANDROID: fuse-bpf: Fix perms on readdir Test: Builds with and without CONFIG_FUSE_BPF Bug: 202785178 Signed-off-by: Paul Lawrence Change-Id: If8b4603dd6f4bd159bfd68fc61c377dcb62ebcd0 --- fs/fuse/dir.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index b2aa101ab2ef..e04b3c77397f 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -1566,8 +1566,10 @@ static int fuse_permission(struct inode *inode, int mask) if (!err && !(inode->i_mode & S_IXUGO)) return -EACCES; } +#ifdef CONFIG_FUSE_BPF } else if (!(mask & MAY_NOT_BLOCK) && fi->backing_inode) { err = fuse_access(inode, mask); +#endif } return err; } From 73ed3d189ee5f2ee2a5133a758d0af3ba6810ceb Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Wed, 12 Jan 2022 19:29:56 +0000 Subject: [PATCH 066/148] UPSTREAM: ANDROID: fuse-bpf: Support inotify Test: fuse_test, atest CtsOsTestCases:android.os.cts.FileObserverTest Bug: 202785178 Signed-off-by: Daniel Rosenberg Signed-off-by: Paul Lawrence Signed-off-by: Alessio Balsini Change-Id: I88719a8ab23c2042fb8f50462f023e247aa4b6c3 --- fs/fuse/backing.c | 23 +++++++++++++++++++++++ fs/fuse/dir.c | 13 +++++++++++++ fs/fuse/fuse_i.h | 10 ++++++++++ 3 files changed, 46 insertions(+) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index d201e59a7086..8f9f4e146023 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1026,6 +1026,29 @@ void *fuse_revalidate_finalize(struct fuse_args *fa, struct inode *dir, return 0; } +int fuse_canonical_path_initialize(struct fuse_args *fa, + struct fuse_dummy_io *fdi, + const struct path *path, + struct path *canonical_path) +{ + fa->opcode = FUSE_CANONICAL_PATH; + return 0; +} + +int fuse_canonical_path_backing(struct fuse_args *fa, const struct path *path, + struct path *canonical_path) +{ + get_fuse_backing_path(path->dentry, canonical_path); + return 0; +} + +void *fuse_canonical_path_finalize(struct fuse_args *fa, + const struct path *path, + struct path *canonical_path) +{ + return NULL; +} + int fuse_mknod_initialize( struct fuse_args *fa, struct fuse_mknod_in *fmi, struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index e04b3c77397f..f4208e8da51d 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -331,6 +331,19 @@ static void fuse_dentry_canonical_path(const struct path *path, struct path *can char *path_name; int err; + +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_dummy_io, + fuse_canonical_path_initialize, + fuse_canonical_path_backing, + fuse_canonical_path_finalize, path, + canonical_path); + if (fer.ret) + return; +#endif + path_name = (char *)get_zeroed_page(GFP_KERNEL); if (!path_name) goto default_path; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 7adb8f4a6224..47bcd722a88b 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1521,6 +1521,16 @@ int fuse_revalidate_backing(struct fuse_args *fa, struct inode *dir, void *fuse_revalidate_finalize(struct fuse_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags); +int fuse_canonical_path_initialize(struct fuse_args *fa, + struct fuse_dummy_io *fdi, + const struct path *path, + struct path *canonical_path); +int fuse_canonical_path_backing(struct fuse_args *fa, const struct path *path, + struct path *canonical_path); +void *fuse_canonical_path_finalize(struct fuse_args *fa, + const struct path *path, + struct path *canonical_path); + struct fuse_getattr_io { struct fuse_getattr_in fgi; struct fuse_attr_out fao; From 0f65460b1028103005d569b53826f135a2924fc3 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Fri, 14 Jan 2022 21:33:59 -0800 Subject: [PATCH 067/148] UPSTREAM: ANDROID: bpf-fuse: Fix Setattr Setattr implementation was mixing up some flags, and missing some of them. Test: atest android.appsecurity.cts.ExternalStorageHostTest Bug: 202785178 Signed-off-by: Daniel Rosenberg Change-Id: Id41fa30881766faad5858b658f5b6871c0ae46b3 --- fs/fuse/backing.c | 54 +++++++++++++++++++++++++++++------------------ 1 file changed, 33 insertions(+), 21 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 8f9f4e146023..3099d3c1c962 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1644,37 +1644,40 @@ void *fuse_getattr_finalize(struct fuse_args *fa, return ERR_PTR(err); } -static void fattr_to_iattr(const struct fuse_setattr_in *arg, +static void fattr_to_iattr(struct fuse_conn *fc, + const struct fuse_setattr_in *arg, struct iattr *iattr) { - unsigned int ivalid = arg->valid; + unsigned int fvalid = arg->valid; - if (ivalid & ATTR_MODE) - iattr->ia_valid |= FATTR_MODE, iattr->ia_mode = arg->mode; - if (ivalid & ATTR_UID) { - iattr->ia_valid |= FATTR_UID; - iattr->ia_uid = KUIDT_INIT(arg->uid); + if (fvalid & FATTR_MODE) + iattr->ia_valid |= ATTR_MODE, iattr->ia_mode = arg->mode; + if (fvalid & FATTR_UID) { + iattr->ia_valid |= ATTR_UID; + iattr->ia_uid = make_kuid(fc->user_ns, arg->uid); } - if (ivalid & ATTR_GID) { - iattr->ia_valid |= FATTR_GID; - iattr->ia_gid = KGIDT_INIT(arg->gid); + if (fvalid & FATTR_GID) { + iattr->ia_valid |= ATTR_GID; + iattr->ia_gid = make_kgid(fc->user_ns, arg->gid); } - if (ivalid & ATTR_SIZE) - iattr->ia_valid |= FATTR_SIZE, iattr->ia_size = arg->size; - if (ivalid & ATTR_ATIME) { - iattr->ia_valid |= FATTR_ATIME; + if (fvalid & FATTR_SIZE) + iattr->ia_valid |= ATTR_SIZE, iattr->ia_size = arg->size; + if (fvalid & FATTR_ATIME) { + iattr->ia_valid |= ATTR_ATIME; iattr->ia_atime.tv_sec = arg->atime; iattr->ia_atime.tv_nsec = arg->atimensec; - if (!(ivalid & ATTR_ATIME_SET)) - iattr->ia_valid |= FATTR_ATIME_NOW; + if (!(fvalid & FATTR_ATIME_NOW)) + iattr->ia_valid |= ATTR_ATIME_SET; } - if (ivalid & ATTR_MTIME) { - iattr->ia_valid |= FATTR_MTIME; + if (fvalid & FATTR_MTIME) { + iattr->ia_valid |= ATTR_MTIME; iattr->ia_mtime.tv_sec = arg->mtime; iattr->ia_mtime.tv_nsec = arg->mtimensec; + if (!(fvalid & FATTR_MTIME_NOW)) + iattr->ia_valid |= ATTR_MTIME_SET; } - if (ivalid & ATTR_CTIME) { - iattr->ia_valid |= FATTR_CTIME; + if (fvalid & FATTR_CTIME) { + iattr->ia_valid |= ATTR_CTIME; iattr->ia_ctime.tv_sec = arg->ctime; iattr->ia_ctime.tv_nsec = arg->ctimensec; } @@ -1705,12 +1708,21 @@ int fuse_setattr_initialize(struct fuse_args *fa, struct fuse_setattr_io *fsio, int fuse_setattr_backing(struct fuse_args *fa, struct dentry *dentry, struct iattr *attr, struct file *file) { + struct fuse_conn *fc = get_fuse_conn(dentry->d_inode); const struct fuse_setattr_in *fsi = fa->in_args[0].value; struct iattr new_attr = {0}; struct path *backing_path = &get_fuse_dentry(dentry)->backing_path; int res; - fattr_to_iattr(fsi, &new_attr); + fattr_to_iattr(fc, fsi, &new_attr); + /* TODO: Some info doesn't get saved by the attr->fattr->attr transition + * When we actually allow the bpf to change these, we may have to consider + * the extra flags more, or pass more info into the bpf. Until then we can + * keep everything except for ATTR_FILE, since we'd need a file on the + * lower fs. For what it's worth, neither f2fs nor ext4 make use of that + * even if it is present. + */ + new_attr.ia_valid = attr->ia_valid & ~ATTR_FILE; inode_lock(d_inode(backing_path->dentry)); res = notify_change(backing_path->dentry, &new_attr, NULL); inode_unlock(d_inode(backing_path->dentry)); From 7009f9192425b9a69114ffefa88ae2dd351ea863 Mon Sep 17 00:00:00 2001 From: Alessio Balsini Date: Mon, 17 Jan 2022 15:09:46 +0000 Subject: [PATCH 068/148] UPSTREAM: ANDROID: fs/fuse: Backing move returns EXDEV if TO not backed In a move operation, if the FROM file has a backing inode associated and is handled in backing, it might happen that the TO file does not have a backing path associated (yet), maybe because the FUSE daemon didn't have the chance to traverse it. Thus this special case would mistakenly trigger EBADF, while EXDEV is more appropriate. Bug: 202785178 Test: mv /storage/emulated/0/Android/data//file /sdcard/DCIM Signed-off-by: Alessio Balsini Change-Id: I513c0e17c128ed9181a6b96fbf9f0b950e78be77 --- fs/fuse/backing.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 3099d3c1c962..6e24c55b8e9a 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1272,7 +1272,13 @@ static int fuse_rename_backing_common( return -EBADF; get_fuse_backing_path(newent, &new_backing_path); if (!new_backing_path.dentry) { - err = -EBADF; + /* + * TODO A file being moved from a backing path to another + * backing path which is not yet instrumented with FUSE-BPF. + * This may be slow and should be substituted with something + * more clever. + */ + err = -EXDEV; goto put_old_path; } if (new_backing_path.mnt != old_backing_path.mnt) { From e8914c1be59850096233719931faf43c5905c209 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Thu, 20 Jan 2022 16:37:23 -0800 Subject: [PATCH 069/148] UPSTREAM: ANDROID: fuse: Fix umasking in backing We should only apply the mask ourselves if we're not using POSIX acls Bug: 215212818 Test: com.android.cts.externalstorageapp.CommonExternalStorageTest #testAllPackageDirsWritable, verify files and cache permissions Signed-off-by: Daniel Rosenberg Change-Id: If105afe62a60b93cbce1ca5ab5caf11f008aa7db --- fs/fuse/backing.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 6e24c55b8e9a..069cb15c4b33 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1091,8 +1091,11 @@ int fuse_mknod_backing( return -EBADF; inode_lock_nested(backing_inode, I_MUTEX_PARENT); + mode = fmi->mode; + if (!IS_POSIXACL(backing_inode)) + mode &= ~fmi->umask; err = vfs_mknod(backing_inode, backing_path.dentry, - fmi->mode & ~fmi->umask, new_decode_dev(fmi->rdev)); + mode, new_decode_dev(fmi->rdev)); inode_unlock(backing_inode); if (err) goto out; @@ -1165,7 +1168,10 @@ int fuse_mkdir_backing( return -EBADF; inode_lock_nested(backing_inode, I_MUTEX_PARENT); - err = vfs_mkdir(backing_inode, backing_path.dentry, fmi->mode & ~fmi->umask); + mode = fmi->mode; + if (!IS_POSIXACL(backing_inode)) + mode &= ~fmi->umask; + err = vfs_mkdir(backing_inode, backing_path.dentry, mode); if (err) goto out; if (d_really_is_negative(backing_path.dentry) || From d6120d038239b423f91eb6190819b8b52439ae06 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Thu, 27 Jan 2022 20:44:30 +0000 Subject: [PATCH 070/148] UPSTREAM: ANDROID: fuse-bpf: Fix truncate Maps would crash every second launch. This was caused by maps receiving an incorrect file size after truncate, then mapping the file. Bug: 215486645 Test: fuse_test + launch maps 100 times & look for native crashes Signed-off-by: Paul Lawrence Change-Id: I13b0211330fb48592864ab53f8ffff60c19aab11 --- fs/fuse/backing.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 069cb15c4b33..98f65cde6bdc 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1738,6 +1738,9 @@ int fuse_setattr_backing(struct fuse_args *fa, inode_lock(d_inode(backing_path->dentry)); res = notify_change(backing_path->dentry, &new_attr, NULL); inode_unlock(d_inode(backing_path->dentry)); + + if (res == 0 && (new_attr.ia_valid & ATTR_SIZE)) + i_size_write(dentry->d_inode, new_attr.ia_size); return res; } From 37966b8d2463bb0262b8ccff223ad821e811eca4 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Mon, 7 Feb 2022 16:01:56 -0800 Subject: [PATCH 071/148] UPSTREAM: ANDROID: fuse: add bpf support for removexattr Bug: 218393120 Test: fuse_test#bpf_test_xattr Signed-off-by: Daniel Rosenberg Change-Id: Idf69b5b70c5dc5f09270146fe8c574ac1bde7be6 --- fs/fuse/backing.c | 33 +++++++++++++++++++++++++++++++++ fs/fuse/fuse_i.h | 8 ++++++++ fs/fuse/xattr.c | 9 ++++++++- 3 files changed, 49 insertions(+), 1 deletion(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 98f65cde6bdc..8da15a080c8f 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -661,6 +661,39 @@ void *fuse_setxattr_finalize(struct fuse_args *fa, struct dentry *dentry, return NULL; } +int fuse_removexattr_initialize(struct fuse_args *fa, + struct fuse_dummy_io *unused, + struct dentry *dentry, const char *name) +{ + *fa = (struct fuse_args) { + .nodeid = get_fuse_inode(dentry->d_inode)->nodeid, + .opcode = FUSE_REMOVEXATTR, + .in_numargs = 1, + .in_args[0] = (struct fuse_in_arg) { + .size = strlen(name) + 1, + .value = name, + }, + }; + + return 0; +} + +int fuse_removexattr_backing(struct fuse_args *fa, + struct dentry *dentry, const char *name) +{ + struct path *backing_path = + &get_fuse_dentry(dentry)->backing_path; + + /* TODO account for changes of the name by prefilter */ + return vfs_removexattr(backing_path->dentry, name); +} + +void *fuse_removexattr_finalize(struct fuse_args *fa, + struct dentry *dentry, const char *name) +{ + return NULL; +} + int fuse_file_read_iter_initialize( struct fuse_args *fa, struct fuse_read_in *fri, struct kiocb *iocb, struct iov_iter *to) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 47bcd722a88b..f4b6f4366ff7 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1474,6 +1474,14 @@ void *fuse_setxattr_finalize(struct fuse_args *fa, struct dentry *dentry, const char *name, const void *value, size_t size, int flags); +int fuse_removexattr_initialize(struct fuse_args *fa, + struct fuse_dummy_io *unused, + struct dentry *dentry, const char *name); +int fuse_removexattr_backing(struct fuse_args *fa, + struct dentry *dentry, const char *name); +void *fuse_removexattr_finalize(struct fuse_args *fa, + struct dentry *dentry, const char *name); + int fuse_file_read_iter_initialize( struct fuse_args *fa, struct fuse_read_in *fri, struct kiocb *iocb, struct iov_iter *to); diff --git a/fs/fuse/xattr.c b/fs/fuse/xattr.c index 3f09fa38d412..3abdf0e2be89 100644 --- a/fs/fuse/xattr.c +++ b/fs/fuse/xattr.c @@ -217,10 +217,17 @@ static int fuse_xattr_set(const struct xattr_handler *handler, #ifdef CONFIG_FUSE_BPF struct fuse_err_ret fer; - fer = fuse_bpf_backing(inode, struct fuse_setxattr_in, + if (value) + fer = fuse_bpf_backing(inode, struct fuse_setxattr_in, fuse_setxattr_initialize, fuse_setxattr_backing, fuse_setxattr_finalize, dentry, name, value, size, flags); + else + fer = fuse_bpf_backing(inode, struct fuse_dummy_io, + fuse_removexattr_initialize, + fuse_removexattr_backing, + fuse_removexattr_finalize, + dentry, name); if (fer.ret) return PTR_ERR(fer.result); #endif From 2e4df81c6378dc276060ab1634c795ca6b7ad790 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Mon, 7 Feb 2022 18:02:17 -0800 Subject: [PATCH 072/148] UPSTREAM: ANDROID: fuse-bpf: fix fuse_create_open_finalize If we hit an error during fuse_create_open, some variables will be undefined during the finalize, so check that they were actually initialized before accessing. Bug: 217570523 Test: attempt to over fill disk Signed-off-by: Daniel Rosenberg Change-Id: I094564b83e49eec2a6bac5bd050b4f7327b0c979 --- fs/fuse/backing.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 8da15a080c8f..ae4eb9d66781 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -339,8 +339,10 @@ void *fuse_create_open_finalize( struct fuse_entry_out *feo = fa->out_args[0].value; struct fuse_open_out *foo = fa->out_args[1].value; - fi->nodeid = feo->nodeid; - ff->fh = foo->fh; + if (fi) + fi->nodeid = feo->nodeid; + if (ff) + ff->fh = foo->fh; return 0; } From e501cd28147540d096dd97867b0425a1ef503336 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Tue, 15 Feb 2022 15:57:14 -0800 Subject: [PATCH 073/148] UPSTREAM: ANDROID: fuse-bpf: Fix filldir filldir used strcpy, potentially leading to writing the ending null past the current page. fuse_dirents are not null terminated, so we switch to using memcpy to avoid adding an extraneous null, which would be overwritten if the name was already byte aligned. Bug: 217570523 Test: generic/027 Signed-off-by: Daniel Rosenberg Change-Id: Ic5d1f1887a113e1a3319998bad47cfbac3d90baa --- fs/fuse/backing.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index ae4eb9d66781..e2c31647307f 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -191,7 +191,8 @@ void *fuse_open_finalize(struct fuse_args *fa, struct fuse_file *ff = file->private_data; struct fuse_open_out *foo = fa->out_args[0].value; - ff->fh = foo->fh; + if (ff) + ff->fh = foo->fh; return 0; } @@ -1987,7 +1988,7 @@ static int filldir(struct dir_context *ctx, const char *name, int namelen, .type = d_type, }; - strcpy(fd->name, name); + memcpy(fd->name, name, namelen); ec->offset += FUSE_DIRENT_SIZE(fd); return 0; From a454ffc049149c3d41b44e93b76f8fc83da6c4d9 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Wed, 2 Feb 2022 17:36:24 -0800 Subject: [PATCH 074/148] UPSTREAM: ANDROID: fuse-bpf: Support FUSE_STATFS Adds support for FUSE_STATFS, needed to run various filesystem tests Bug: 217570523 Test: bpf_test_statfs Change-Id: I5ee13e880118c5c79c4ca17bb2e902a3e17a7eb8 Signed-off-by: Daniel Rosenberg --- fs/fuse/backing.c | 50 +++++++++++++++++++++++++++++++++++++++++++++++ fs/fuse/fuse_i.h | 37 +++++++++++++++++++++++++++++++++++ fs/fuse/inode.c | 28 ++++++++++++-------------- 3 files changed, 99 insertions(+), 16 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index e2c31647307f..f7629c0f096c 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1786,6 +1786,56 @@ void *fuse_setattr_finalize(struct fuse_args *fa, return NULL; } +int fuse_statfs_initialize( + struct fuse_args *fa, struct fuse_statfs_out *fso, + struct dentry *dentry, struct kstatfs *buf) +{ + *fso = (struct fuse_statfs_out) {0}; + *fa = (struct fuse_args) { + .nodeid = get_node_id(d_inode(dentry)), + .opcode = FUSE_STATFS, + .out_numargs = 1, + .out_numargs = 1, + .out_args[0].size = sizeof(fso), + .out_args[0].value = fso, + }; + + return 0; +} + +int fuse_statfs_backing( + struct fuse_args *fa, + struct dentry *dentry, struct kstatfs *buf) +{ + int err = 0; + struct path backing_path; + struct fuse_statfs_out *fso = fa->out_args[0].value; + + get_fuse_backing_path(dentry, &backing_path); + if (!backing_path.dentry) + return -EBADF; + err = vfs_statfs(&backing_path, buf); + path_put(&backing_path); + buf->f_type = FUSE_SUPER_MAGIC; + + //TODO Provide postfilter opportunity to modify + if (!err) + convert_statfs_to_fuse(&fso->st, buf); + + return err; +} + +void *fuse_statfs_finalize( + struct fuse_args *fa, + struct dentry *dentry, struct kstatfs *buf) +{ + struct fuse_statfs_out *fso = fa->out_args[0].value; + + if (!fa->error_in) + convert_fuse_statfs(buf, &fso->st); + return NULL; +} + int fuse_get_link_initialize(struct fuse_args *fa, struct fuse_dummy_io *unused, struct inode *inode, struct dentry *dentry, struct delayed_call *callback, const char **out) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index f4b6f4366ff7..0b932cc8a163 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -33,6 +33,9 @@ #include #include #include +#include + +#define FUSE_SUPER_MAGIC 0x65735546 /** Default max number of pages that can be used in a single read request */ #define FUSE_DEFAULT_MAX_PAGES_PER_REQ 32 @@ -1565,6 +1568,13 @@ int fuse_setattr_backing(struct fuse_args *fa, void *fuse_setattr_finalize(struct fuse_args *fa, struct dentry *dentry, struct iattr *attr, struct file *file); +int fuse_statfs_initialize(struct fuse_args *fa, struct fuse_statfs_out *fso, + struct dentry *dentry, struct kstatfs *buf); +int fuse_statfs_backing(struct fuse_args *fa, + struct dentry *dentry, struct kstatfs *buf); +void *fuse_statfs_finalize(struct fuse_args *fa, + struct dentry *dentry, struct kstatfs *buf); + int fuse_get_link_initialize(struct fuse_args *fa, struct fuse_dummy_io *dummy, struct inode *inode, struct dentry *dentry, struct delayed_call *callback, const char **out); @@ -1706,6 +1716,33 @@ static inline int finalize_attr(struct inode *inode, struct fuse_attr_out *outar return err; } +static inline void convert_statfs_to_fuse(struct fuse_kstatfs *attr, struct kstatfs *stbuf) +{ + attr->bsize = stbuf->f_bsize; + attr->frsize = stbuf->f_frsize; + attr->blocks = stbuf->f_blocks; + attr->bfree = stbuf->f_bfree; + attr->bavail = stbuf->f_bavail; + attr->files = stbuf->f_files; + attr->ffree = stbuf->f_ffree; + attr->namelen = stbuf->f_namelen; + /* fsid is left zero */ +} + +static inline void convert_fuse_statfs(struct kstatfs *stbuf, struct fuse_kstatfs *attr) +{ + stbuf->f_type = FUSE_SUPER_MAGIC; + stbuf->f_bsize = attr->bsize; + stbuf->f_frsize = attr->frsize; + stbuf->f_blocks = attr->blocks; + stbuf->f_bfree = attr->bfree; + stbuf->f_bavail = attr->bavail; + stbuf->f_files = attr->files; + stbuf->f_ffree = attr->ffree; + stbuf->f_namelen = attr->namelen; + /* fsid is left zero */ +} + #ifdef CONFIG_FUSE_BPF struct fuse_err_ret { void *result; diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 4e1003775aa0..f2c3abfeeb9a 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -50,8 +50,6 @@ MODULE_PARM_DESC(max_user_congthresh, "Global limit for the maximum congestion threshold an " "unprivileged user can set"); -#define FUSE_SUPER_MAGIC 0x65735546 - #define FUSE_DEFAULT_BLKSIZE 512 /** Maximum number of outstanding background requests */ @@ -554,20 +552,6 @@ static void fuse_put_super(struct super_block *sb) fuse_mount_put(fm); } -static void convert_fuse_statfs(struct kstatfs *stbuf, struct fuse_kstatfs *attr) -{ - stbuf->f_type = FUSE_SUPER_MAGIC; - stbuf->f_bsize = attr->bsize; - stbuf->f_frsize = attr->frsize; - stbuf->f_blocks = attr->blocks; - stbuf->f_bfree = attr->bfree; - stbuf->f_bavail = attr->bavail; - stbuf->f_files = attr->files; - stbuf->f_ffree = attr->ffree; - stbuf->f_namelen = attr->namelen; - /* fsid is left zero */ -} - static int fuse_statfs(struct dentry *dentry, struct kstatfs *buf) { struct super_block *sb = dentry->d_sb; @@ -575,12 +559,24 @@ static int fuse_statfs(struct dentry *dentry, struct kstatfs *buf) FUSE_ARGS(args); struct fuse_statfs_out outarg; int err; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; +#endif if (!fuse_allow_current_process(fm->fc)) { buf->f_type = FUSE_SUPER_MAGIC; return 0; } +#ifdef CONFIG_FUSE_BPF + fer = fuse_bpf_backing(dentry->d_inode, struct fuse_statfs_out, + fuse_statfs_initialize, fuse_statfs_backing, + fuse_statfs_finalize, + dentry, buf); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + memset(&outarg, 0, sizeof(outarg)); args.in_numargs = 0; args.opcode = FUSE_STATFS; From 64445d52ca800be3d70629a88abf40250842a416 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Thu, 27 Jan 2022 16:53:22 -0800 Subject: [PATCH 075/148] UPSTREAM: ANDROID: fuse: allow mounting with no userspace daemon This is useful for testing fuse-bpf directly on a backing folder. Bug: 217570523 Test: mount -t fuse [DEVNAME] [mntpoint] -o user_id=0,group_id=0,rootmode=0040000, no_daemon,root_dir=[backingfd] Change-Id: I9ac13c3f707d71cbb74dba10eda5778bf3e83233 Signed-off-by: Daniel Rosenberg --- fs/fuse/fuse_i.h | 4 ++++ fs/fuse/inode.c | 40 ++++++++++++++++++++++++++-------------- 2 files changed, 30 insertions(+), 14 deletions(-) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 0b932cc8a163..49f63eb80695 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -544,6 +544,7 @@ struct fuse_fs_context { bool no_force_umount:1; bool legacy_opts_show:1; bool dax:1; + bool no_daemon:1; unsigned int max_read; unsigned int blksize; const char *subtype; @@ -792,6 +793,9 @@ struct fuse_conn { /** Passthrough mode for read/write IO */ unsigned int passthrough:1; + /** BPF Only, no Daemon running */ + unsigned int no_daemon:1; + /** The number of requests waiting for completion */ atomic_t num_waiting; diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index f2c3abfeeb9a..9036e427b768 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -603,6 +603,7 @@ enum { OPT_BLKSIZE, OPT_ROOT_BPF, OPT_ROOT_DIR, + OPT_NO_DAEMON, OPT_ERR }; @@ -619,6 +620,7 @@ static const struct fs_parameter_spec fuse_param_specs[] = { fsparam_string ("subtype", OPT_SUBTYPE), fsparam_u32 ("root_bpf", OPT_ROOT_BPF), fsparam_u32 ("root_dir", OPT_ROOT_DIR), + fsparam_flag ("no_daemon", OPT_NO_DAEMON), {} }; @@ -722,6 +724,11 @@ static int fuse_parse_param(struct fs_context *fc, struct fs_parameter *param) return invalf(fc, "Unable to open root directory"); break; + case OPT_NO_DAEMON: + ctx->no_daemon = true; + ctx->fd_present = true; + break; + default: return -EINVAL; } @@ -1277,7 +1284,7 @@ void fuse_send_init(struct fuse_mount *fm) ia->args.nocreds = true; ia->args.end = process_init_reply; - if (fuse_simple_background(fm, &ia->args, GFP_KERNEL) != 0) + if (unlikely(fm->fc->no_daemon) || fuse_simple_background(fm, &ia->args, GFP_KERNEL) != 0) process_init_reply(fm, &ia->args, -ENOTCONN); } EXPORT_SYMBOL_GPL(fuse_send_init); @@ -1522,6 +1529,7 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) fc->destroy = ctx->destroy; fc->no_control = ctx->no_control; fc->no_force_umount = ctx->no_force_umount; + fc->no_daemon = ctx->no_daemon; err = -ENOMEM; root = fuse_get_root_inode(sb, ctx->rootmode, ctx->root_bpf, @@ -1573,18 +1581,20 @@ static int fuse_fill_super(struct super_block *sb, struct fs_context *fsc) struct fuse_mount *fm; err = -EINVAL; - file = fget(ctx->fd); - if (!file) - goto err; + if (!ctx->no_daemon) { + file = fget(ctx->fd); + if (!file) + goto err; - /* - * Require mount to happen from the same user namespace which - * opened /dev/fuse to prevent potential attacks. - */ - if ((file->f_op != &fuse_dev_operations) || - (file->f_cred->user_ns != sb->s_user_ns)) - goto err_fput; - ctx->fudptr = &file->private_data; + /* + * Require mount to happen from the same user namespace which + * opened /dev/fuse to prevent potential attacks. + */ + if ((file->f_op != &fuse_dev_operations) || + (file->f_cred->user_ns != sb->s_user_ns)) + goto err_fput; + ctx->fudptr = &file->private_data; + } fc = kmalloc(sizeof(*fc), GFP_KERNEL); err = -ENOMEM; @@ -1610,7 +1620,8 @@ static int fuse_fill_super(struct super_block *sb, struct fs_context *fsc) * memory barrier for file->private_data to be visible on all * CPUs after this */ - fput(file); + if (!ctx->no_daemon) + fput(file); fuse_send_init(get_fuse_mount_super(sb)); return 0; @@ -1618,7 +1629,8 @@ static int fuse_fill_super(struct super_block *sb, struct fs_context *fsc) fuse_mount_put(fm); sb->s_fs_info = NULL; err_fput: - fput(file); + if (!ctx->no_daemon) + fput(file); err: return err; } From 6ba82128544386405a7988399a7e15f65f0032a1 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Fri, 25 Feb 2022 23:35:30 +0000 Subject: [PATCH 076/148] UPSTREAM: ANDROID: fuse-bpf: Update attributes on file write Bug: 221093504 Test: atest android.hardware.cts.CameraTest#testJpegExif Signed-off-by: Paul Lawrence Change-Id: Ideb4f4d95e60594aed9000df4c21bacfaeac3a55 --- fs/fuse/backing.c | 4 ++++ fs/fuse/fuse_i.h | 1 + fs/fuse/passthrough.c | 2 +- 3 files changed, 6 insertions(+), 1 deletion(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index f7629c0f096c..c1e57a06c469 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -797,6 +797,10 @@ int fuse_file_write_iter_backing(struct fuse_args *fa, /* TODO This just plain ignores any change to fuse_write_in */ fwo->size = vfs_iter_write(ff->backing_file, from, &iocb->ki_pos, 0); + /* Must reflect change in size of backing file to upper file */ + if (fwo->size > 0) + fuse_copyattr(file, ff->backing_file); + if (fwo->size < 0) return fwo->size; return 0; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 49f63eb80695..6149062aaf9e 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1297,6 +1297,7 @@ bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment); void fuse_dax_cancel_work(struct fuse_conn *fc); /* passthrough.c */ +void fuse_copyattr(struct file *dst_file, struct file *src_file); int fuse_passthrough_open(struct fuse_dev *fud, u32 lower_fd); int fuse_passthrough_setup(struct fuse_conn *fc, struct fuse_file *ff, struct fuse_open_out *openarg); diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c index 95368ddfb53a..013e2095ab9a 100644 --- a/fs/fuse/passthrough.c +++ b/fs/fuse/passthrough.c @@ -46,7 +46,7 @@ static void fuse_file_accessed(struct file *dst_file, struct file *src_file) touch_atime(&dst_file->f_path); } -static void fuse_copyattr(struct file *dst_file, struct file *src_file) +void fuse_copyattr(struct file *dst_file, struct file *src_file) { struct inode *dst = file_inode(dst_file); struct inode *src = file_inode(src_file); From 4f66dbf477fd5c02b7709dd38ba84014645a97f0 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Thu, 17 Feb 2022 16:10:58 -0800 Subject: [PATCH 077/148] UPSTREAM: ANDROID: fuse-bpf: Move bpf earlier in fuse_permission In the backing case, we initially want to call out to the fuse-bpf implementation, and only fall back to userspace if that requests it. Otherwise we end up making requests to userspace that the daemon may not be equiped to respond to. Change-Id: If3780aa8b7c45558717a9efba0b1781e8d63a3c0 Bug: 217570523 Test: generic/099 Signed-off-by: Daniel Rosenberg --- fs/fuse/dir.c | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index f4208e8da51d..e2d79af866b2 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -1505,7 +1505,6 @@ static int fuse_perm_getattr(struct inode *inode, int mask) return -ECHILD; forget_all_cached_acls(inode); - /* TODO: BPF stuff here? But we have no dentry for path for vfs_getattr */ return fuse_do_getattr(inode, NULL, NULL); } @@ -1528,6 +1527,9 @@ static int fuse_permission(struct inode *inode, int mask) bool refreshed = false; int err = 0; struct fuse_inode *fi = get_fuse_inode(inode); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; +#endif if (fuse_is_bad(inode)) return -EIO; @@ -1535,6 +1537,14 @@ static int fuse_permission(struct inode *inode, int mask) if (!fuse_allow_current_process(fc)) return -EACCES; +#ifdef CONFIG_FUSE_BPF + fer = fuse_bpf_backing(inode, struct fuse_access_in, + fuse_access_initialize, fuse_access_backing, + fuse_access_finalize, inode, mask); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + /* * If attributes are needed, refresh them before proceeding */ @@ -1579,10 +1589,6 @@ static int fuse_permission(struct inode *inode, int mask) if (!err && !(inode->i_mode & S_IXUGO)) return -EACCES; } -#ifdef CONFIG_FUSE_BPF - } else if (!(mask & MAY_NOT_BLOCK) && fi->backing_inode) { - err = fuse_access(inode, mask); -#endif } return err; } From 36351aded0557fa18e7e7ccf59a47edb293e5f0b Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Wed, 2 Mar 2022 15:18:27 -0800 Subject: [PATCH 078/148] UPSTREAM: ANDROID: fuse-bpf: Invalidate if lower is unhashed If the lower filesystem has deleted a file or folder behind our back, we should drop it as well. Bug: 221093504 Test: atest android.hardware.cts.CameraTest#testJpegThumbnailSize several times in a row Signed-off-by: Daniel Rosenberg Change-Id: Ibd92dea83a82dd7ab21269ae5d2533e4826f5fb7 --- fs/fuse/backing.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index c1e57a06c469..3c38a3c6cd43 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1055,6 +1055,13 @@ int fuse_revalidate_backing(struct fuse_args *fa, struct inode *dir, struct fuse_dentry *fuse_dentry = get_fuse_dentry(entry); struct dentry *backing_entry = fuse_dentry->backing_path.dentry; + spin_lock(&backing_entry->d_lock); + if (d_unhashed(backing_entry)) { + spin_unlock(&backing_entry->d_lock); + return 0; + } + spin_unlock(&backing_entry->d_lock); + if (unlikely(backing_entry->d_flags & DCACHE_OP_REVALIDATE)) return backing_entry->d_op->d_revalidate(backing_entry, flags); return 1; From 095ffde519d156e69d5438ae5b0ecc72b6918222 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Fri, 4 Mar 2022 20:11:59 +0000 Subject: [PATCH 079/148] UPSTREAM: ANDROID: fuse-bpf: Move fd operations to be synchronous Bug: 222619123 Test: fuse_test passes, on cuttlefish CtsCameraTestCases passes Signed-off-by: Paul Lawrence Change-Id: I54c148206b5ad5ae5737939bcb076cbe6c40129c --- fs/fuse/backing.c | 78 ++++++++++++----------------------------------- fs/fuse/dev.c | 12 ++++++++ fs/fuse/dir.c | 34 ++++++++++++--------- fs/fuse/fuse_i.h | 6 +--- fs/fuse/inode.c | 3 -- 5 files changed, 53 insertions(+), 80 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 3c38a3c6cd43..7cb9a17eb71c 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -13,53 +13,12 @@ #include "../internal.h" -/* Reimplement these functions since fget_task is not exported */ -static struct file *fuse__fget_files(struct files_struct *files, - unsigned int fd, fmode_t mask, unsigned int refs) +struct bpf_prog *fuse_get_bpf_prog(struct file *file) { - struct file *file; - - rcu_read_lock(); -loop: - file = fcheck_files(files, fd); - if (file) { - /* File object ref couldn't be taken. - * dup2() atomicity guarantee is the reason - * we loop to catch the new file (or NULL pointer) - */ - if (file->f_mode & mask) - file = NULL; - else if (!get_file_rcu_many(file, refs)) - goto loop; - } - rcu_read_unlock(); - return file; -} - -static struct file *fuse_fget_task(struct task_struct *task, unsigned int fd) -{ - struct file *file = NULL; - - task_lock(task); - if (task->files) - file = fuse__fget_files(task->files, fd, 0, 1); - task_unlock(task); - - return file; -} - -struct file *fuse_fget(struct fuse_conn *fc, unsigned int fd) -{ - return fuse_fget_task(fc->task, fd); -} - -struct bpf_prog *fuse_get_bpf_prog(struct fuse_conn *fc, unsigned int fd) -{ - struct file *bpf_file = fuse_fget(fc, fd); struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); - if (!bpf_file) - goto out; + if (!file || IS_ERR(file)) + return bpf_prog; /** * Two ways of getting a bpf prog from another task's fd, since * bpf_prog_get_type_dev only works with an fd @@ -75,10 +34,10 @@ struct bpf_prog *fuse_get_bpf_prog(struct fuse_conn *fc, unsigned int fd) * compilable as a module. */ #if 0 - if (bpf_file->f_op != &bpf_prog_fops) + if (file->f_op != &bpf_prog_fops) goto out; - bpf_prog = bpf_file->private_data; + bpf_prog = file->private_data; if (bpf_prog->type == BPF_PROG_TYPE_FUSE) bpf_prog_inc(bpf_prog); else @@ -86,24 +45,25 @@ struct bpf_prog *fuse_get_bpf_prog(struct fuse_conn *fc, unsigned int fd) #else { - int task_fd = get_unused_fd_flags(bpf_file->f_flags); + int task_fd = get_unused_fd_flags(file->f_flags); if (task_fd < 0) goto out; - fd_install(task_fd, bpf_file); + + fd_install(task_fd, file); bpf_prog = bpf_prog_get_type_dev(task_fd, BPF_PROG_TYPE_FUSE, false); - __close_fd(current->files, task_fd); - /* TODO I think this file is probably being leaked */ - bpf_file = NULL; + /* Close the fd, which also closes the file */ + __close_fd(current->files, task_fd); + file = NULL; } #endif out: - if (bpf_file) - fput(bpf_file); + if (file) + fput(file); return bpf_prog; } @@ -992,8 +952,11 @@ struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, break; case FUSE_ACTION_REPLACE: { - struct fuse_conn *fc = get_fuse_mount(dir)->fc; - struct bpf_prog *bpf_prog = fuse_get_bpf_prog(fc, febo->bpf_fd); + struct file *bpf_file = (struct file*) febo->bpf_fd; + struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); + + if (bpf_file && !IS_ERR(bpf_file)) + bpf_prog = fuse_get_bpf_prog(bpf_file); if (IS_ERR(bpf_prog)) return ERR_PTR(PTR_ERR(bpf_prog)); @@ -1022,9 +985,8 @@ struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, struct file *backing_file; fc = get_fuse_mount(dir)->fc; - backing_file = fuse_fget(fc, febo->backing_fd); - __close_fd(fc->task->files, febo->backing_fd); - if (!backing_file) + backing_file = (struct file *) febo->backing_fd; + if (!backing_file || IS_ERR(backing_file)) return ERR_PTR(-EIO); iput(get_fuse_inode(inode)->backing_inode); diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index 2dc094bfe3e4..3c9764124b6b 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -1959,6 +1959,18 @@ static ssize_t fuse_dev_do_write(struct fuse_dev *fud, req->out.h.error = kern_path(path, 0, req->args->canonical_path); } + if (!err && (req->in.h.opcode == FUSE_LOOKUP || + req->in.h.opcode == (FUSE_LOOKUP | FUSE_POSTFILTER)) && + req->args->out_args[1].size == sizeof(struct fuse_entry_bpf_out)) { + struct fuse_entry_bpf_out *febo = (struct fuse_entry_bpf_out *) + req->args->out_args[1].value; + + if (febo->backing_action == FUSE_ACTION_REPLACE) + febo->backing_fd = (uint64_t) fget(febo->backing_fd); + if (febo->bpf_action == FUSE_ACTION_REPLACE) + febo->bpf_fd = (uint64_t) fget(febo->bpf_fd); + } + spin_lock(&fpq->lock); clear_bit(FR_LOCKED, &req->flags); if (!fpq->connected) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index e2d79af866b2..106c3aeb51b9 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -243,8 +243,19 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) * change the backing file ever, so not sure what is correct * here yet, especially as we can't return an error to user */ - if (bpf_outarg.backing_action == FUSE_ACTION_REPLACE) - __close_fd(fm->fc->task->files, bpf_outarg.backing_fd); + if (bpf_outarg.backing_action == FUSE_ACTION_REPLACE) { + struct file *file = (struct file *) bpf_outarg.backing_fd; + + if (file && !IS_ERR(file)) + fput(file); + } + + if (bpf_outarg.bpf_action == FUSE_ACTION_REPLACE) { + struct file *file = (struct file *) bpf_outarg.bpf_fd; + + if (file && !IS_ERR(file)) + fput(file); + } /* Zero nodeid is same as -ENOENT */ if (!ret && !outarg.nodeid) @@ -528,26 +539,21 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name if (bpf_outarg->backing_action != FUSE_ACTION_REPLACE) goto out_queue_forget; - backing_file = fuse_fget(fm->fc, bpf_outarg->backing_fd); - if (!backing_file) + backing_file = (struct file *) bpf_outarg->backing_fd; + if (!backing_file || IS_ERR(backing_file)) goto out_queue_forget; - /* TODO userspace doesn't really know when the right time to - * close the passed fd is. This because after replying to the - * driver request, so assume that after a lookup with bpf_args, - * the daemon passes the fd ownership to the kernel, which also - * takes care of closing it at the right time. - */ - __close_fd(fm->fc->task->files, bpf_outarg->backing_fd); - backing_inode = backing_file->f_inode; *inode = fuse_iget_backing(sb, backing_inode); if (!*inode) goto bpf_outarg_out; if (bpf_outarg->bpf_action == FUSE_ACTION_REPLACE) { - struct bpf_prog *bpf_prog = fuse_get_bpf_prog(fm->fc, - bpf_outarg->bpf_fd); + struct file *bpf_file = (struct file*) bpf_outarg->bpf_fd; + struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); + + if (bpf_file && !IS_ERR(bpf_file)) + bpf_prog = fuse_get_bpf_prog(bpf_file);; if (IS_ERR(bpf_prog)) { iput(*inode); diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 6149062aaf9e..c2dcb4e16ae3 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -845,9 +845,6 @@ struct fuse_conn { /** Protects passthrough_req */ spinlock_t passthrough_req_lock; - - /** task_struct for fd lookups in fuse-bpf */ - struct task_struct *task; }; /* @@ -1308,8 +1305,7 @@ ssize_t fuse_passthrough_mmap(struct file *file, struct vm_area_struct *vma); /* backing.c */ -struct file *fuse_fget(struct fuse_conn *fc, unsigned int fd); -struct bpf_prog *fuse_get_bpf_prog(struct fuse_conn *fc, unsigned int fd); +struct bpf_prog *fuse_get_bpf_prog(struct file *file); /* * Dummy io passed to fuse_bpf_backing when io operation needs no scratch space diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 9036e427b768..2ff5282bb648 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1228,7 +1228,6 @@ static void process_init_reply(struct fuse_mount *fm, struct fuse_args *args, fc->minor = arg->minor; fc->max_write = arg->minor < 5 ? 4096 : arg->max_write; fc->max_write = max_t(unsigned, 4096, fc->max_write); - fc->task = get_task_struct(current); fc->conn_init = 1; } kfree(ia); @@ -1305,8 +1304,6 @@ void fuse_free_conn(struct fuse_conn *fc) idr_for_each(&fc->passthrough_req, free_fuse_passthrough, NULL); idr_destroy(&fc->passthrough_req); kfree_rcu(fc, rcu); - if (fc->task) - put_task_struct(fc->task); } EXPORT_SYMBOL_GPL(fuse_free_conn); From cbe5f9cd71a754842704733bba8f95ef18f9171d Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Mon, 7 Mar 2022 20:56:03 +0000 Subject: [PATCH 080/148] UPSTREAM: ANDROID: fuse-bpf: Fix CONFIG_FUSE_BPF typo in FUSE_FSYNCDIR Bug: 222497969 Test: idle kingdom launches sucessfully with no mediaprovider crashes Signed-off-by: Paul Lawrence Change-Id: Ia5f842fd4a0bf9a21d7c88874b13b84d42ab2c4d --- fs/fuse/dir.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 106c3aeb51b9..85f2b608e9f0 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -1718,7 +1718,7 @@ static int fuse_dir_fsync(struct file *file, loff_t start, loff_t end, if (fuse_is_bad(inode)) return -EIO; -#ifdef CONFIG_FUSE_BFP +#ifdef CONFIG_FUSE_BPF { struct fuse_err_ret fer; From 66d1fc6d062ef936e636aa869ce2f56bbba92adc Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Tue, 8 Mar 2022 17:23:14 -0800 Subject: [PATCH 081/148] UPSTREAM: ANDROID: fuse-bpf: Avoid reusing uint64_t for file This moves the backing/fd files to their own space, instead of reusing the userspace provided fds. Bug: 222619123 Test: fuse_test passes, on cuttlefish CtsCameraTestCases passes Signed-off-by: Daniel Rosenberg Change-Id: I5d3b1ea8299f249ef5adc1ce2b7f45404a041208 --- fs/fuse/backing.c | 9 +++++---- fs/fuse/dev.c | 5 +++-- fs/fuse/dir.c | 35 +++++++++++++++++------------------ fs/fuse/fuse_i.h | 3 +-- fs/fuse/inode.c | 6 ++---- include/uapi/linux/fuse.h | 6 ++++++ 6 files changed, 34 insertions(+), 30 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 7cb9a17eb71c..63bfd60897b1 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -874,8 +874,8 @@ int fuse_lookup_initialize(struct fuse_args *fa, struct fuse_lookup_io *fli, .value = &fli->feo, }, .out_args[1] = (struct fuse_arg) { - .size = sizeof(fli->febo), - .value = &fli->febo, + .size = sizeof(fli->feb.out), + .value = &fli->feb.out, }, }; @@ -917,6 +917,7 @@ struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, struct inode *inode, *backing_inode; struct fuse_entry_out *feo = fa->out_args[0].value; struct fuse_entry_bpf_out *febo = fa->out_args[1].value; + struct fuse_entry_bpf *feb = container_of(febo, struct fuse_entry_bpf, out); fd = get_fuse_dentry(entry); if (!fd) @@ -952,7 +953,7 @@ struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, break; case FUSE_ACTION_REPLACE: { - struct file *bpf_file = (struct file*) febo->bpf_fd; + struct file *bpf_file = feb->bpf_file; struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); if (bpf_file && !IS_ERR(bpf_file)) @@ -985,7 +986,7 @@ struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, struct file *backing_file; fc = get_fuse_mount(dir)->fc; - backing_file = (struct file *) febo->backing_fd; + backing_file = feb->backing_file; if (!backing_file || IS_ERR(backing_file)) return ERR_PTR(-EIO); diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index 3c9764124b6b..4b0e5a97f03c 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -1964,11 +1964,12 @@ static ssize_t fuse_dev_do_write(struct fuse_dev *fud, req->args->out_args[1].size == sizeof(struct fuse_entry_bpf_out)) { struct fuse_entry_bpf_out *febo = (struct fuse_entry_bpf_out *) req->args->out_args[1].value; + struct fuse_entry_bpf *feb = container_of(febo, struct fuse_entry_bpf, out); if (febo->backing_action == FUSE_ACTION_REPLACE) - febo->backing_fd = (uint64_t) fget(febo->backing_fd); + feb->backing_file = fget(febo->backing_fd); if (febo->bpf_action == FUSE_ACTION_REPLACE) - febo->bpf_fd = (uint64_t) fget(febo->bpf_fd); + feb->bpf_file = fget(febo->bpf_fd); } spin_lock(&fpq->lock); diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 85f2b608e9f0..ceda69466011 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -196,7 +196,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) else if (time_before64(fuse_dentry_time(entry), get_jiffies_64()) || (flags & LOOKUP_REVAL)) { struct fuse_entry_out outarg; - struct fuse_entry_bpf_out bpf_outarg; + struct fuse_entry_bpf bpf_arg; FUSE_ARGS(args); struct fuse_forget_link *forget; u64 attr_version; @@ -234,7 +234,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) parent = dget_parent(entry); fuse_lookup_init(fm->fc, &args, get_node_id(d_inode(parent)), - &entry->d_name, &outarg, &bpf_outarg); + &entry->d_name, &outarg, &bpf_arg.out); ret = fuse_simple_request(fm, &args); dput(parent); @@ -243,15 +243,15 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) * change the backing file ever, so not sure what is correct * here yet, especially as we can't return an error to user */ - if (bpf_outarg.backing_action == FUSE_ACTION_REPLACE) { - struct file *file = (struct file *) bpf_outarg.backing_fd; + if (bpf_arg.out.backing_action == FUSE_ACTION_REPLACE) { + struct file *file = bpf_arg.backing_file; if (file && !IS_ERR(file)) fput(file); } - if (bpf_outarg.bpf_action == FUSE_ACTION_REPLACE) { - struct file *file = (struct file *) bpf_outarg.bpf_fd; + if (bpf_arg.out.bpf_action == FUSE_ACTION_REPLACE) { + struct file *file = bpf_arg.bpf_file; if (file && !IS_ERR(file)) fput(file); @@ -498,12 +498,12 @@ bool fuse_invalid_attr(struct fuse_attr *attr) int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name, struct fuse_entry_out *outarg, - struct fuse_entry_bpf_out *bpf_outarg, struct dentry *entry, struct inode **inode) { struct fuse_mount *fm = get_fuse_mount_super(sb); FUSE_ARGS(args); + struct fuse_entry_bpf bpf_arg = {0}; struct fuse_forget_link *forget; u64 attr_version; int err; @@ -521,11 +521,11 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name attr_version = fuse_get_attr_version(fm->fc); - fuse_lookup_init(fm->fc, &args, nodeid, name, outarg, bpf_outarg); + fuse_lookup_init(fm->fc, &args, nodeid, name, outarg, &bpf_arg.out); err = fuse_simple_request(fm, &args); #ifdef CONFIG_FUSE_BPF - if (err == sizeof(*bpf_outarg)) { + if (err == sizeof(bpf_arg.out)) { /* TODO Make sure this handles invalid handles */ /* TODO Do we need the same code in revalidate */ struct file *backing_file; @@ -536,20 +536,20 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name goto out_queue_forget; err = -EINVAL; - if (bpf_outarg->backing_action != FUSE_ACTION_REPLACE) + if (bpf_arg.out.backing_action != FUSE_ACTION_REPLACE) goto out_queue_forget; - backing_file = (struct file *) bpf_outarg->backing_fd; + backing_file = bpf_arg.backing_file; if (!backing_file || IS_ERR(backing_file)) goto out_queue_forget; backing_inode = backing_file->f_inode; *inode = fuse_iget_backing(sb, backing_inode); if (!*inode) - goto bpf_outarg_out; + goto bpf_arg_out; - if (bpf_outarg->bpf_action == FUSE_ACTION_REPLACE) { - struct file *bpf_file = (struct file*) bpf_outarg->bpf_fd; + if (bpf_arg.out.bpf_action == FUSE_ACTION_REPLACE) { + struct file *bpf_file = bpf_arg.bpf_file; struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); if (bpf_file && !IS_ERR(bpf_file)) @@ -559,7 +559,7 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name iput(*inode); *inode = NULL; err = PTR_ERR(bpf_prog); - goto bpf_outarg_out; + goto bpf_arg_out; } get_fuse_inode(*inode)->bpf = bpf_prog; } @@ -567,7 +567,7 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name get_fuse_dentry(entry)->backing_path = backing_file->f_path; path_get(&get_fuse_dentry(entry)->backing_path); -bpf_outarg_out: +bpf_arg_out: fput(backing_file); } else #endif @@ -608,7 +608,6 @@ static struct dentry *fuse_lookup(struct inode *dir, struct dentry *entry, { int err; struct fuse_entry_out outarg; - struct fuse_entry_bpf_out bpf_outarg = {0}; struct inode *inode; struct dentry *newent; bool outarg_valid = true; @@ -630,7 +629,7 @@ static struct dentry *fuse_lookup(struct inode *dir, struct dentry *entry, locked = fuse_lock_inode(dir); err = fuse_lookup_name(dir->i_sb, get_node_id(dir), &entry->d_name, - &outarg, &bpf_outarg, entry, &inode); + &outarg, entry, &inode); fuse_unlock_inode(dir, locked); if (err == -ENOENT) { outarg_valid = false; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index c2dcb4e16ae3..a5a070461524 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -949,7 +949,6 @@ struct inode *fuse_iget(struct super_block *sb, u64 nodeid, int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name, struct fuse_entry_out *outarg, - struct fuse_entry_bpf_out *bpf_outarg, struct dentry *entry, struct inode **inode); /** @@ -1519,7 +1518,7 @@ void *fuse_file_fallocate_finalize(struct fuse_args *fa, struct fuse_lookup_io { struct fuse_entry_out feo; - struct fuse_entry_bpf_out febo; + struct fuse_entry_bpf feb; }; int fuse_lookup_initialize(struct fuse_args *fa, struct fuse_lookup_io *feo, diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 2ff5282bb648..8025cf345196 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -933,14 +933,13 @@ static struct dentry *fuse_get_dentry(struct super_block *sb, inode = ilookup5(sb, handle->nodeid, fuse_inode_eq, &fii); if (!inode) { struct fuse_entry_out outarg; - struct fuse_entry_bpf_out bpf_outarg; const struct qstr name = QSTR_INIT(".", 1); if (!fc->export_support) goto out_err; err = fuse_lookup_name(sb, handle->nodeid, &name, &outarg, - &bpf_outarg, NULL, &inode); + NULL, &inode); if (err && err != -ENOENT) goto out_err; if (err || !inode) { @@ -1034,7 +1033,6 @@ static struct dentry *fuse_get_parent(struct dentry *child) struct inode *inode; struct dentry *parent; struct fuse_entry_out outarg; - struct fuse_entry_bpf_out bpf_outarg; const struct qstr name = QSTR_INIT("..", 2); int err; @@ -1042,7 +1040,7 @@ static struct dentry *fuse_get_parent(struct dentry *child) return ERR_PTR(-ESTALE); err = fuse_lookup_name(child_inode->i_sb, get_node_id(child_inode), - &name, &outarg, &bpf_outarg, NULL, &inode); + &name, &outarg, NULL, &inode); if (err) { if (err == -ENOENT) return ERR_PTR(-ESTALE); diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index dfb20abbf536..a0c34ffc97fe 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -527,6 +527,12 @@ struct fuse_entry_bpf_out { uint64_t bpf_fd; }; +struct fuse_entry_bpf { + struct fuse_entry_bpf_out out; + struct file *backing_file; + struct file *bpf_file; +}; + struct fuse_forget_in { uint64_t nlookup; }; From d932a916a2d7ba2724f84d9feeb44796c6bc7d46 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Wed, 9 Mar 2022 17:29:18 -0800 Subject: [PATCH 082/148] UPSTREAM: ANDROID: fuse-bpf: Report errors to finalize Ensure finalize has access to the returned error code, if it had one. Test: generic/377 Bug: 217570523 Signed-off-by: Daniel Rosenberg Change-Id: Id9a0aca10c3158f8b414e0cbc2dd1282bb5ef16b --- fs/fuse/backing.c | 9 +++++++-- fs/fuse/fuse_i.h | 4 ++-- 2 files changed, 9 insertions(+), 4 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 63bfd60897b1..c351d04ccd9a 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -557,12 +557,15 @@ int fuse_listxattr_backing(struct fuse_args *fa, struct dentry *dentry, vfs_listxattr(get_fuse_dentry(dentry)->backing_path.dentry, list, size); + if (ret < 0) + return ret; + if (fa->out_argvar) fa->out_args[0].size = ret; else ((struct fuse_getxattr_out *)fa->out_args[0].value)->size = ret; - return 0; + return ret; } void *fuse_listxattr_finalize(struct fuse_args *fa, struct dentry *dentry, @@ -570,11 +573,13 @@ void *fuse_listxattr_finalize(struct fuse_args *fa, struct dentry *dentry, { struct fuse_getxattr_out *fgo; + if (fa->error_in) + return NULL; + if (fa->out_argvar) return ERR_PTR(fa->out_args[0].size); fgo = fa->out_args[0].value; - return ERR_PTR(fgo->size); } diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index a5a070461524..d15d0b896e85 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1843,12 +1843,12 @@ struct fuse_err_ret { ERR_PTR(backing(&fa, args)), \ true, \ }; \ + if (IS_ERR(fer.result)) \ + fa.error_in = PTR_ERR(fer.result); \ if (!(ext_flags & FUSE_BPF_POST_FILTER)) \ break; \ \ fa.opcode |= FUSE_POSTFILTER; \ - if (IS_ERR(fer.result)) \ - fa.error_in = PTR_ERR(fer.result); \ for (i = 0; i < fa.out_numargs; ++i) \ fa.in_args[fa.in_numargs++] = \ (struct fuse_in_arg) { \ From 03c20048f7cd7346724660b90504a4dfa034c16f Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Thu, 10 Mar 2022 18:05:44 -0800 Subject: [PATCH 083/148] UPSTREAM: ANDROID: fuse-bpf: Add support for FUSE_COPY_FILE_RANGE Bug: 217570523 Test: generic/075 Signed-off-by: Daniel Rosenberg Change-Id: I5feb75c83bc8bca2f44700d731f9f43e1eacf77c --- fs/fuse/backing.c | 58 +++++++++++++++++++++++++++++++++++++++++++++++ fs/fuse/file.c | 12 ++++++++++ fs/fuse/fuse_i.h | 19 ++++++++++++++++ 3 files changed, 89 insertions(+) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index c351d04ccd9a..efd76dc08f3d 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -403,6 +403,64 @@ void *fuse_flush_finalize(struct fuse_args *fa, struct file *file, fl_owner_t id return NULL; } +int fuse_copy_file_range_initialize(struct fuse_args *fa, struct fuse_copy_file_range_io *fcf, + struct file *file_in, loff_t pos_in, struct file *file_out, + loff_t pos_out, size_t len, unsigned int flags) +{ + struct fuse_file *fuse_file_in = file_in->private_data; + struct fuse_file *fuse_file_out = file_out->private_data; + + + fcf->fci = (struct fuse_copy_file_range_in) { + .fh_in = fuse_file_in->fh, + .off_in = pos_in, + .nodeid_out = fuse_file_out->nodeid, + .fh_out = fuse_file_out->fh, + .off_out = pos_out, + .len = len, + .flags = flags, + }; + + *fa = (struct fuse_args) { + .nodeid = get_node_id(file_in->f_inode), + .opcode = FUSE_COPY_FILE_RANGE, + .in_numargs = 1, + .in_args[0].size = sizeof(fcf->fci), + .in_args[0].value = &fcf->fci, + .out_numargs = 1, + .out_args[0].size = sizeof(fcf->fwo), + .out_args[0].value = &fcf->fwo, + }; + + return 0; +} + +int fuse_copy_file_range_backing(struct fuse_args *fa, struct file *file_in, loff_t pos_in, + struct file *file_out, loff_t pos_out, size_t len, + unsigned int flags) +{ + const struct fuse_copy_file_range_in *fci = fa->in_args[0].value; + struct fuse_file *fuse_file_in = file_in->private_data; + struct file *backing_file_in = fuse_file_in->backing_file; + struct fuse_file *fuse_file_out = file_out->private_data; + struct file *backing_file_out = fuse_file_out->backing_file; + + /* TODO: Handle changing of in/out files */ + if (backing_file_out) + return vfs_copy_file_range(backing_file_in, fci->off_in, backing_file_out, + fci->off_out, fci->len, fci->flags); + else + return generic_copy_file_range(file_in, pos_in, file_out, pos_out, len, + flags); +} + +void *fuse_copy_file_range_finalize(struct fuse_args *fa, struct file *file_in, loff_t pos_in, + struct file *file_out, loff_t pos_out, size_t len, + unsigned int flags) +{ + return NULL; +} + int fuse_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, struct file *file, loff_t start, loff_t end, int datasync) { diff --git a/fs/fuse/file.c b/fs/fuse/file.c index c3346b090954..deb12dc1157d 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -3540,6 +3540,18 @@ static ssize_t __fuse_copy_file_range(struct file *file_in, loff_t pos_in, bool is_unstable = (!fc->writeback_cache) && ((pos_out + len) > inode_out->i_size); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(file_in->f_inode, struct fuse_copy_file_range_io, + fuse_copy_file_range_initialize, + fuse_copy_file_range_backing, + fuse_copy_file_range_finalize, + file_in, pos_in, file_out, pos_out, len, flags); + if (fer.ret) + return PTR_ERR(fer.result); +#endif + if (fc->no_copy_file_range) return -EOPNOTSUPP; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index d15d0b896e85..5be9bebe584c 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1431,6 +1431,25 @@ int fuse_flush_backing(struct fuse_args *fa, struct file *file, fl_owner_t id); void *fuse_flush_finalize(struct fuse_args *fa, struct file *file, fl_owner_t id); +struct fuse_copy_file_range_io { + struct fuse_copy_file_range_in fci; + struct fuse_write_out fwo; +}; + +int fuse_copy_file_range_initialize(struct fuse_args *fa, + struct fuse_copy_file_range_io *fcf, + struct file *file_in, loff_t pos_in, + struct file *file_out, loff_t pos_out, + size_t len, unsigned int flags); +int fuse_copy_file_range_backing(struct fuse_args *fa, + struct file *file_in, loff_t pos_in, + struct file *file_out, loff_t pos_out, + size_t len, unsigned int flags); +void *fuse_copy_file_range_finalize(struct fuse_args *fa, + struct file *file_in, loff_t pos_in, + struct file *file_out, loff_t pos_out, + size_t len, unsigned int flags); + int fuse_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, struct file *file, loff_t start, loff_t end, int datasync); int fuse_fsync_backing(struct fuse_args *fa, From c49b8607f2fe0f0f7320e80dbd109c8b2f48544b Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Mon, 14 Mar 2022 18:21:02 -0700 Subject: [PATCH 084/148] UPSTREAM: ANDROID: fuse-bpf: support FUSE_LSEEK Adds support for lseek via fuse-bpf Bug: 224855060 Test: bpf_test_lseek Signed-off-by: Daniel Rosenberg Change-Id: Ic282940d53b9bb44a291cb3a5dfe09847b4e5c9a --- fs/fuse/backing.c | 59 +++++++++++++++++++++++++++++++++++++++++++++++ fs/fuse/file.c | 11 +++++++++ fs/fuse/fuse_i.h | 10 ++++++++ 3 files changed, 80 insertions(+) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index efd76dc08f3d..70e2fe352371 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -403,6 +403,65 @@ void *fuse_flush_finalize(struct fuse_args *fa, struct file *file, fl_owner_t id return NULL; } +int fuse_lseek_initialize(struct fuse_args *fa, struct fuse_lseek_io *flio, + struct file *file, loff_t offset, int whence) +{ + struct fuse_file *fuse_file = file->private_data; + + flio->fli = (struct fuse_lseek_in) { + .fh = fuse_file->fh, + .offset = offset, + .whence = whence, + }; + + *fa = (struct fuse_args) { + .nodeid = get_node_id(file->f_inode), + .opcode = FUSE_LSEEK, + .in_numargs = 1, + .in_args[0].size = sizeof(flio->fli), + .in_args[0].value = &flio->fli, + .out_numargs = 1, + .out_args[0].size = sizeof(flio->flo), + .out_args[0].value = &flio->flo, + }; + + return 0; +} + +int fuse_lseek_backing(struct fuse_args *fa, struct file *file, loff_t offset, int whence) +{ + const struct fuse_lseek_in *fli = fa->in_args[0].value; + struct fuse_lseek_out *flo = fa->out_args[0].value; + struct fuse_file *fuse_file = file->private_data; + struct file *backing_file = fuse_file->backing_file; + loff_t ret; + + /* TODO: Handle changing of the file handle */ + if (offset == 0) { + if (whence == SEEK_CUR) + return file->f_pos; + + if (whence == SEEK_SET) + return vfs_setpos(file, 0, 0); + } + + inode_lock(file->f_inode); + backing_file->f_pos = file->f_pos; + ret = vfs_llseek(backing_file, fli->offset, fli->whence); + flo->offset = ret; + inode_unlock(file->f_inode); + return ret; +} + +void *fuse_lseek_finalize(struct fuse_args *fa, struct file *file, loff_t offset, int whence) +{ + struct fuse_lseek_out *flo = fa->out_args[0].value; + + if (!fa->error_in) + file->f_pos = flo->offset; + return ERR_PTR(flo->offset); +} + int fuse_copy_file_range_initialize(struct fuse_args *fa, struct fuse_copy_file_range_io *fcf, struct file *file_in, loff_t pos_in, struct file *file_out, loff_t pos_out, size_t len, unsigned int flags) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index deb12dc1157d..004d81b198ac 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -2779,6 +2779,17 @@ static loff_t fuse_file_llseek(struct file *file, loff_t offset, int whence) { loff_t retval; struct inode *inode = file_inode(file); +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; + + fer = fuse_bpf_backing(inode, struct fuse_lseek_io, + fuse_lseek_initialize, + fuse_lseek_backing, + fuse_lseek_finalize, + file, offset, whence); + if (fer.ret) + return PTR_ERR(fer.result); +#endif switch (whence) { case SEEK_SET: diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 5be9bebe584c..8871b39e4a5e 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1431,6 +1431,16 @@ int fuse_flush_backing(struct fuse_args *fa, struct file *file, fl_owner_t id); void *fuse_flush_finalize(struct fuse_args *fa, struct file *file, fl_owner_t id); +struct fuse_lseek_io { + struct fuse_lseek_in fli; + struct fuse_lseek_out flo; +}; + +int fuse_lseek_initialize(struct fuse_args *fa, struct fuse_lseek_io *fli, + struct file *file, loff_t offset, int whence); +int fuse_lseek_backing(struct fuse_args *fa, struct file *file, loff_t offset, int whence); +void *fuse_lseek_finalize(struct fuse_args *fa, struct file *file, loff_t offset, int whence); + struct fuse_copy_file_range_io { struct fuse_copy_file_range_in fci; struct fuse_write_out fwo; From 671d07d583f6f2c343ff2a6b451ef96c6766bbb1 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Tue, 15 Mar 2022 17:11:54 -0700 Subject: [PATCH 085/148] UPSTREAM: ANDROID: fuse-bpf: fix special devices Removes incorrect usage of new_decode_dev Bug: 217570523 Test: generic/184 Change-Id: Ia9b85c025bb54879196545bcd4c2e42102d5a07f Signed-off-by: Daniel Rosenberg --- fs/fuse/inode.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 8025cf345196..7142a25d974f 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -315,8 +315,7 @@ static void fuse_init_inode(struct inode *inode, struct fuse_attr *attr) else if (S_ISCHR(inode->i_mode) || S_ISBLK(inode->i_mode) || S_ISFIFO(inode->i_mode) || S_ISSOCK(inode->i_mode)) { fuse_init_common(inode); - init_special_inode(inode, inode->i_mode, - new_decode_dev(attr->rdev)); + init_special_inode(inode, inode->i_mode, attr->rdev); } else BUG(); } From 472479a68a357a8ff0deb56fd005fd1859c1b72a Mon Sep 17 00:00:00 2001 From: Jiufei Xue Date: Wed, 20 Nov 2019 17:45:25 +0800 Subject: [PATCH 086/148] BACKPORT: vfs: add vfs_iocb_iter_[read|write] helper functions This doesn't cause any behavior changes and will be used by overlay async IO implementation. Remove 'kiocb_clone' from passthrough.c as this is the proper upstream commit which adds the function. Change-Id: Ic3d0ebd5cc2d945358f7295697ae02a829e2e919 Signed-off-by: Jiufei Xue Signed-off-by: Miklos Szeredi --- fs/fuse/passthrough.c | 12 ---------- fs/read_write.c | 56 +++++++++++++++++++++++++++++++++++++++++++ include/linux/fs.h | 16 +++++++++++++ 3 files changed, 72 insertions(+), 12 deletions(-) diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c index 013e2095ab9a..6031e72f3b5a 100644 --- a/fs/fuse/passthrough.c +++ b/fs/fuse/passthrough.c @@ -14,18 +14,6 @@ struct fuse_aio_req { struct kiocb *iocb_fuse; }; -static inline void kiocb_clone(struct kiocb *kiocb, struct kiocb *kiocb_src, - struct file *filp) -{ - *kiocb = (struct kiocb){ - .ki_filp = filp, - .ki_flags = kiocb_src->ki_flags, - .ki_hint = kiocb_src->ki_hint, - .ki_ioprio = kiocb_src->ki_ioprio, - .ki_pos = kiocb_src->ki_pos, - }; -} - static void fuse_file_accessed(struct file *dst_file, struct file *src_file) { struct inode *dst_inode; diff --git a/fs/read_write.c b/fs/read_write.c index 93d06b4c5f78..301c4a4ba1d0 100644 --- a/fs/read_write.c +++ b/fs/read_write.c @@ -939,6 +939,34 @@ out: return ret; } +ssize_t vfs_iocb_iter_read(struct file *file, struct kiocb *iocb, + struct iov_iter *iter) +{ + size_t tot_len; + ssize_t ret = 0; + + if (!file->f_op->read_iter) + return -EINVAL; + if (!(file->f_mode & FMODE_READ)) + return -EBADF; + if (!(file->f_mode & FMODE_CAN_READ)) + return -EINVAL; + + tot_len = iov_iter_count(iter); + if (!tot_len) + goto out; + ret = rw_verify_area(READ, file, &iocb->ki_pos, tot_len); + if (ret < 0) + return ret; + + ret = call_read_iter(file, iocb, iter); +out: + if (ret >= 0) + fsnotify_access(file); + return ret; +} +EXPORT_SYMBOL(vfs_iocb_iter_read); + ssize_t vfs_iter_read(struct file *file, struct iov_iter *iter, loff_t *ppos, rwf_t flags) { @@ -975,6 +1003,34 @@ static ssize_t do_iter_write(struct file *file, struct iov_iter *iter, return ret; } +ssize_t vfs_iocb_iter_write(struct file *file, struct kiocb *iocb, + struct iov_iter *iter) +{ + size_t tot_len; + ssize_t ret = 0; + + if (!file->f_op->write_iter) + return -EINVAL; + if (!(file->f_mode & FMODE_WRITE)) + return -EBADF; + if (!(file->f_mode & FMODE_CAN_WRITE)) + return -EINVAL; + + tot_len = iov_iter_count(iter); + if (!tot_len) + return 0; + ret = rw_verify_area(WRITE, file, &iocb->ki_pos, tot_len); + if (ret < 0) + return ret; + + ret = call_write_iter(file, iocb, iter); + if (ret > 0) + fsnotify_modify(file); + + return ret; +} +EXPORT_SYMBOL(vfs_iocb_iter_write); + ssize_t vfs_iter_write(struct file *file, struct iov_iter *iter, loff_t *ppos, rwf_t flags) { diff --git a/include/linux/fs.h b/include/linux/fs.h index 53be8d36da9c..e79b51ff5807 100644 --- a/include/linux/fs.h +++ b/include/linux/fs.h @@ -2167,6 +2167,18 @@ static inline void init_sync_kiocb(struct kiocb *kiocb, struct file *filp) }; } +static inline void kiocb_clone(struct kiocb *kiocb, struct kiocb *kiocb_src, + struct file *filp) +{ + *kiocb = (struct kiocb) { + .ki_filp = filp, + .ki_flags = kiocb_src->ki_flags, + .ki_hint = kiocb_src->ki_hint, + .ki_ioprio = kiocb_src->ki_ioprio, + .ki_pos = kiocb_src->ki_pos, + }; +} + /* * Inode state bits. Protected by inode->i_lock * @@ -3220,6 +3232,10 @@ ssize_t vfs_iter_read(struct file *file, struct iov_iter *iter, loff_t *ppos, rwf_t flags); ssize_t vfs_iter_write(struct file *file, struct iov_iter *iter, loff_t *ppos, rwf_t flags); +ssize_t vfs_iocb_iter_read(struct file *file, struct kiocb *iocb, + struct iov_iter *iter); +ssize_t vfs_iocb_iter_write(struct file *file, struct kiocb *iocb, + struct iov_iter *iter); /* fs/block_dev.c */ extern ssize_t blkdev_read_iter(struct kiocb *iocb, struct iov_iter *to); From 11a6e05d2c8e8dce46fcba09df25821e73400de3 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Wed, 16 Mar 2022 17:06:55 -0700 Subject: [PATCH 087/148] UPSTREAM: ANDROID: fuse-bpf: fix read_iter and write_iter Properly handle the async case. The existing bpf operations will likely need to be reworked. Given that they don't allow altering anything as is, this change just incrementally moves us in the right direction. Signed-off-by: Daniel Rosenberg Test: generic/467 and fuse_test Bug: 217570523 Change-Id: I31c0b48bf3d674efecad4bff4ea8b482c4e7da45 --- fs/fuse/backing.c | 149 ++++++++++++++++++++++++++++++++++++++++------ fs/fuse/fuse_i.h | 4 ++ 2 files changed, 136 insertions(+), 17 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 70e2fe352371..118bd9d12945 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -13,6 +13,33 @@ #include "../internal.h" +#define FUSE_BPF_IOCB_MASK (IOCB_APPEND | IOCB_DSYNC | IOCB_HIPRI | IOCB_NOWAIT | IOCB_SYNC) + +struct fuse_bpf_aio_req { + struct kiocb iocb; + struct kiocb *iocb_fuse; +}; + +static void fuse_file_accessed(struct file *dst_file, struct file *src_file) +{ + struct inode *dst_inode; + struct inode *src_inode; + + if (dst_file->f_flags & O_NOATIME) + return; + + dst_inode = file_inode(dst_file); + src_inode = file_inode(src_file); + + if ((!timespec64_equal(&dst_inode->i_mtime, &src_inode->i_mtime) || + !timespec64_equal(&dst_inode->i_ctime, &src_inode->i_ctime))) { + dst_inode->i_mtime = src_inode->i_mtime; + dst_inode->i_ctime = src_inode->i_ctime; + } + + touch_atime(&dst_file->f_path); +} + struct bpf_prog *fuse_get_bpf_prog(struct file *file) { struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); @@ -779,6 +806,33 @@ void *fuse_removexattr_finalize(struct fuse_args *fa, return NULL; } +static void fuse_bpf_aio_cleanup_handler(struct fuse_bpf_aio_req *aio_req) +{ + struct kiocb *iocb = &aio_req->iocb; + struct kiocb *iocb_fuse = aio_req->iocb_fuse; + + if (iocb->ki_flags & IOCB_WRITE) { + __sb_writers_acquired(file_inode(iocb->ki_filp)->i_sb, + SB_FREEZE_WRITE); + file_end_write(iocb->ki_filp); + fuse_copyattr(iocb_fuse->ki_filp, iocb->ki_filp); + } + + iocb_fuse->ki_pos = iocb->ki_pos; + kfree(aio_req); +} + +static void fuse_bpf_aio_rw_complete(struct kiocb *iocb, long res, long res2) +{ + struct fuse_bpf_aio_req *aio_req = + container_of(iocb, struct fuse_bpf_aio_req, iocb); + struct kiocb *iocb_fuse = aio_req->iocb_fuse; + + fuse_bpf_aio_cleanup_handler(aio_req); + iocb_fuse->ki_complete(iocb_fuse, res, res2); +} + + int fuse_file_read_iter_initialize( struct fuse_args *fa, struct fuse_read_in *fri, struct kiocb *iocb, struct iov_iter *to) @@ -820,17 +874,44 @@ int fuse_file_read_iter_backing(struct fuse_args *fa, { struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; - ssize_t result; + ssize_t ret; + + if (!iov_iter_count(to)) + return 0; + + if ((iocb->ki_flags & IOCB_DIRECT) && + (!ff->backing_file->f_mapping->a_ops || + !ff->backing_file->f_mapping->a_ops->direct_IO)) + return -EINVAL; /* TODO This just plain ignores any change to fuse_read_in */ - result = vfs_iter_read(ff->backing_file, to, &iocb->ki_pos, 0); + if (is_sync_kiocb(iocb)) { + ret = vfs_iter_read(ff->backing_file, to, &iocb->ki_pos, + iocb_to_rw_flags(iocb->ki_flags, FUSE_BPF_IOCB_MASK)); + } else { + struct fuse_bpf_aio_req *aio_req; - if (result < 0) - return result; + ret = -ENOMEM; + aio_req = kzalloc(sizeof(struct fuse_bpf_aio_req), GFP_KERNEL); + if (!aio_req) + goto out; + aio_req->iocb_fuse = iocb; + kiocb_clone(&aio_req->iocb, iocb, ff->backing_file); + aio_req->iocb.ki_complete = fuse_bpf_aio_rw_complete; + ret = vfs_iocb_iter_read(ff->backing_file, &aio_req->iocb, to); + if (ret != -EIOCBQUEUED) + fuse_bpf_aio_cleanup_handler(aio_req); + } + + if (ret >= 0) + fa->out_args[0].size = ret; /* TODO Need to point value at the buffer for post-modification */ - fa->out_args[0].size = result; - return result; + +out: + fuse_file_accessed(file, ff->backing_file); + + return ret; } void *fuse_file_read_iter_finalize(struct fuse_args *fa, @@ -862,8 +943,8 @@ int fuse_file_write_iter_initialize( .in_args[1].size = fwio->fwi.size, .in_args[1].value = from->kvec->iov_base, .out_numargs = 1, - .out_args[0].size = sizeof(fwio->fwo), - .out_args[0].value = &fwio->fwo, + .out_args[0].size = sizeof(fwio->fwio), + .out_args[0].value = &fwio->fwio, }; return 0; @@ -874,26 +955,60 @@ int fuse_file_write_iter_backing(struct fuse_args *fa, { struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; - struct fuse_write_out *fwo = fa->out_args[0].value; + struct fuse_write_iter_out *fwio = fa->out_args[0].value; + ssize_t ret; + + if (!iov_iter_count(from)) + return 0; /* TODO This just plain ignores any change to fuse_write_in */ - fwo->size = vfs_iter_write(ff->backing_file, from, &iocb->ki_pos, 0); + /* TODO uint32_t seems smaller than ssize_t.... right? */ + inode_lock(file_inode(file)); - /* Must reflect change in size of backing file to upper file */ - if (fwo->size > 0) - fuse_copyattr(file, ff->backing_file); + fuse_copyattr(file, ff->backing_file); - if (fwo->size < 0) - return fwo->size; + if (is_sync_kiocb(iocb)) { + file_start_write(ff->backing_file); + ret = vfs_iter_write(ff->backing_file, from, &iocb->ki_pos, + iocb_to_rw_flags(iocb->ki_flags, FUSE_BPF_IOCB_MASK)); + file_end_write(ff->backing_file); + + /* Must reflect change in size of backing file to upper file */ + if (ret > 0) + fuse_copyattr(file, ff->backing_file); + } else { + struct fuse_bpf_aio_req *aio_req; + + ret = -ENOMEM; + /* TODO get this from a cache? */ + aio_req = kzalloc(sizeof(struct fuse_bpf_aio_req), GFP_KERNEL); + if (!aio_req) + goto out; + + file_start_write(ff->backing_file); + __sb_writers_release(file_inode(ff->backing_file)->i_sb, SB_FREEZE_WRITE); + aio_req->iocb_fuse = iocb; + kiocb_clone(&aio_req->iocb, iocb, ff->backing_file); + aio_req->iocb.ki_complete = fuse_bpf_aio_rw_complete; + ret = vfs_iocb_iter_write(ff->backing_file, &aio_req->iocb, from); + if (ret != -EIOCBQUEUED) + fuse_bpf_aio_cleanup_handler(aio_req); + } + +out: + inode_unlock(file_inode(file)); + fwio->ret = ret; + if (ret < 0) + return ret; return 0; } void *fuse_file_write_iter_finalize(struct fuse_args *fa, struct kiocb *iocb, struct iov_iter *from) { - struct fuse_write_out *fwo = fa->out_args[0].value; + struct fuse_write_iter_out *fwio = fa->out_args[0].value; - return ERR_PTR(fwo->size); + return ERR_PTR(fwio->ret); } ssize_t fuse_backing_mmap(struct file *file, struct vm_area_struct *vma) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 8871b39e4a5e..e536c68f7d41 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1522,9 +1522,13 @@ int fuse_file_read_iter_backing(struct fuse_args *fa, void *fuse_file_read_iter_finalize(struct fuse_args *fa, struct kiocb *iocb, struct iov_iter *to); +struct fuse_write_iter_out { + uint64_t ret; +}; struct fuse_file_write_iter_io { struct fuse_write_in fwi; struct fuse_write_out fwo; + struct fuse_write_iter_out fwio; }; int fuse_file_write_iter_initialize( From 2921af70c5ed3cf615c1c3dff4bf39f9ddb4555d Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Mon, 11 Apr 2022 12:33:39 -0700 Subject: [PATCH 088/148] UPSTREAM: ANDROID: fuse-bpf: Fix lseek return value for offset 0 Bug: 227160050 Test: audible app now works Signed-off-by: Paul Lawrence Change-Id: Ib14765285190b5838f28c25a69c91935d02c34f4 --- fs/fuse/backing.c | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 118bd9d12945..dc11545dcd48 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -465,11 +465,15 @@ int fuse_lseek_backing(struct fuse_args *fa, struct file *file, loff_t offset, i /* TODO: Handle changing of the file handle */ if (offset == 0) { - if (whence == SEEK_CUR) - return file->f_pos; + if (whence == SEEK_CUR) { + flo->offset = file->f_pos; + return flo->offset; + } - if (whence == SEEK_SET) - return vfs_setpos(file, 0, 0); + if (whence == SEEK_SET) { + flo->offset = vfs_setpos(file, 0, 0); + return flo->offset; + } } inode_lock(file->f_inode); From f1135a3332406535140e592282ec10b903a2f283 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Wed, 23 Mar 2022 16:43:25 -0700 Subject: [PATCH 089/148] UPSTREAM: ANDROID: fuse-bpf: Fix readdir Fuse uses generic_file_llseek, so we must account for that in readdir to ensure we read from the correct offset in the lower filesystem. Bug: 226655281 Test: generic/257, fuse_test Signed-off-by: Daniel Rosenberg Change-Id: Ie752c1c645e95b7c03ef9497562758a5c42b514a --- fs/fuse/backing.c | 44 +++++++++++++++++++++++++++++++++++++------- fs/fuse/fuse_i.h | 10 +++------- fs/fuse/readdir.c | 12 ++++++++---- 3 files changed, 48 insertions(+), 18 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index dc11545dcd48..250d99af90fe 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -2189,7 +2189,7 @@ void *fuse_symlink_finalize( int fuse_readdir_initialize(struct fuse_args *fa, struct fuse_read_io *frio, struct file *file, struct dir_context *ctx, - bool *force_again, bool *allow_force) + bool *force_again, bool *allow_force, bool is_continued) { struct fuse_file *ff = file->private_data; u8 *page = (u8 *)__get_free_page(GFP_KERNEL); @@ -2259,9 +2259,35 @@ static int filldir(struct dir_context *ctx, const char *name, int namelen, return 0; } +static int parse_dirfile(char *buf, size_t nbytes, struct dir_context *ctx) +{ + while (nbytes >= FUSE_NAME_OFFSET) { + struct fuse_dirent *dirent = (struct fuse_dirent *) buf; + size_t reclen = FUSE_DIRENT_SIZE(dirent); + + if (!dirent->namelen || dirent->namelen > FUSE_NAME_MAX) + return -EIO; + if (reclen > nbytes) + break; + if (memchr(dirent->name, '/', dirent->namelen) != NULL) + return -EIO; + + ctx->pos = dirent->off; + if (!dir_emit(ctx, dirent->name, dirent->namelen, dirent->ino, + dirent->type)) + break; + + buf += reclen; + nbytes -= reclen; + } + + return 0; +} + + int fuse_readdir_backing(struct fuse_args *fa, struct file *file, struct dir_context *ctx, - bool *force_again, bool *allow_force) + bool *force_again, bool *allow_force, bool is_continued) { struct fuse_file *ff = file->private_data; struct file *backing_dir = ff->backing_file; @@ -2278,6 +2304,9 @@ int fuse_readdir_backing(struct fuse_args *fa, if (!ec.addr) return -ENOMEM; + if (!is_continued) + backing_dir->f_pos = file->f_pos; + err = iterate_dir(backing_dir, &ec.ctx); if (ec.offset == 0) *allow_force = false; @@ -2290,18 +2319,19 @@ int fuse_readdir_backing(struct fuse_args *fa, void *fuse_readdir_finalize(struct fuse_args *fa, struct file *file, struct dir_context *ctx, - bool *force_again, bool *allow_force) + bool *force_again, bool *allow_force, bool is_continued) { - int err = 0; + struct fuse_read_out *fro = fa->out_args[0].value; struct fuse_file *ff = file->private_data; struct file *backing_dir = ff->backing_file; - struct fuse_read_out *fro = fa->out_args[0].value; + int err = 0; - err = fuse_parse_dirfile(fa->out_args[1].value, - fa->out_args[1].size, file, ctx); + err = parse_dirfile(fa->out_args[1].value, fa->out_args[1].size, ctx); *force_again = !!fro->again; if (*force_again && !*allow_force) err = -EINVAL; + + ctx->pos = fro->offset; backing_dir->f_pos = fro->offset; free_page((unsigned long) fa->out_args[1].value); diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index e536c68f7d41..15c68072f90a 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -986,10 +986,6 @@ struct fuse_io_args { void fuse_read_args_fill(struct fuse_io_args *ia, struct file *file, loff_t pos, size_t count, int opcode); - -int fuse_parse_dirfile(char *buf, size_t nbytes, struct file *file, - struct dir_context *ctx); - /** * Send OPEN or OPENDIR request */ @@ -1635,13 +1631,13 @@ struct fuse_read_io { int fuse_readdir_initialize(struct fuse_args *fa, struct fuse_read_io *frio, struct file *file, struct dir_context *ctx, - bool *force_again, bool *allow_force); + bool *force_again, bool *allow_force, bool is_continued); int fuse_readdir_backing(struct fuse_args *fa, struct file *file, struct dir_context *ctx, - bool *force_again, bool *allow_force); + bool *force_again, bool *allow_force, bool is_continued); void *fuse_readdir_finalize(struct fuse_args *fa, struct file *file, struct dir_context *ctx, - bool *force_again, bool *allow_force); + bool *force_again, bool *allow_force, bool is_continued); int fuse_access_initialize(struct fuse_args *fa, struct fuse_access_in *fai, struct inode *inode, int mask); diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index 1ca60c98cf9f..be0df513a2e0 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -121,7 +121,7 @@ static bool fuse_emit(struct file *file, struct dir_context *ctx, dirent->type); } -int fuse_parse_dirfile(char *buf, size_t nbytes, struct file *file, +static int parse_dirfile(char *buf, size_t nbytes, struct file *file, struct dir_context *ctx) { while (nbytes >= FUSE_NAME_OFFSET) { @@ -368,7 +368,7 @@ static int fuse_readdir_uncached(struct file *file, struct dir_context *ctx) res = parse_dirplusfile(page_address(page), res, file, ctx, attr_version); } else { - res = fuse_parse_dirfile(page_address(page), res, file, + res = parse_dirfile(page_address(page), res, file, ctx); } } @@ -582,13 +582,17 @@ int fuse_readdir(struct file *file, struct dir_context *ctx) #ifdef CONFIG_FUSE_BPF struct fuse_err_ret fer; bool force_again, allow_force; + bool is_continued = false; + again: fer = fuse_bpf_backing(inode, struct fuse_read_io, fuse_readdir_initialize, fuse_readdir_backing, fuse_readdir_finalize, - file, ctx, &force_again, &allow_force); - if (force_again && !IS_ERR(fer.result)) + file, ctx, &force_again, &allow_force, is_continued); + if (force_again && !IS_ERR(fer.result)) { + is_continued = true; goto again; + } if (fer.ret) return PTR_ERR(fer.result); From 0ae10ca8d1cdd3becb95fc475f8c210fc3ba6a63 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Mon, 4 Apr 2022 18:11:22 -0700 Subject: [PATCH 090/148] UPSTREAM: ANDROID: fuse-bpf: Fix fixattr in rename Existing fixattr was adjusting the same node twice. Bug: 226655982 Test: generic/241 generic/269 Signed-off-by: Daniel Rosenberg Change-Id: I4b1cb6d626ee6bd9010012ac126b78f14d6157d0 --- fs/fuse/backing.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 250d99af90fe..7f089d2231e3 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1565,7 +1565,7 @@ static int fuse_rename_backing_common( if (target_inode) fsstack_copy_attr_all(target_inode, get_fuse_inode(target_inode)->backing_inode); - fsstack_copy_attr_all(newdir, d_inode(new_backing_dir_dentry)); + fsstack_copy_attr_all(d_inode(oldent), d_inode(old_backing_dentry)); unlock: unlock_rename(old_backing_dir_dentry, new_backing_dir_dentry); put_parents: From 20edac301ac488aa814877bc23778bc49288750f Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Fri, 8 Apr 2022 19:06:56 -0700 Subject: [PATCH 091/148] UPSTREAM: ANDROID: fuse-bpf: Rename iocb_fuse to iocb_orig The current name is a bit confusing. iocb_fuse could refer to the iocb passed to fuse or created by fuse. The new name unambiguously refers to the one passed in to fuse. Test: compiles, behavior unchanged Bug: 217570523 Signed-off-by: Daniel Rosenberg Change-Id: I955500eb8a3186252427fd06ca6e99b4fec469b6 --- fs/fuse/backing.c | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 7f089d2231e3..6dd34a05a464 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -17,7 +17,7 @@ struct fuse_bpf_aio_req { struct kiocb iocb; - struct kiocb *iocb_fuse; + struct kiocb *iocb_orig; }; static void fuse_file_accessed(struct file *dst_file, struct file *src_file) @@ -813,16 +813,16 @@ void *fuse_removexattr_finalize(struct fuse_args *fa, static void fuse_bpf_aio_cleanup_handler(struct fuse_bpf_aio_req *aio_req) { struct kiocb *iocb = &aio_req->iocb; - struct kiocb *iocb_fuse = aio_req->iocb_fuse; + struct kiocb *iocb_orig = aio_req->iocb_orig; if (iocb->ki_flags & IOCB_WRITE) { __sb_writers_acquired(file_inode(iocb->ki_filp)->i_sb, SB_FREEZE_WRITE); file_end_write(iocb->ki_filp); - fuse_copyattr(iocb_fuse->ki_filp, iocb->ki_filp); + fuse_copyattr(iocb_orig->ki_filp, iocb->ki_filp); } - iocb_fuse->ki_pos = iocb->ki_pos; + iocb_orig->ki_pos = iocb->ki_pos; kfree(aio_req); } @@ -830,10 +830,10 @@ static void fuse_bpf_aio_rw_complete(struct kiocb *iocb, long res, long res2) { struct fuse_bpf_aio_req *aio_req = container_of(iocb, struct fuse_bpf_aio_req, iocb); - struct kiocb *iocb_fuse = aio_req->iocb_fuse; + struct kiocb *iocb_orig = aio_req->iocb_orig; fuse_bpf_aio_cleanup_handler(aio_req); - iocb_fuse->ki_complete(iocb_fuse, res, res2); + iocb_orig->ki_complete(iocb_orig, res, res2); } @@ -899,7 +899,7 @@ int fuse_file_read_iter_backing(struct fuse_args *fa, aio_req = kzalloc(sizeof(struct fuse_bpf_aio_req), GFP_KERNEL); if (!aio_req) goto out; - aio_req->iocb_fuse = iocb; + aio_req->iocb_orig = iocb; kiocb_clone(&aio_req->iocb, iocb, ff->backing_file); aio_req->iocb.ki_complete = fuse_bpf_aio_rw_complete; ret = vfs_iocb_iter_read(ff->backing_file, &aio_req->iocb, to); @@ -991,7 +991,7 @@ int fuse_file_write_iter_backing(struct fuse_args *fa, file_start_write(ff->backing_file); __sb_writers_release(file_inode(ff->backing_file)->i_sb, SB_FREEZE_WRITE); - aio_req->iocb_fuse = iocb; + aio_req->iocb_orig = iocb; kiocb_clone(&aio_req->iocb, iocb, ff->backing_file); aio_req->iocb.ki_complete = fuse_bpf_aio_rw_complete; ret = vfs_iocb_iter_write(ff->backing_file, &aio_req->iocb, from); From 426f86751b7ea842115051030464a53c8cab4f71 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Fri, 8 Apr 2022 19:08:53 -0700 Subject: [PATCH 092/148] UPSTREAM: ANDROID: fuse-bpf: Use cache and refcount Allocates the iocb we create for asynchronous IO from a cache instead of a regular kzalloc Test: generic/467 and fuse_test Bug: 217570523 Signed-off-by: Daniel Rosenberg Change-Id: I27dcec89cd585835f6a8e80e1ae30c503f4038c8 --- fs/fuse/backing.c | 38 +++++++++++++++++++++++++++++++++----- fs/fuse/control.c | 2 +- fs/fuse/fuse_i.h | 5 ++++- fs/fuse/inode.c | 7 +++++++ 4 files changed, 45 insertions(+), 7 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 6dd34a05a464..9bc1c51f8f9e 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -17,9 +17,12 @@ struct fuse_bpf_aio_req { struct kiocb iocb; + refcount_t ref; struct kiocb *iocb_orig; }; +static struct kmem_cache *fuse_bpf_aio_request_cachep; + static void fuse_file_accessed(struct file *dst_file, struct file *src_file) { struct inode *dst_inode; @@ -810,6 +813,12 @@ void *fuse_removexattr_finalize(struct fuse_args *fa, return NULL; } +static inline void fuse_bpf_aio_put(struct fuse_bpf_aio_req *aio_req) +{ + if (refcount_dec_and_test(&aio_req->ref)) + kmem_cache_free(fuse_bpf_aio_request_cachep, aio_req); +} + static void fuse_bpf_aio_cleanup_handler(struct fuse_bpf_aio_req *aio_req) { struct kiocb *iocb = &aio_req->iocb; @@ -821,9 +830,8 @@ static void fuse_bpf_aio_cleanup_handler(struct fuse_bpf_aio_req *aio_req) file_end_write(iocb->ki_filp); fuse_copyattr(iocb_orig->ki_filp, iocb->ki_filp); } - iocb_orig->ki_pos = iocb->ki_pos; - kfree(aio_req); + fuse_bpf_aio_put(aio_req); } static void fuse_bpf_aio_rw_complete(struct kiocb *iocb, long res, long res2) @@ -896,13 +904,16 @@ int fuse_file_read_iter_backing(struct fuse_args *fa, struct fuse_bpf_aio_req *aio_req; ret = -ENOMEM; - aio_req = kzalloc(sizeof(struct fuse_bpf_aio_req), GFP_KERNEL); + aio_req = kmem_cache_zalloc(fuse_bpf_aio_request_cachep, GFP_KERNEL); if (!aio_req) goto out; + aio_req->iocb_orig = iocb; kiocb_clone(&aio_req->iocb, iocb, ff->backing_file); aio_req->iocb.ki_complete = fuse_bpf_aio_rw_complete; + refcount_set(&aio_req->ref, 2); ret = vfs_iocb_iter_read(ff->backing_file, &aio_req->iocb, to); + fuse_bpf_aio_put(aio_req); if (ret != -EIOCBQUEUED) fuse_bpf_aio_cleanup_handler(aio_req); } @@ -984,8 +995,7 @@ int fuse_file_write_iter_backing(struct fuse_args *fa, struct fuse_bpf_aio_req *aio_req; ret = -ENOMEM; - /* TODO get this from a cache? */ - aio_req = kzalloc(sizeof(struct fuse_bpf_aio_req), GFP_KERNEL); + aio_req = kmem_cache_zalloc(fuse_bpf_aio_request_cachep, GFP_KERNEL); if (!aio_req) goto out; @@ -994,7 +1004,9 @@ int fuse_file_write_iter_backing(struct fuse_args *fa, aio_req->iocb_orig = iocb; kiocb_clone(&aio_req->iocb, iocb, ff->backing_file); aio_req->iocb.ki_complete = fuse_bpf_aio_rw_complete; + refcount_set(&aio_req->ref, 2); ret = vfs_iocb_iter_write(ff->backing_file, &aio_req->iocb, from); + fuse_bpf_aio_put(aio_req); if (ret != -EIOCBQUEUED) fuse_bpf_aio_cleanup_handler(aio_req); } @@ -2369,3 +2381,19 @@ void *fuse_access_finalize(struct fuse_args *fa, struct inode *inode, int mask) { return NULL; } + +int __init fuse_bpf_init(void) +{ + fuse_bpf_aio_request_cachep = kmem_cache_create("fuse_bpf_aio_req", + sizeof(struct fuse_bpf_aio_req), + 0, SLAB_HWCACHE_ALIGN, NULL); + if (!fuse_bpf_aio_request_cachep) + return -ENOMEM; + + return 0; +} + +void __exit fuse_bpf_cleanup(void) +{ + kmem_cache_destroy(fuse_bpf_aio_request_cachep); +} diff --git a/fs/fuse/control.c b/fs/fuse/control.c index 24b4d9db231d..9b46669f8f85 100644 --- a/fs/fuse/control.c +++ b/fs/fuse/control.c @@ -395,7 +395,7 @@ int __init fuse_ctl_init(void) return register_filesystem(&fuse_ctl_fs_type); } -void __exit fuse_ctl_cleanup(void) +void fuse_ctl_cleanup(void) { unregister_filesystem(&fuse_ctl_fs_type); } diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 15c68072f90a..7d6b5ee9d1e5 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1054,7 +1054,7 @@ int fuse_dev_init(void); void fuse_dev_cleanup(void); int fuse_ctl_init(void); -void __exit fuse_ctl_cleanup(void); +void fuse_ctl_cleanup(void); /** * Simple request sending that does request allocation and freeing @@ -1778,6 +1778,9 @@ struct fuse_err_ret { bool ret; }; +int __init fuse_bpf_init(void); +void __exit fuse_bpf_cleanup(void); + /* * expression statement to wrap the backing filter logic * struct inode *inode: inode with bpf and backing inode diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 7142a25d974f..0e803d857f9f 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1911,11 +1911,17 @@ static int __init fuse_init(void) if (res) goto err_sysfs_cleanup; + res = fuse_bpf_init(); + if (res) + goto err_ctl_cleanup; + sanitize_global_limit(&max_user_bgreq); sanitize_global_limit(&max_user_congthresh); return 0; + err_ctl_cleanup: + fuse_ctl_cleanup(); err_sysfs_cleanup: fuse_sysfs_cleanup(); err_dev_cleanup: @@ -1933,6 +1939,7 @@ static void __exit fuse_exit(void) fuse_ctl_cleanup(); fuse_sysfs_cleanup(); fuse_fs_cleanup(); + fuse_bpf_cleanup(); fuse_dev_cleanup(); } From d0df659ec79e1a77802ba396f0cd407e2e63d896 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Fri, 8 Apr 2022 19:09:50 -0700 Subject: [PATCH 093/148] UPSTREAM: ANDROID: fuse-bpf: Fix read_iter We had a size mismatch for the return value, leading to EIOCBQUEUED getting interpreted as a return size instead of an error code. Test: generic/467, generic/013, and fuse_test Bug: 217570523 Signed-off-by: Daniel Rosenberg Change-Id: I64f9d5263f8b37d3c0e286467f9351997b294cc2 --- fs/fuse/backing.c | 24 +++++++++++++++--------- fs/fuse/file.c | 2 +- fs/fuse/fuse_i.h | 10 +++++++++- 3 files changed, 25 insertions(+), 11 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 9bc1c51f8f9e..fd3831472a2e 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -846,29 +846,33 @@ static void fuse_bpf_aio_rw_complete(struct kiocb *iocb, long res, long res2) int fuse_file_read_iter_initialize( - struct fuse_args *fa, struct fuse_read_in *fri, + struct fuse_args *fa, struct fuse_file_read_iter_io *fri, struct kiocb *iocb, struct iov_iter *to) { struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; - *fri = (struct fuse_read_in) { + fri->fri = (struct fuse_read_in) { .fh = ff->fh, .offset = iocb->ki_pos, .size = to->count, }; + fri->frio = (struct fuse_read_iter_out) { + .ret = fri->fri.size, + }; + /* TODO we can't assume 'to' is a kvec */ /* TODO we also can't assume the vector has only one component */ *fa = (struct fuse_args) { .opcode = FUSE_READ, .nodeid = ff->nodeid, .in_numargs = 1, - .in_args[0].size = sizeof(*fri), - .in_args[0].value = fri, + .in_args[0].size = sizeof(fri->fri), + .in_args[0].value = &fri->fri, .out_numargs = 1, - .out_args[0].size = fri->size, - .out_args[0].value = to->kvec->iov_base, + .out_args[0].size = sizeof(fri->frio), + .out_args[0].value = &fri->frio, /* * TODO Design this properly. * Possible approach: do not pass buf to bpf @@ -884,6 +888,7 @@ int fuse_file_read_iter_initialize( int fuse_file_read_iter_backing(struct fuse_args *fa, struct kiocb *iocb, struct iov_iter *to) { + struct fuse_read_iter_out *frio = fa->out_args[0].value; struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; ssize_t ret; @@ -918,8 +923,7 @@ int fuse_file_read_iter_backing(struct fuse_args *fa, fuse_bpf_aio_cleanup_handler(aio_req); } - if (ret >= 0) - fa->out_args[0].size = ret; + frio->ret = ret; /* TODO Need to point value at the buffer for post-modification */ @@ -932,7 +936,9 @@ out: void *fuse_file_read_iter_finalize(struct fuse_args *fa, struct kiocb *iocb, struct iov_iter *to) { - return ERR_PTR(fa->out_args[0].size); + struct fuse_read_iter_out *frio = fa->out_args[0].value; + + return ERR_PTR(frio->ret); } int fuse_file_write_iter_initialize( diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 004d81b198ac..04fd62f4562f 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1687,7 +1687,7 @@ static ssize_t fuse_file_read_iter(struct kiocb *iocb, struct iov_iter *to) { struct fuse_err_ret fer; - fer = fuse_bpf_backing(inode, struct fuse_read_in, + fer = fuse_bpf_backing(inode, struct fuse_file_read_iter_io, fuse_file_read_iter_initialize, fuse_file_read_iter_backing, fuse_file_read_iter_finalize, diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 7d6b5ee9d1e5..b1fb8d83117c 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1510,8 +1510,16 @@ int fuse_removexattr_backing(struct fuse_args *fa, void *fuse_removexattr_finalize(struct fuse_args *fa, struct dentry *dentry, const char *name); +struct fuse_read_iter_out { + uint64_t ret; +}; +struct fuse_file_read_iter_io { + struct fuse_read_in fri; + struct fuse_read_iter_out frio; +}; + int fuse_file_read_iter_initialize( - struct fuse_args *fa, struct fuse_read_in *fri, + struct fuse_args *fa, struct fuse_file_read_iter_io *fri, struct kiocb *iocb, struct iov_iter *to); int fuse_file_read_iter_backing(struct fuse_args *fa, struct kiocb *iocb, struct iov_iter *to); From 29efcdd0dbfd0e07a4c4c65de170c9b4a469f0bd Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Wed, 30 Mar 2022 15:19:00 -0700 Subject: [PATCH 094/148] BACKPORT: ANDROID: fuse-bpf: Use fuse_bpf_args in uapi fuse_args is not suitable for use in the uapi - it is not stable, and contains internal pointers. Replace with stable equivalent. The end_offset values are currently unused and unset, but will be used in a follow up patch by the verifier. Test: fuse_test, atest ScopedStorageDeviceTest pass Bug: 202785178 Signed-off-by: Daniel Rosenberg Change-Id: Ic1c12f9706aeae233cc30a0d68ed2533030e485b --- fs/fuse/backing.c | 394 +++++++++++++++++++++----------------- fs/fuse/fuse_i.h | 248 ++++++++++++++---------- include/linux/bpf_types.h | 2 +- include/uapi/linux/fuse.h | 50 ++--- kernel/bpf/bpf_fuse.c | 6 +- 5 files changed, 395 insertions(+), 305 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index fd3831472a2e..2959a6944c1a 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -97,7 +97,7 @@ out: return bpf_prog; } -int fuse_open_initialize(struct fuse_args *fa, struct fuse_open_io *foio, +int fuse_open_initialize(struct fuse_bpf_args *fa, struct fuse_open_io *foio, struct inode *inode, struct file *file, bool isdir) { foio->foi = (struct fuse_open_in) { @@ -106,16 +106,16 @@ int fuse_open_initialize(struct fuse_args *fa, struct fuse_open_io *foio, foio->foo = (struct fuse_open_out) {0}; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(inode)->nodeid, .opcode = isdir ? FUSE_OPENDIR : FUSE_OPEN, .in_numargs = 1, .out_numargs = 1, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(foio->foi), .value = &foio->foi, }, - .out_args[0] = (struct fuse_arg) { + .out_args[0] = (struct fuse_bpf_arg) { .size = sizeof(foio->foo), .value = &foio->foo, }, @@ -124,7 +124,7 @@ int fuse_open_initialize(struct fuse_args *fa, struct fuse_open_io *foio, return 0; } -int fuse_open_backing(struct fuse_args *fa, +int fuse_open_backing(struct fuse_bpf_args *fa, struct inode *inode, struct file *file, bool isdir) { struct fuse_mount *fm = get_fuse_mount(inode); @@ -175,7 +175,7 @@ int fuse_open_backing(struct fuse_args *fa, return 0; } -void *fuse_open_finalize(struct fuse_args *fa, +void *fuse_open_finalize(struct fuse_bpf_args *fa, struct inode *inode, struct file *file, bool isdir) { struct fuse_file *ff = file->private_data; @@ -187,7 +187,7 @@ void *fuse_open_finalize(struct fuse_args *fa, } int fuse_create_open_initialize( - struct fuse_args *fa, struct fuse_create_open_io *fcoio, + struct fuse_bpf_args *fa, struct fuse_create_open_io *fcoio, struct inode *dir, struct dentry *entry, struct file *file, unsigned int flags, umode_t mode) { @@ -199,24 +199,24 @@ int fuse_create_open_initialize( fcoio->feo = (struct fuse_entry_out) {0}; fcoio->foo = (struct fuse_open_out) {0}; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(dir), .opcode = FUSE_CREATE, .in_numargs = 2, .out_numargs = 2, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(fcoio->fci), .value = &fcoio->fci, }, - .in_args[1] = (struct fuse_in_arg) { + .in_args[1] = (struct fuse_bpf_in_arg) { .size = entry->d_name.len + 1, .value = entry->d_name.name, }, - .out_args[0] = (struct fuse_arg) { + .out_args[0] = (struct fuse_bpf_arg) { .size = sizeof(fcoio->feo), .value = &fcoio->feo, }, - .out_args[1] = (struct fuse_arg) { + .out_args[1] = (struct fuse_bpf_arg) { .size = sizeof(fcoio->foo), .value = &fcoio->foo, }, @@ -251,7 +251,7 @@ static int fuse_open_file_backing(struct inode *inode, struct file *file) } int fuse_create_open_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, struct file *file, unsigned int flags, umode_t mode) { @@ -321,7 +321,7 @@ out: } void *fuse_create_open_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, struct file *file, unsigned int flags, umode_t mode) { @@ -337,7 +337,7 @@ void *fuse_create_open_finalize( return 0; } -int fuse_release_initialize(struct fuse_args *fa, struct fuse_release_in *fri, +int fuse_release_initialize(struct fuse_bpf_args *fa, struct fuse_release_in *fri, struct inode *inode, struct file *file) { struct fuse_file *fuse_file = file->private_data; @@ -349,7 +349,7 @@ int fuse_release_initialize(struct fuse_args *fa, struct fuse_release_in *fri, .fh = ((struct fuse_file *)(file->private_data))->fh, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(inode)->nodeid, .opcode = FUSE_RELEASE, .in_numargs = 1, @@ -360,7 +360,7 @@ int fuse_release_initialize(struct fuse_args *fa, struct fuse_release_in *fri, return 0; } -int fuse_releasedir_initialize(struct fuse_args *fa, +int fuse_releasedir_initialize(struct fuse_bpf_args *fa, struct fuse_release_in *fri, struct inode *inode, struct file *file) { @@ -373,7 +373,7 @@ int fuse_releasedir_initialize(struct fuse_args *fa, .fh = ((struct fuse_file *)(file->private_data))->fh, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(inode)->nodeid, .opcode = FUSE_RELEASEDIR, .in_numargs = 1, @@ -384,20 +384,20 @@ int fuse_releasedir_initialize(struct fuse_args *fa, return 0; } -int fuse_release_backing(struct fuse_args *fa, +int fuse_release_backing(struct fuse_bpf_args *fa, struct inode *inode, struct file *file) { return 0; } -void *fuse_release_finalize(struct fuse_args *fa, +void *fuse_release_finalize(struct fuse_bpf_args *fa, struct inode *inode, struct file *file) { fuse_file_free(file->private_data); return NULL; } -int fuse_flush_initialize(struct fuse_args *fa, struct fuse_flush_in *ffi, +int fuse_flush_initialize(struct fuse_bpf_args *fa, struct fuse_flush_in *ffi, struct file *file, fl_owner_t id) { struct fuse_file *fuse_file = file->private_data; @@ -406,19 +406,19 @@ int fuse_flush_initialize(struct fuse_args *fa, struct fuse_flush_in *ffi, .fh = fuse_file->fh, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(file->f_inode), .opcode = FUSE_FLUSH, .in_numargs = 1, .in_args[0].size = sizeof(*ffi), .in_args[0].value = ffi, - .force = true, + .flags = FUSE_BPF_FORCE, }; return 0; } -int fuse_flush_backing(struct fuse_args *fa, struct file *file, fl_owner_t id) +int fuse_flush_backing(struct fuse_bpf_args *fa, struct file *file, fl_owner_t id) { struct fuse_file *fuse_file = file->private_data; struct file *backing_file = fuse_file->backing_file; @@ -428,12 +428,12 @@ int fuse_flush_backing(struct fuse_args *fa, struct file *file, fl_owner_t id) return 0; } -void *fuse_flush_finalize(struct fuse_args *fa, struct file *file, fl_owner_t id) +void *fuse_flush_finalize(struct fuse_bpf_args *fa, struct file *file, fl_owner_t id) { return NULL; } -int fuse_lseek_initialize(struct fuse_args *fa, struct fuse_lseek_io *flio, +int fuse_lseek_initialize(struct fuse_bpf_args *fa, struct fuse_lseek_io *flio, struct file *file, loff_t offset, int whence) { struct fuse_file *fuse_file = file->private_data; @@ -444,7 +444,7 @@ int fuse_lseek_initialize(struct fuse_args *fa, struct fuse_lseek_io *flio, .whence = whence, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(file->f_inode), .opcode = FUSE_LSEEK, .in_numargs = 1, @@ -458,7 +458,7 @@ int fuse_lseek_initialize(struct fuse_args *fa, struct fuse_lseek_io *flio, return 0; } -int fuse_lseek_backing(struct fuse_args *fa, struct file *file, loff_t offset, int whence) +int fuse_lseek_backing(struct fuse_bpf_args *fa, struct file *file, loff_t offset, int whence) { const struct fuse_lseek_in *fli = fa->in_args[0].value; struct fuse_lseek_out *flo = fa->out_args[0].value; @@ -487,7 +487,7 @@ int fuse_lseek_backing(struct fuse_args *fa, struct file *file, loff_t offset, i return ret; } -void *fuse_lseek_finalize(struct fuse_args *fa, struct file *file, loff_t offset, int whence) +void *fuse_lseek_finalize(struct fuse_bpf_args *fa, struct file *file, loff_t offset, int whence) { struct fuse_lseek_out *flo = fa->out_args[0].value; @@ -496,7 +496,7 @@ void *fuse_lseek_finalize(struct fuse_args *fa, struct file *file, loff_t offset return ERR_PTR(flo->offset); } -int fuse_copy_file_range_initialize(struct fuse_args *fa, struct fuse_copy_file_range_io *fcf, +int fuse_copy_file_range_initialize(struct fuse_bpf_args *fa, struct fuse_copy_file_range_io *fcf, struct file *file_in, loff_t pos_in, struct file *file_out, loff_t pos_out, size_t len, unsigned int flags) { @@ -514,7 +514,7 @@ int fuse_copy_file_range_initialize(struct fuse_args *fa, struct fuse_copy_file_ .flags = flags, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(file_in->f_inode), .opcode = FUSE_COPY_FILE_RANGE, .in_numargs = 1, @@ -528,7 +528,7 @@ int fuse_copy_file_range_initialize(struct fuse_args *fa, struct fuse_copy_file_ return 0; } -int fuse_copy_file_range_backing(struct fuse_args *fa, struct file *file_in, loff_t pos_in, +int fuse_copy_file_range_backing(struct fuse_bpf_args *fa, struct file *file_in, loff_t pos_in, struct file *file_out, loff_t pos_out, size_t len, unsigned int flags) { @@ -547,14 +547,14 @@ int fuse_copy_file_range_backing(struct fuse_args *fa, struct file *file_in, lof flags); } -void *fuse_copy_file_range_finalize(struct fuse_args *fa, struct file *file_in, loff_t pos_in, +void *fuse_copy_file_range_finalize(struct fuse_bpf_args *fa, struct file *file_in, loff_t pos_in, struct file *file_out, loff_t pos_out, size_t len, unsigned int flags) { return NULL; } -int fuse_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, +int fuse_fsync_initialize(struct fuse_bpf_args *fa, struct fuse_fsync_in *ffi, struct file *file, loff_t start, loff_t end, int datasync) { struct fuse_file *fuse_file = file->private_data; @@ -564,19 +564,19 @@ int fuse_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, .fsync_flags = datasync ? FUSE_FSYNC_FDATASYNC : 0, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(file->f_inode)->nodeid, .opcode = FUSE_FSYNC, .in_numargs = 1, .in_args[0].size = sizeof(*ffi), .in_args[0].value = ffi, - .force = true, + .flags = FUSE_BPF_FORCE, }; return 0; } -int fuse_fsync_backing(struct fuse_args *fa, +int fuse_fsync_backing(struct fuse_bpf_args *fa, struct file *file, loff_t start, loff_t end, int datasync) { struct fuse_file *fuse_file = file->private_data; @@ -587,13 +587,13 @@ int fuse_fsync_backing(struct fuse_args *fa, return vfs_fsync(backing_file, new_datasync); } -void *fuse_fsync_finalize(struct fuse_args *fa, +void *fuse_fsync_finalize(struct fuse_bpf_args *fa, struct file *file, loff_t start, loff_t end, int datasync) { return NULL; } -int fuse_dir_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, +int fuse_dir_fsync_initialize(struct fuse_bpf_args *fa, struct fuse_fsync_in *ffi, struct file *file, loff_t start, loff_t end, int datasync) { struct fuse_file *fuse_file = file->private_data; @@ -603,19 +603,19 @@ int fuse_dir_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, .fsync_flags = datasync ? FUSE_FSYNC_FDATASYNC : 0, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(file->f_inode)->nodeid, .opcode = FUSE_FSYNCDIR, .in_numargs = 1, .in_args[0].size = sizeof(*ffi), .in_args[0].value = ffi, - .force = true, + .flags = FUSE_BPF_FORCE, }; return 0; } -int fuse_getxattr_initialize(struct fuse_args *fa, +int fuse_getxattr_initialize(struct fuse_bpf_args *fa, struct fuse_getxattr_io *fgio, struct dentry *dentry, const char *name, void *value, size_t size) @@ -624,35 +624,34 @@ int fuse_getxattr_initialize(struct fuse_args *fa, .fgi.size = size, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(dentry->d_inode)->nodeid, .opcode = FUSE_GETXATTR, .in_numargs = 2, .out_numargs = 1, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(fgio->fgi), .value = &fgio->fgi, }, - .in_args[1] = (struct fuse_in_arg) { + .in_args[1] = (struct fuse_bpf_in_arg) { .size = strlen(name) + 1, .value = name, }, - .out_argvar = size ? true : false, + .flags = size ? FUSE_BPF_OUT_ARGVAR : 0, .out_args[0].size = size ? size : sizeof(fgio->fgo), .out_args[0].value = size ? value : &fgio->fgo, }; - return 0; } -int fuse_getxattr_backing(struct fuse_args *fa, +int fuse_getxattr_backing(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name, void *value, size_t size) { ssize_t ret = vfs_getxattr(get_fuse_dentry(dentry)->backing_path.dentry, fa->in_args[1].value, value, size); - if (fa->out_argvar) + if (fa->flags & FUSE_BPF_OUT_ARGVAR) fa->out_args[0].size = ret; else ((struct fuse_getxattr_out *)fa->out_args[0].value)->size = ret; @@ -660,13 +659,13 @@ int fuse_getxattr_backing(struct fuse_args *fa, return 0; } -void *fuse_getxattr_finalize(struct fuse_args *fa, +void *fuse_getxattr_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name, void *value, size_t size) { struct fuse_getxattr_out *fgo; - if (fa->out_argvar) + if (fa->flags & FUSE_BPF_OUT_ARGVAR) return ERR_PTR(fa->out_args[0].size); fgo = fa->out_args[0].value; @@ -675,7 +674,7 @@ void *fuse_getxattr_finalize(struct fuse_args *fa, } -int fuse_listxattr_initialize(struct fuse_args *fa, +int fuse_listxattr_initialize(struct fuse_bpf_args *fa, struct fuse_getxattr_io *fgio, struct dentry *dentry, char *list, size_t size) { @@ -683,17 +682,17 @@ int fuse_listxattr_initialize(struct fuse_args *fa, .fgi.size = size, }; - *fa = (struct fuse_args){ + *fa = (struct fuse_bpf_args){ .nodeid = get_fuse_inode(dentry->d_inode)->nodeid, .opcode = FUSE_LISTXATTR, .in_numargs = 1, .out_numargs = 1, .in_args[0] = - (struct fuse_in_arg){ + (struct fuse_bpf_in_arg){ .size = sizeof(fgio->fgi), .value = &fgio->fgi, }, - .out_argvar = size ? true : false, + .flags = size ? FUSE_BPF_OUT_ARGVAR : 0, .out_args[0].size = size ? size : sizeof(fgio->fgo), .out_args[0].value = size ? (void *)list : &fgio->fgo, }; @@ -701,7 +700,7 @@ int fuse_listxattr_initialize(struct fuse_args *fa, return 0; } -int fuse_listxattr_backing(struct fuse_args *fa, struct dentry *dentry, +int fuse_listxattr_backing(struct fuse_bpf_args *fa, struct dentry *dentry, char *list, size_t size) { ssize_t ret = @@ -711,7 +710,7 @@ int fuse_listxattr_backing(struct fuse_args *fa, struct dentry *dentry, if (ret < 0) return ret; - if (fa->out_argvar) + if (fa->flags & FUSE_BPF_OUT_ARGVAR) fa->out_args[0].size = ret; else ((struct fuse_getxattr_out *)fa->out_args[0].value)->size = ret; @@ -719,7 +718,7 @@ int fuse_listxattr_backing(struct fuse_args *fa, struct dentry *dentry, return ret; } -void *fuse_listxattr_finalize(struct fuse_args *fa, struct dentry *dentry, +void *fuse_listxattr_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, char *list, size_t size) { struct fuse_getxattr_out *fgo; @@ -727,14 +726,14 @@ void *fuse_listxattr_finalize(struct fuse_args *fa, struct dentry *dentry, if (fa->error_in) return NULL; - if (fa->out_argvar) + if (fa->flags & FUSE_BPF_OUT_ARGVAR) return ERR_PTR(fa->out_args[0].size); fgo = fa->out_args[0].value; return ERR_PTR(fgo->size); } -int fuse_setxattr_initialize(struct fuse_args *fa, +int fuse_setxattr_initialize(struct fuse_bpf_args *fa, struct fuse_setxattr_in *fsxi, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) @@ -744,19 +743,19 @@ int fuse_setxattr_initialize(struct fuse_args *fa, .flags = flags, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(dentry->d_inode)->nodeid, .opcode = FUSE_SETXATTR, .in_numargs = 3, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(*fsxi), .value = fsxi, }, - .in_args[1] = (struct fuse_in_arg) { + .in_args[1] = (struct fuse_bpf_in_arg) { .size = strlen(name) + 1, .value = name, }, - .in_args[2] = (struct fuse_in_arg) { + .in_args[2] = (struct fuse_bpf_in_arg) { .size = size, .value = value, }, @@ -765,7 +764,7 @@ int fuse_setxattr_initialize(struct fuse_args *fa, return 0; } -int fuse_setxattr_backing(struct fuse_args *fa, struct dentry *dentry, +int fuse_setxattr_backing(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { @@ -773,22 +772,22 @@ int fuse_setxattr_backing(struct fuse_args *fa, struct dentry *dentry, value, size, flags); } -void *fuse_setxattr_finalize(struct fuse_args *fa, struct dentry *dentry, +void *fuse_setxattr_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { return NULL; } -int fuse_removexattr_initialize(struct fuse_args *fa, +int fuse_removexattr_initialize(struct fuse_bpf_args *fa, struct fuse_dummy_io *unused, struct dentry *dentry, const char *name) { - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(dentry->d_inode)->nodeid, .opcode = FUSE_REMOVEXATTR, .in_numargs = 1, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = strlen(name) + 1, .value = name, }, @@ -797,7 +796,7 @@ int fuse_removexattr_initialize(struct fuse_args *fa, return 0; } -int fuse_removexattr_backing(struct fuse_args *fa, +int fuse_removexattr_backing(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name) { struct path *backing_path = @@ -807,7 +806,7 @@ int fuse_removexattr_backing(struct fuse_args *fa, return vfs_removexattr(backing_path->dentry, name); } -void *fuse_removexattr_finalize(struct fuse_args *fa, +void *fuse_removexattr_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name) { return NULL; @@ -846,7 +845,7 @@ static void fuse_bpf_aio_rw_complete(struct kiocb *iocb, long res, long res2) int fuse_file_read_iter_initialize( - struct fuse_args *fa, struct fuse_file_read_iter_io *fri, + struct fuse_bpf_args *fa, struct fuse_file_read_iter_io *fri, struct kiocb *iocb, struct iov_iter *to) { struct file *file = iocb->ki_filp; @@ -864,7 +863,7 @@ int fuse_file_read_iter_initialize( /* TODO we can't assume 'to' is a kvec */ /* TODO we also can't assume the vector has only one component */ - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .opcode = FUSE_READ, .nodeid = ff->nodeid, .in_numargs = 1, @@ -885,7 +884,7 @@ int fuse_file_read_iter_initialize( return 0; } -int fuse_file_read_iter_backing(struct fuse_args *fa, +int fuse_file_read_iter_backing(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *to) { struct fuse_read_iter_out *frio = fa->out_args[0].value; @@ -933,7 +932,7 @@ out: return ret; } -void *fuse_file_read_iter_finalize(struct fuse_args *fa, +void *fuse_file_read_iter_finalize(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *to) { struct fuse_read_iter_out *frio = fa->out_args[0].value; @@ -942,7 +941,7 @@ void *fuse_file_read_iter_finalize(struct fuse_args *fa, } int fuse_file_write_iter_initialize( - struct fuse_args *fa, struct fuse_file_write_iter_io *fwio, + struct fuse_bpf_args *fa, struct fuse_file_write_iter_io *fwio, struct kiocb *iocb, struct iov_iter *from) { struct file *file = iocb->ki_filp; @@ -955,7 +954,7 @@ int fuse_file_write_iter_initialize( }; /* TODO we can't assume 'from' is a kvec */ - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .opcode = FUSE_WRITE, .nodeid = ff->nodeid, .in_numargs = 2, @@ -971,7 +970,7 @@ int fuse_file_write_iter_initialize( return 0; } -int fuse_file_write_iter_backing(struct fuse_args *fa, +int fuse_file_write_iter_backing(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *from) { struct file *file = iocb->ki_filp; @@ -1025,7 +1024,7 @@ out: return 0; } -void *fuse_file_write_iter_finalize(struct fuse_args *fa, +void *fuse_file_write_iter_finalize(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *from) { struct fuse_write_iter_out *fwio = fa->out_args[0].value; @@ -1071,7 +1070,7 @@ ssize_t fuse_backing_mmap(struct file *file, struct vm_area_struct *vma) return ret; } -int fuse_file_fallocate_initialize(struct fuse_args *fa, +int fuse_file_fallocate_initialize(struct fuse_bpf_args *fa, struct fuse_fallocate_in *ffi, struct file *file, int mode, loff_t offset, loff_t length) { @@ -1084,7 +1083,7 @@ int fuse_file_fallocate_initialize(struct fuse_args *fa, .mode = mode }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .opcode = FUSE_FALLOCATE, .nodeid = ff->nodeid, .in_numargs = 1, @@ -1095,7 +1094,7 @@ int fuse_file_fallocate_initialize(struct fuse_args *fa, return 0; } -int fuse_file_fallocate_backing(struct fuse_args *fa, +int fuse_file_fallocate_backing(struct fuse_bpf_args *fa, struct file *file, int mode, loff_t offset, loff_t length) { const struct fuse_fallocate_in *ffi = fa->in_args[0].value; @@ -1105,7 +1104,7 @@ int fuse_file_fallocate_backing(struct fuse_args *fa, ffi->length); } -void *fuse_file_fallocate_finalize(struct fuse_args *fa, +void *fuse_file_fallocate_finalize(struct fuse_bpf_args *fa, struct file *file, int mode, loff_t offset, loff_t length) { return NULL; @@ -1115,24 +1114,24 @@ void *fuse_file_fallocate_finalize(struct fuse_args *fa, * Directory operations after here * ******************************************************************************/ -int fuse_lookup_initialize(struct fuse_args *fa, struct fuse_lookup_io *fli, +int fuse_lookup_initialize(struct fuse_bpf_args *fa, struct fuse_lookup_io *fli, struct inode *dir, struct dentry *entry, unsigned int flags) { - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(dir)->nodeid, .opcode = FUSE_LOOKUP, .in_numargs = 1, .out_numargs = 2, - .out_argvar = 1, - .in_args[0] = (struct fuse_in_arg) { + .flags = FUSE_BPF_OUT_ARGVAR, + .in_args[0] = (struct fuse_bpf_in_arg) { .size = entry->d_name.len + 1, .value = entry->d_name.name, }, - .out_args[0] = (struct fuse_arg) { + .out_args[0] = (struct fuse_bpf_arg) { .size = sizeof(fli->feo), .value = &fli->feo, }, - .out_args[1] = (struct fuse_arg) { + .out_args[1] = (struct fuse_bpf_arg) { .size = sizeof(fli->feb.out), .value = &fli->feb.out, }, @@ -1141,7 +1140,7 @@ int fuse_lookup_initialize(struct fuse_args *fa, struct fuse_lookup_io *fli, return 0; } -int fuse_lookup_backing(struct fuse_args *fa, struct inode *dir, +int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags) { struct fuse_dentry *fuse_entry = get_fuse_dentry(entry); @@ -1168,7 +1167,7 @@ int fuse_lookup_backing(struct fuse_args *fa, struct inode *dir, return 0; } -struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, +struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags) { struct fuse_dentry *fd; @@ -1271,7 +1270,7 @@ struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, return d_splice_alias(inode, entry); } -int fuse_revalidate_backing(struct fuse_args *fa, struct inode *dir, +int fuse_revalidate_backing(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags) { struct fuse_dentry *fuse_dentry = get_fuse_dentry(entry); @@ -1289,13 +1288,13 @@ int fuse_revalidate_backing(struct fuse_args *fa, struct inode *dir, return 1; } -void *fuse_revalidate_finalize(struct fuse_args *fa, struct inode *dir, +void *fuse_revalidate_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags) { return 0; } -int fuse_canonical_path_initialize(struct fuse_args *fa, +int fuse_canonical_path_initialize(struct fuse_bpf_args *fa, struct fuse_dummy_io *fdi, const struct path *path, struct path *canonical_path) @@ -1304,14 +1303,14 @@ int fuse_canonical_path_initialize(struct fuse_args *fa, return 0; } -int fuse_canonical_path_backing(struct fuse_args *fa, const struct path *path, +int fuse_canonical_path_backing(struct fuse_bpf_args *fa, const struct path *path, struct path *canonical_path) { get_fuse_backing_path(path->dentry, canonical_path); return 0; } -void *fuse_canonical_path_finalize(struct fuse_args *fa, +void *fuse_canonical_path_finalize(struct fuse_bpf_args *fa, const struct path *path, struct path *canonical_path) { @@ -1319,7 +1318,7 @@ void *fuse_canonical_path_finalize(struct fuse_args *fa, } int fuse_mknod_initialize( - struct fuse_args *fa, struct fuse_mknod_in *fmi, + struct fuse_bpf_args *fa, struct fuse_mknod_in *fmi, struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) { *fmi = (struct fuse_mknod_in) { @@ -1327,15 +1326,15 @@ int fuse_mknod_initialize( .rdev = new_encode_dev(rdev), .umask = current_umask(), }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(dir), .opcode = FUSE_MKNOD, .in_numargs = 2, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(*fmi), .value = fmi, }, - .in_args[1] = (struct fuse_in_arg) { + .in_args[1] = (struct fuse_bpf_in_arg) { .size = entry->d_name.len + 1, .value = entry->d_name.name, }, @@ -1345,7 +1344,7 @@ int fuse_mknod_initialize( } int fuse_mknod_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) { int err = 0; @@ -1389,29 +1388,29 @@ out: } void *fuse_mknod_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) { return NULL; } int fuse_mkdir_initialize( - struct fuse_args *fa, struct fuse_mkdir_in *fmi, + struct fuse_bpf_args *fa, struct fuse_mkdir_in *fmi, struct inode *dir, struct dentry *entry, umode_t mode) { *fmi = (struct fuse_mkdir_in) { .mode = mode, .umask = current_umask(), }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(dir), .opcode = FUSE_MKDIR, .in_numargs = 2, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(*fmi), .value = fmi, }, - .in_args[1] = (struct fuse_in_arg) { + .in_args[1] = (struct fuse_bpf_in_arg) { .size = entry->d_name.len + 1, .value = entry->d_name.name, }, @@ -1421,7 +1420,7 @@ int fuse_mkdir_initialize( } int fuse_mkdir_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, umode_t mode) { int err = 0; @@ -1467,21 +1466,21 @@ out: } void *fuse_mkdir_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, umode_t mode) { return NULL; } int fuse_rmdir_initialize( - struct fuse_args *fa, struct fuse_dummy_io *dummy, + struct fuse_bpf_args *fa, struct fuse_dummy_io *dummy, struct inode *dir, struct dentry *entry) { - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(dir), .opcode = FUSE_RMDIR, .in_numargs = 1, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = entry->d_name.len + 1, .value = entry->d_name.name, }, @@ -1491,7 +1490,7 @@ int fuse_rmdir_initialize( } int fuse_rmdir_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry) { int err = 0; @@ -1520,7 +1519,7 @@ int fuse_rmdir_backing( } void *fuse_rmdir_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry) { return NULL; @@ -1596,7 +1595,7 @@ put_old_path: return err; } -int fuse_rename2_initialize(struct fuse_args *fa, struct fuse_rename2_in *fri, +int fuse_rename2_initialize(struct fuse_bpf_args *fa, struct fuse_rename2_in *fri, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent, unsigned int flags) @@ -1605,19 +1604,19 @@ int fuse_rename2_initialize(struct fuse_args *fa, struct fuse_rename2_in *fri, .newdir = get_node_id(newdir), .flags = flags, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(olddir), .opcode = FUSE_RENAME2, .in_numargs = 3, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(*fri), .value = fri, }, - .in_args[1] = (struct fuse_in_arg) { + .in_args[1] = (struct fuse_bpf_in_arg) { .size = oldent->d_name.len + 1, .value = oldent->d_name.name, }, - .in_args[2] = (struct fuse_in_arg) { + .in_args[2] = (struct fuse_bpf_in_arg) { .size = newent->d_name.len + 1, .value = newent->d_name.name, }, @@ -1626,7 +1625,7 @@ int fuse_rename2_initialize(struct fuse_args *fa, struct fuse_rename2_in *fri, return 0; } -int fuse_rename2_backing(struct fuse_args *fa, +int fuse_rename2_backing(struct fuse_bpf_args *fa, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent, unsigned int flags) @@ -1637,7 +1636,7 @@ int fuse_rename2_backing(struct fuse_args *fa, return fuse_rename_backing_common(olddir, oldent, newdir, newent, fri->flags); } -void *fuse_rename2_finalize(struct fuse_args *fa, +void *fuse_rename2_finalize(struct fuse_bpf_args *fa, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent, unsigned int flags) @@ -1645,26 +1644,26 @@ void *fuse_rename2_finalize(struct fuse_args *fa, return NULL; } -int fuse_rename_initialize(struct fuse_args *fa, struct fuse_rename_in *fri, +int fuse_rename_initialize(struct fuse_bpf_args *fa, struct fuse_rename_in *fri, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent) { *fri = (struct fuse_rename_in) { .newdir = get_node_id(newdir), }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(olddir), .opcode = FUSE_RENAME, .in_numargs = 3, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(*fri), .value = fri, }, - .in_args[1] = (struct fuse_in_arg) { + .in_args[1] = (struct fuse_bpf_in_arg) { .size = oldent->d_name.len + 1, .value = oldent->d_name.name, }, - .in_args[2] = (struct fuse_in_arg) { + .in_args[2] = (struct fuse_bpf_in_arg) { .size = newent->d_name.len + 1, .value = newent->d_name.name, }, @@ -1673,7 +1672,7 @@ int fuse_rename_initialize(struct fuse_args *fa, struct fuse_rename_in *fri, return 0; } -int fuse_rename_backing(struct fuse_args *fa, +int fuse_rename_backing(struct fuse_bpf_args *fa, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent) { @@ -1681,7 +1680,7 @@ int fuse_rename_backing(struct fuse_args *fa, return fuse_rename_backing_common(olddir, oldent, newdir, newent, 0); } -void *fuse_rename_finalize(struct fuse_args *fa, +void *fuse_rename_finalize(struct fuse_bpf_args *fa, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent) { @@ -1689,14 +1688,14 @@ void *fuse_rename_finalize(struct fuse_args *fa, } int fuse_unlink_initialize( - struct fuse_args *fa, struct fuse_dummy_io *dummy, + struct fuse_bpf_args *fa, struct fuse_dummy_io *dummy, struct inode *dir, struct dentry *entry) { - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(dir), .opcode = FUSE_UNLINK, .in_numargs = 1, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = entry->d_name.len + 1, .value = entry->d_name.name, }, @@ -1706,7 +1705,7 @@ int fuse_unlink_initialize( } int fuse_unlink_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry) { int err = 0; @@ -1735,13 +1734,13 @@ int fuse_unlink_backing( } void *fuse_unlink_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry) { return NULL; } -int fuse_link_initialize(struct fuse_args *fa, struct fuse_link_in *fli, +int fuse_link_initialize(struct fuse_bpf_args *fa, struct fuse_link_in *fli, struct dentry *entry, struct inode *dir, struct dentry *newent) { @@ -1761,7 +1760,7 @@ int fuse_link_initialize(struct fuse_args *fa, struct fuse_link_in *fli, return 0; } -int fuse_link_backing(struct fuse_args *fa, struct dentry *entry, +int fuse_link_backing(struct fuse_bpf_args *fa, struct dentry *entry, struct inode *dir, struct dentry *newent) { int err = 0; @@ -1815,13 +1814,13 @@ err_dst_path: return err; } -void *fuse_link_finalize(struct fuse_args *fa, struct dentry *entry, +void *fuse_link_finalize(struct fuse_bpf_args *fa, struct dentry *entry, struct inode *dir, struct dentry *newent) { return NULL; } -int fuse_getattr_initialize(struct fuse_args *fa, struct fuse_getattr_io *fgio, +int fuse_getattr_initialize(struct fuse_bpf_args *fa, struct fuse_getattr_io *fgio, const struct dentry *entry, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -1832,16 +1831,16 @@ int fuse_getattr_initialize(struct fuse_args *fa, struct fuse_getattr_io *fgio, fgio->fao = (struct fuse_attr_out) {0}; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(entry->d_inode), .opcode = FUSE_GETATTR, .in_numargs = 1, .out_numargs = 1, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(fgio->fgi), .value = &fgio->fgi, }, - .out_args[0] = (struct fuse_arg) { + .out_args[0] = (struct fuse_bpf_arg) { .size = sizeof(fgio->fao), .value = &fgio->fao, }, @@ -1886,7 +1885,7 @@ static void fuse_stat_to_attr(struct fuse_conn *fc, struct inode *inode, attr->blksize = 1 << blkbits; } -int fuse_getattr_backing(struct fuse_args *fa, +int fuse_getattr_backing(struct fuse_bpf_args *fa, const struct dentry *entry, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -1909,7 +1908,7 @@ int fuse_getattr_backing(struct fuse_args *fa, return err; } -void *fuse_getattr_finalize(struct fuse_args *fa, +void *fuse_getattr_finalize(struct fuse_bpf_args *fa, const struct dentry *entry, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -1964,7 +1963,7 @@ static void fattr_to_iattr(struct fuse_conn *fc, } } -int fuse_setattr_initialize(struct fuse_args *fa, struct fuse_setattr_io *fsio, +int fuse_setattr_initialize(struct fuse_bpf_args *fa, struct fuse_setattr_io *fsio, struct dentry *dentry, struct iattr *attr, struct file *file) { struct fuse_conn *fc = get_fuse_conn(dentry->d_inode); @@ -1972,7 +1971,7 @@ int fuse_setattr_initialize(struct fuse_args *fa, struct fuse_setattr_io *fsio, *fsio = (struct fuse_setattr_io) {0}; iattr_to_fattr(fc, attr, &fsio->fsi, true); - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .opcode = FUSE_SETATTR, .nodeid = get_node_id(dentry->d_inode), .in_numargs = 1, @@ -1986,7 +1985,7 @@ int fuse_setattr_initialize(struct fuse_args *fa, struct fuse_setattr_io *fsio, return 0; } -int fuse_setattr_backing(struct fuse_args *fa, +int fuse_setattr_backing(struct fuse_bpf_args *fa, struct dentry *dentry, struct iattr *attr, struct file *file) { struct fuse_conn *fc = get_fuse_conn(dentry->d_inode); @@ -2013,18 +2012,18 @@ int fuse_setattr_backing(struct fuse_args *fa, return res; } -void *fuse_setattr_finalize(struct fuse_args *fa, +void *fuse_setattr_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, struct iattr *attr, struct file *file) { return NULL; } int fuse_statfs_initialize( - struct fuse_args *fa, struct fuse_statfs_out *fso, + struct fuse_bpf_args *fa, struct fuse_statfs_out *fso, struct dentry *dentry, struct kstatfs *buf) { *fso = (struct fuse_statfs_out) {0}; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(d_inode(dentry)), .opcode = FUSE_STATFS, .out_numargs = 1, @@ -2037,7 +2036,7 @@ int fuse_statfs_initialize( } int fuse_statfs_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct dentry *dentry, struct kstatfs *buf) { int err = 0; @@ -2059,7 +2058,7 @@ int fuse_statfs_backing( } void *fuse_statfs_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct dentry *dentry, struct kstatfs *buf) { struct fuse_statfs_out *fso = fa->out_args[0].value; @@ -2069,7 +2068,7 @@ void *fuse_statfs_finalize( return NULL; } -int fuse_get_link_initialize(struct fuse_args *fa, struct fuse_dummy_io *unused, +int fuse_get_link_initialize(struct fuse_bpf_args *fa, struct fuse_dummy_io *unused, struct inode *inode, struct dentry *dentry, struct delayed_call *callback, const char **out) { @@ -2085,11 +2084,11 @@ int fuse_get_link_initialize(struct fuse_args *fa, struct fuse_dummy_io *unused, * We ought to only make that buffer if it's been requested, so leaving * this unimplemented for the moment */ - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .opcode = FUSE_READLINK, .nodeid = get_node_id(inode), .in_numargs = 1, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = dentry->d_name.len + 1, .value = dentry->d_name.name, }, @@ -2104,7 +2103,7 @@ int fuse_get_link_initialize(struct fuse_args *fa, struct fuse_dummy_io *unused, return 0; } -int fuse_get_link_backing(struct fuse_args *fa, +int fuse_get_link_backing(struct fuse_bpf_args *fa, struct inode *inode, struct dentry *dentry, struct delayed_call *callback, const char **out) { @@ -2131,7 +2130,7 @@ int fuse_get_link_backing(struct fuse_args *fa, return 0; } -void *fuse_get_link_finalize(struct fuse_args *fa, +void *fuse_get_link_finalize(struct fuse_bpf_args *fa, struct inode *inode, struct dentry *dentry, struct delayed_call *callback, const char **out) { @@ -2139,18 +2138,18 @@ void *fuse_get_link_finalize(struct fuse_args *fa, } int fuse_symlink_initialize( - struct fuse_args *fa, struct fuse_dummy_io *unused, + struct fuse_bpf_args *fa, struct fuse_dummy_io *unused, struct inode *dir, struct dentry *entry, const char *link, int len) { - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = get_node_id(dir), .opcode = FUSE_SYMLINK, .in_numargs = 2, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = entry->d_name.len + 1, .value = entry->d_name.name, }, - .in_args[1] = (struct fuse_in_arg) { + .in_args[1] = (struct fuse_bpf_in_arg) { .size = len, .value = link, }, @@ -2160,7 +2159,7 @@ int fuse_symlink_initialize( } int fuse_symlink_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, const char *link, int len) { int err = 0; @@ -2199,13 +2198,13 @@ out: } void *fuse_symlink_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, const char *link, int len) { return NULL; } -int fuse_readdir_initialize(struct fuse_args *fa, struct fuse_read_io *frio, +int fuse_readdir_initialize(struct fuse_bpf_args *fa, struct fuse_read_io *frio, struct file *file, struct dir_context *ctx, bool *force_again, bool *allow_force, bool is_continued) { @@ -2215,21 +2214,21 @@ int fuse_readdir_initialize(struct fuse_args *fa, struct fuse_read_io *frio, if (!page) return -ENOMEM; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .nodeid = ff->nodeid, .opcode = FUSE_READDIR, .in_numargs = 1, - .out_argvar = true, + .flags = FUSE_BPF_OUT_ARGVAR, .out_numargs = 2, - .in_args[0] = (struct fuse_in_arg) { + .in_args[0] = (struct fuse_bpf_in_arg) { .size = sizeof(frio->fri), .value = &frio->fri, }, - .out_args[0] = (struct fuse_arg) { + .out_args[0] = (struct fuse_bpf_arg) { .size = sizeof(frio->fro), .value = &frio->fro, }, - .out_args[1] = (struct fuse_arg) { + .out_args[1] = (struct fuse_bpf_arg) { .size = PAGE_SIZE, .value = page, }, @@ -2303,7 +2302,7 @@ static int parse_dirfile(char *buf, size_t nbytes, struct dir_context *ctx) } -int fuse_readdir_backing(struct fuse_args *fa, +int fuse_readdir_backing(struct fuse_bpf_args *fa, struct file *file, struct dir_context *ctx, bool *force_again, bool *allow_force, bool is_continued) { @@ -2335,7 +2334,7 @@ int fuse_readdir_backing(struct fuse_args *fa, return err; } -void *fuse_readdir_finalize(struct fuse_args *fa, +void *fuse_readdir_finalize(struct fuse_bpf_args *fa, struct file *file, struct dir_context *ctx, bool *force_again, bool *allow_force, bool is_continued) { @@ -2356,14 +2355,14 @@ void *fuse_readdir_finalize(struct fuse_args *fa, return ERR_PTR(err); } -int fuse_access_initialize(struct fuse_args *fa, struct fuse_access_in *fai, +int fuse_access_initialize(struct fuse_bpf_args *fa, struct fuse_access_in *fai, struct inode *inode, int mask) { *fai = (struct fuse_access_in) { .mask = mask, }; - *fa = (struct fuse_args) { + *fa = (struct fuse_bpf_args) { .opcode = FUSE_ACCESS, .nodeid = get_node_id(inode), .in_numargs = 1, @@ -2374,7 +2373,7 @@ int fuse_access_initialize(struct fuse_args *fa, struct fuse_access_in *fai, return 0; } -int fuse_access_backing(struct fuse_args *fa, struct inode *inode, int mask) +int fuse_access_backing(struct fuse_bpf_args *fa, struct inode *inode, int mask) { struct fuse_inode *fi = get_fuse_inode(inode); const struct fuse_access_in *fai = fa->in_args[0].value; @@ -2383,7 +2382,7 @@ int fuse_access_backing(struct fuse_args *fa, struct inode *inode, int mask) fi->backing_inode, fai->mask); } -void *fuse_access_finalize(struct fuse_args *fa, struct inode *inode, int mask) +void *fuse_access_finalize(struct fuse_bpf_args *fa, struct inode *inode, int mask) { return NULL; } @@ -2403,3 +2402,54 @@ void __exit fuse_bpf_cleanup(void) { kmem_cache_destroy(fuse_bpf_aio_request_cachep); } + +ssize_t fuse_bpf_simple_request(struct fuse_mount *fm, struct fuse_bpf_args *bpf_args) +{ + int i; + ssize_t res; + struct fuse_args args = { + .nodeid = bpf_args->nodeid, + .opcode = bpf_args->opcode, + .error_in = bpf_args->error_in, + .in_numargs = bpf_args->in_numargs, + .out_numargs = bpf_args->out_numargs, + .force = !!(bpf_args->flags & FUSE_BPF_FORCE), + .out_argvar = !!(bpf_args->flags & FUSE_BPF_OUT_ARGVAR), + }; + + for (i = 0; i < args.in_numargs; ++i) + args.in_args[i] = (struct fuse_in_arg) { + .size = bpf_args->in_args[i].size, + .value = bpf_args->in_args[i].value, + }; + for (i = 0; i < args.out_numargs; ++i) + args.out_args[i] = (struct fuse_arg) { + .size = bpf_args->out_args[i].size, + .value = bpf_args->out_args[i].value, + }; + + res = fuse_simple_request(fm, &args); + + *bpf_args = (struct fuse_bpf_args) { + .nodeid = args.nodeid, + .opcode = args.opcode, + .error_in = args.error_in, + .in_numargs = args.in_numargs, + .out_numargs = args.out_numargs, + }; + if (args.force) + bpf_args->flags |= FUSE_BPF_FORCE; + if (args.out_args) + bpf_args->flags |= FUSE_BPF_OUT_ARGVAR; + for (i = 0; i < args.in_numargs; ++i) + bpf_args->in_args[i] = (struct fuse_bpf_in_arg) { + .size = args.in_args[i].size, + .value = args.in_args[i].value, + }; + for (i = 0; i < args.out_numargs; ++i) + bpf_args->out_args[i] = (struct fuse_bpf_arg) { + .size = args.out_args[i].size, + .value = args.out_args[i].value, + }; + return res; +} diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index b1fb8d83117c..2f07d3a1ff28 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -308,12 +308,48 @@ struct fuse_file { bool flock:1; }; +/** One input argument of a request */ +struct fuse_in_arg { + unsigned size; + const void *value; +}; + +/** One output argument of a request */ +struct fuse_arg { + unsigned size; + void *value; +}; + /** FUSE page descriptor */ struct fuse_page_desc { unsigned int length; unsigned int offset; }; +struct fuse_args { + uint64_t nodeid; + uint32_t opcode; + uint32_t error_in; + unsigned short in_numargs; + unsigned short out_numargs; + bool force:1; + bool noreply:1; + bool nocreds:1; + bool in_pages:1; + bool out_pages:1; + bool user_pages:1; + bool out_argvar:1; + bool page_zeroing:1; + bool page_replace:1; + bool may_block:1; + struct fuse_in_arg in_args[FUSE_MAX_IN_ARGS]; + struct fuse_arg out_args[FUSE_MAX_OUT_ARGS]; + void (*end)(struct fuse_mount *fm, struct fuse_args *args, int error); + + /* Path used for completing d_canonical_path */ + struct path *canonical_path; +}; + struct fuse_args_pages { struct fuse_args args; struct page **pages; @@ -1314,11 +1350,11 @@ struct fuse_open_io { struct fuse_open_out foo; }; -int fuse_open_initialize(struct fuse_args *fa, struct fuse_open_io *foi, +int fuse_open_initialize(struct fuse_bpf_args *fa, struct fuse_open_io *foi, struct inode *inode, struct file *file, bool isdir); -int fuse_open_backing(struct fuse_args *fa, +int fuse_open_backing(struct fuse_bpf_args *fa, struct inode *inode, struct file *file, bool isdir); -void *fuse_open_finalize(struct fuse_args *fa, +void *fuse_open_finalize(struct fuse_bpf_args *fa, struct inode *inode, struct file *file, bool isdir); struct fuse_create_open_io { @@ -1328,103 +1364,103 @@ struct fuse_create_open_io { }; int fuse_create_open_initialize( - struct fuse_args *fa, struct fuse_create_open_io *fcoi, + struct fuse_bpf_args *fa, struct fuse_create_open_io *fcoi, struct inode *dir, struct dentry *entry, struct file *file, unsigned int flags, umode_t mode); int fuse_create_open_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, struct file *file, unsigned int flags, umode_t mode); void *fuse_create_open_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, struct file *file, unsigned int flags, umode_t mode); int fuse_mknod_initialize( - struct fuse_args *fa, struct fuse_mknod_in *fmi, + struct fuse_bpf_args *fa, struct fuse_mknod_in *fmi, struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev); int fuse_mknod_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev); void *fuse_mknod_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev); int fuse_mkdir_initialize( - struct fuse_args *fa, struct fuse_mkdir_in *fmi, + struct fuse_bpf_args *fa, struct fuse_mkdir_in *fmi, struct inode *dir, struct dentry *entry, umode_t mode); int fuse_mkdir_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, umode_t mode); void *fuse_mkdir_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, umode_t mode); int fuse_rmdir_initialize( - struct fuse_args *fa, struct fuse_dummy_io *fmi, + struct fuse_bpf_args *fa, struct fuse_dummy_io *fmi, struct inode *dir, struct dentry *entry); int fuse_rmdir_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry); void *fuse_rmdir_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry); -int fuse_rename2_initialize(struct fuse_args *fa, struct fuse_rename2_in *fri, +int fuse_rename2_initialize(struct fuse_bpf_args *fa, struct fuse_rename2_in *fri, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent, unsigned int flags); -int fuse_rename2_backing(struct fuse_args *fa, +int fuse_rename2_backing(struct fuse_bpf_args *fa, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent, unsigned int flags); -void *fuse_rename2_finalize(struct fuse_args *fa, +void *fuse_rename2_finalize(struct fuse_bpf_args *fa, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent, unsigned int flags); -int fuse_rename_initialize(struct fuse_args *fa, struct fuse_rename_in *fri, +int fuse_rename_initialize(struct fuse_bpf_args *fa, struct fuse_rename_in *fri, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent); -int fuse_rename_backing(struct fuse_args *fa, +int fuse_rename_backing(struct fuse_bpf_args *fa, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent); -void *fuse_rename_finalize(struct fuse_args *fa, +void *fuse_rename_finalize(struct fuse_bpf_args *fa, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent); int fuse_unlink_initialize( - struct fuse_args *fa, struct fuse_dummy_io *fmi, + struct fuse_bpf_args *fa, struct fuse_dummy_io *fmi, struct inode *dir, struct dentry *entry); int fuse_unlink_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry); void *fuse_unlink_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry); -int fuse_link_initialize(struct fuse_args *fa, struct fuse_link_in *fli, +int fuse_link_initialize(struct fuse_bpf_args *fa, struct fuse_link_in *fli, struct dentry *entry, struct inode *dir, struct dentry *newent); -int fuse_link_backing(struct fuse_args *fa, struct dentry *entry, +int fuse_link_backing(struct fuse_bpf_args *fa, struct dentry *entry, struct inode *dir, struct dentry *newent); -void *fuse_link_finalize(struct fuse_args *fa, struct dentry *entry, +void *fuse_link_finalize(struct fuse_bpf_args *fa, struct dentry *entry, struct inode *dir, struct dentry *newent); -int fuse_release_initialize(struct fuse_args *fa, struct fuse_release_in *fri, +int fuse_release_initialize(struct fuse_bpf_args *fa, struct fuse_release_in *fri, struct inode *inode, struct file *file); -int fuse_releasedir_initialize(struct fuse_args *fa, +int fuse_releasedir_initialize(struct fuse_bpf_args *fa, struct fuse_release_in *fri, struct inode *inode, struct file *file); -int fuse_release_backing(struct fuse_args *fa, +int fuse_release_backing(struct fuse_bpf_args *fa, struct inode *inode, struct file *file); -void *fuse_release_finalize(struct fuse_args *fa, +void *fuse_release_finalize(struct fuse_bpf_args *fa, struct inode *inode, struct file *file); -int fuse_flush_initialize(struct fuse_args *fa, struct fuse_flush_in *ffi, +int fuse_flush_initialize(struct fuse_bpf_args *fa, struct fuse_flush_in *ffi, struct file *file, fl_owner_t id); -int fuse_flush_backing(struct fuse_args *fa, struct file *file, fl_owner_t id); -void *fuse_flush_finalize(struct fuse_args *fa, +int fuse_flush_backing(struct fuse_bpf_args *fa, struct file *file, fl_owner_t id); +void *fuse_flush_finalize(struct fuse_bpf_args *fa, struct file *file, fl_owner_t id); struct fuse_lseek_io { @@ -1432,37 +1468,37 @@ struct fuse_lseek_io { struct fuse_lseek_out flo; }; -int fuse_lseek_initialize(struct fuse_args *fa, struct fuse_lseek_io *fli, +int fuse_lseek_initialize(struct fuse_bpf_args *fa, struct fuse_lseek_io *fli, struct file *file, loff_t offset, int whence); -int fuse_lseek_backing(struct fuse_args *fa, struct file *file, loff_t offset, int whence); -void *fuse_lseek_finalize(struct fuse_args *fa, struct file *file, loff_t offset, int whence); +int fuse_lseek_backing(struct fuse_bpf_args *fa, struct file *file, loff_t offset, int whence); +void *fuse_lseek_finalize(struct fuse_bpf_args *fa, struct file *file, loff_t offset, int whence); struct fuse_copy_file_range_io { struct fuse_copy_file_range_in fci; struct fuse_write_out fwo; }; -int fuse_copy_file_range_initialize(struct fuse_args *fa, +int fuse_copy_file_range_initialize(struct fuse_bpf_args *fa, struct fuse_copy_file_range_io *fcf, struct file *file_in, loff_t pos_in, struct file *file_out, loff_t pos_out, size_t len, unsigned int flags); -int fuse_copy_file_range_backing(struct fuse_args *fa, +int fuse_copy_file_range_backing(struct fuse_bpf_args *fa, struct file *file_in, loff_t pos_in, struct file *file_out, loff_t pos_out, size_t len, unsigned int flags); -void *fuse_copy_file_range_finalize(struct fuse_args *fa, +void *fuse_copy_file_range_finalize(struct fuse_bpf_args *fa, struct file *file_in, loff_t pos_in, struct file *file_out, loff_t pos_out, size_t len, unsigned int flags); -int fuse_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, +int fuse_fsync_initialize(struct fuse_bpf_args *fa, struct fuse_fsync_in *ffi, struct file *file, loff_t start, loff_t end, int datasync); -int fuse_fsync_backing(struct fuse_args *fa, +int fuse_fsync_backing(struct fuse_bpf_args *fa, struct file *file, loff_t start, loff_t end, int datasync); -void *fuse_fsync_finalize(struct fuse_args *fa, +void *fuse_fsync_finalize(struct fuse_bpf_args *fa, struct file *file, loff_t start, loff_t end, int datasync); -int fuse_dir_fsync_initialize(struct fuse_args *fa, struct fuse_fsync_in *ffi, +int fuse_dir_fsync_initialize(struct fuse_bpf_args *fa, struct fuse_fsync_in *ffi, struct file *file, loff_t start, loff_t end, int datasync); struct fuse_getxattr_io { @@ -1471,43 +1507,43 @@ struct fuse_getxattr_io { }; int fuse_getxattr_initialize( - struct fuse_args *fa, struct fuse_getxattr_io *fgio, + struct fuse_bpf_args *fa, struct fuse_getxattr_io *fgio, struct dentry *dentry, const char *name, void *value, size_t size); int fuse_getxattr_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct dentry *dentry, const char *name, void *value, size_t size); void *fuse_getxattr_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct dentry *dentry, const char *name, void *value, size_t size); -int fuse_listxattr_initialize(struct fuse_args *fa, +int fuse_listxattr_initialize(struct fuse_bpf_args *fa, struct fuse_getxattr_io *fgio, struct dentry *dentry, char *list, size_t size); -int fuse_listxattr_backing(struct fuse_args *fa, struct dentry *dentry, +int fuse_listxattr_backing(struct fuse_bpf_args *fa, struct dentry *dentry, char *list, size_t size); -void *fuse_listxattr_finalize(struct fuse_args *fa, struct dentry *dentry, +void *fuse_listxattr_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, char *list, size_t size); -int fuse_setxattr_initialize(struct fuse_args *fa, +int fuse_setxattr_initialize(struct fuse_bpf_args *fa, struct fuse_setxattr_in *fsxi, struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int fuse_setxattr_backing(struct fuse_args *fa, struct dentry *dentry, +int fuse_setxattr_backing(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -void *fuse_setxattr_finalize(struct fuse_args *fa, struct dentry *dentry, +void *fuse_setxattr_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int fuse_removexattr_initialize(struct fuse_args *fa, +int fuse_removexattr_initialize(struct fuse_bpf_args *fa, struct fuse_dummy_io *unused, struct dentry *dentry, const char *name); -int fuse_removexattr_backing(struct fuse_args *fa, +int fuse_removexattr_backing(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name); -void *fuse_removexattr_finalize(struct fuse_args *fa, +void *fuse_removexattr_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, const char *name); struct fuse_read_iter_out { @@ -1519,11 +1555,11 @@ struct fuse_file_read_iter_io { }; int fuse_file_read_iter_initialize( - struct fuse_args *fa, struct fuse_file_read_iter_io *fri, + struct fuse_bpf_args *fa, struct fuse_file_read_iter_io *fri, struct kiocb *iocb, struct iov_iter *to); -int fuse_file_read_iter_backing(struct fuse_args *fa, +int fuse_file_read_iter_backing(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *to); -void *fuse_file_read_iter_finalize(struct fuse_args *fa, +void *fuse_file_read_iter_finalize(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *to); struct fuse_write_iter_out { @@ -1536,21 +1572,21 @@ struct fuse_file_write_iter_io { }; int fuse_file_write_iter_initialize( - struct fuse_args *fa, struct fuse_file_write_iter_io *fwio, + struct fuse_bpf_args *fa, struct fuse_file_write_iter_io *fwio, struct kiocb *iocb, struct iov_iter *from); -int fuse_file_write_iter_backing(struct fuse_args *fa, +int fuse_file_write_iter_backing(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *from); -void *fuse_file_write_iter_finalize(struct fuse_args *fa, +void *fuse_file_write_iter_finalize(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *from); ssize_t fuse_backing_mmap(struct file *file, struct vm_area_struct *vma); -int fuse_file_fallocate_initialize(struct fuse_args *fa, +int fuse_file_fallocate_initialize(struct fuse_bpf_args *fa, struct fuse_fallocate_in *ffi, struct file *file, int mode, loff_t offset, loff_t length); -int fuse_file_fallocate_backing(struct fuse_args *fa, +int fuse_file_fallocate_backing(struct fuse_bpf_args *fa, struct file *file, int mode, loff_t offset, loff_t length); -void *fuse_file_fallocate_finalize(struct fuse_args *fa, +void *fuse_file_fallocate_finalize(struct fuse_bpf_args *fa, struct file *file, int mode, loff_t offset, loff_t length); struct fuse_lookup_io { @@ -1558,24 +1594,24 @@ struct fuse_lookup_io { struct fuse_entry_bpf feb; }; -int fuse_lookup_initialize(struct fuse_args *fa, struct fuse_lookup_io *feo, +int fuse_lookup_initialize(struct fuse_bpf_args *fa, struct fuse_lookup_io *feo, struct inode *dir, struct dentry *entry, unsigned int flags); -int fuse_lookup_backing(struct fuse_args *fa, struct inode *dir, +int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags); -struct dentry *fuse_lookup_finalize(struct fuse_args *fa, struct inode *dir, +struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags); -int fuse_revalidate_backing(struct fuse_args *fa, struct inode *dir, +int fuse_revalidate_backing(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags); -void *fuse_revalidate_finalize(struct fuse_args *fa, struct inode *dir, +void *fuse_revalidate_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags); -int fuse_canonical_path_initialize(struct fuse_args *fa, +int fuse_canonical_path_initialize(struct fuse_bpf_args *fa, struct fuse_dummy_io *fdi, const struct path *path, struct path *canonical_path); -int fuse_canonical_path_backing(struct fuse_args *fa, const struct path *path, +int fuse_canonical_path_backing(struct fuse_bpf_args *fa, const struct path *path, struct path *canonical_path); -void *fuse_canonical_path_finalize(struct fuse_args *fa, +void *fuse_canonical_path_finalize(struct fuse_bpf_args *fa, const struct path *path, struct path *canonical_path); @@ -1583,13 +1619,13 @@ struct fuse_getattr_io { struct fuse_getattr_in fgi; struct fuse_attr_out fao; }; -int fuse_getattr_initialize(struct fuse_args *fa, struct fuse_getattr_io *fgio, +int fuse_getattr_initialize(struct fuse_bpf_args *fa, struct fuse_getattr_io *fgio, const struct dentry *entry, struct kstat *stat, u32 request_mask, unsigned int flags); -int fuse_getattr_backing(struct fuse_args *fa, +int fuse_getattr_backing(struct fuse_bpf_args *fa, const struct dentry *entry, struct kstat *stat, u32 request_mask, unsigned int flags); -void *fuse_getattr_finalize(struct fuse_args *fa, +void *fuse_getattr_finalize(struct fuse_bpf_args *fa, const struct dentry *entry, struct kstat *stat, u32 request_mask, unsigned int flags); @@ -1598,38 +1634,38 @@ struct fuse_setattr_io { struct fuse_attr_out fao; }; -int fuse_setattr_initialize(struct fuse_args *fa, struct fuse_setattr_io *fsi, +int fuse_setattr_initialize(struct fuse_bpf_args *fa, struct fuse_setattr_io *fsi, struct dentry *dentry, struct iattr *attr, struct file *file); -int fuse_setattr_backing(struct fuse_args *fa, +int fuse_setattr_backing(struct fuse_bpf_args *fa, struct dentry *dentry, struct iattr *attr, struct file *file); -void *fuse_setattr_finalize(struct fuse_args *fa, +void *fuse_setattr_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, struct iattr *attr, struct file *file); -int fuse_statfs_initialize(struct fuse_args *fa, struct fuse_statfs_out *fso, +int fuse_statfs_initialize(struct fuse_bpf_args *fa, struct fuse_statfs_out *fso, struct dentry *dentry, struct kstatfs *buf); -int fuse_statfs_backing(struct fuse_args *fa, +int fuse_statfs_backing(struct fuse_bpf_args *fa, struct dentry *dentry, struct kstatfs *buf); -void *fuse_statfs_finalize(struct fuse_args *fa, +void *fuse_statfs_finalize(struct fuse_bpf_args *fa, struct dentry *dentry, struct kstatfs *buf); -int fuse_get_link_initialize(struct fuse_args *fa, struct fuse_dummy_io *dummy, +int fuse_get_link_initialize(struct fuse_bpf_args *fa, struct fuse_dummy_io *dummy, struct inode *inode, struct dentry *dentry, struct delayed_call *callback, const char **out); -int fuse_get_link_backing(struct fuse_args *fa, +int fuse_get_link_backing(struct fuse_bpf_args *fa, struct inode *inode, struct dentry *dentry, struct delayed_call *callback, const char **out); -void *fuse_get_link_finalize(struct fuse_args *fa, +void *fuse_get_link_finalize(struct fuse_bpf_args *fa, struct inode *inode, struct dentry *dentry, struct delayed_call *callback, const char **out); int fuse_symlink_initialize( - struct fuse_args *fa, struct fuse_dummy_io *unused, + struct fuse_bpf_args *fa, struct fuse_dummy_io *unused, struct inode *dir, struct dentry *entry, const char *link, int len); int fuse_symlink_backing( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, const char *link, int len); void *fuse_symlink_finalize( - struct fuse_args *fa, + struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, const char *link, int len); struct fuse_read_io { @@ -1637,20 +1673,20 @@ struct fuse_read_io { struct fuse_read_out fro; }; -int fuse_readdir_initialize(struct fuse_args *fa, struct fuse_read_io *frio, +int fuse_readdir_initialize(struct fuse_bpf_args *fa, struct fuse_read_io *frio, struct file *file, struct dir_context *ctx, bool *force_again, bool *allow_force, bool is_continued); -int fuse_readdir_backing(struct fuse_args *fa, +int fuse_readdir_backing(struct fuse_bpf_args *fa, struct file *file, struct dir_context *ctx, bool *force_again, bool *allow_force, bool is_continued); -void *fuse_readdir_finalize(struct fuse_args *fa, +void *fuse_readdir_finalize(struct fuse_bpf_args *fa, struct file *file, struct dir_context *ctx, bool *force_again, bool *allow_force, bool is_continued); -int fuse_access_initialize(struct fuse_args *fa, struct fuse_access_in *fai, +int fuse_access_initialize(struct fuse_bpf_args *fa, struct fuse_access_in *fai, struct inode *inode, int mask); -int fuse_access_backing(struct fuse_args *fa, struct inode *inode, int mask); -void *fuse_access_finalize(struct fuse_args *fa, struct inode *inode, int mask); +int fuse_access_backing(struct fuse_bpf_args *fa, struct inode *inode, int mask); +void *fuse_access_finalize(struct fuse_bpf_args *fa, struct inode *inode, int mask); /* * FUSE caches dentries and attributes with separate timeout. The @@ -1789,16 +1825,18 @@ struct fuse_err_ret { int __init fuse_bpf_init(void); void __exit fuse_bpf_cleanup(void); +ssize_t fuse_bpf_simple_request(struct fuse_mount *fm, struct fuse_bpf_args *args); + /* * expression statement to wrap the backing filter logic * struct inode *inode: inode with bpf and backing inode * typedef io: (typically complex) type whose components fuse_args can point to. * An instance of this type is created locally and passed to initialize - * void initialize(struct fuse_args *fa, io *in_out, args...): function that sets + * void initialize(struct fuse_bpf_args *fa, io *in_out, args...): function that sets * up fa and io based on args - * int backing(struct fuse_args *fa, args...): function that actually performs + * int backing(struct fuse_bpf_args *fa, args...): function that actually performs * the backing io operation - * void *finalize(struct fuse_args *, args...): function that performs any final + * void *finalize(struct fuse_bpf_args *, args...): function that performs any final * work needed to commit the backing io */ #define fuse_bpf_backing(inode, io, initialize, backing, finalize, \ @@ -1809,7 +1847,7 @@ void __exit fuse_bpf_cleanup(void); struct fuse_inode *fuse_inode = get_fuse_inode(inode); \ struct fuse_mount *fm = get_fuse_mount(inode); \ io feo = {0}; \ - struct fuse_args fa = {0}, fa_backup = {0}; \ + struct fuse_bpf_args fa = {0}, fa_backup = {0}; \ bool locked; \ ssize_t res; \ void *err; \ @@ -1833,7 +1871,7 @@ void __exit fuse_bpf_cleanup(void); fa_backup = fa; \ fa.opcode |= FUSE_PREFILTER; \ for (i = 0; i < fa.in_numargs; ++i) \ - fa.out_args[i] = (struct fuse_arg) { \ + fa.out_args[i] = (struct fuse_bpf_arg) { \ .size = fa.in_args[i].size, \ .value = (void *)fa.in_args[i].value, \ }; \ @@ -1852,7 +1890,7 @@ void __exit fuse_bpf_cleanup(void); \ if (ext_flags & FUSE_BPF_USER_FILTER) { \ locked = fuse_lock_inode(inode); \ - res = fuse_simple_request(fm, &fa); \ + res = fuse_bpf_simple_request(fm, &fa); \ fuse_unlock_inode(inode, locked); \ if (res < 0) { \ fer = (struct fuse_err_ret) { \ @@ -1868,12 +1906,12 @@ void __exit fuse_bpf_cleanup(void); \ fa.opcode &= ~FUSE_PREFILTER; \ for (i = 0; i < fa.in_numargs; ++i) \ - fa.in_args[i] = (struct fuse_in_arg) { \ + fa.in_args[i] = (struct fuse_bpf_in_arg) { \ .size = fa.out_args[i].size, \ .value = fa.out_args[i].value, \ }; \ - for (i = 0; i < fa.out_numargs; ++i) \ - fa.out_args[i] = (struct fuse_arg) { \ + for (i = 0; i < fa_backup.out_numargs; ++i) \ + fa.out_args[i] = (struct fuse_bpf_arg) { \ .size = fa_backup.out_args[i].size, \ .value = fa_backup.out_args[i].value, \ }; \ @@ -1891,7 +1929,7 @@ void __exit fuse_bpf_cleanup(void); fa.opcode |= FUSE_POSTFILTER; \ for (i = 0; i < fa.out_numargs; ++i) \ fa.in_args[fa.in_numargs++] = \ - (struct fuse_in_arg) { \ + (struct fuse_bpf_in_arg) { \ .size = fa.out_args[i].size, \ .value = fa.out_args[i].value, \ }; \ @@ -1910,7 +1948,7 @@ void __exit fuse_bpf_cleanup(void); fa.out_args[1].size = fa_backup.out_args[1].size; \ fa.out_numargs = fa_backup.out_numargs; \ locked = fuse_lock_inode(inode); \ - res = fuse_simple_request(fm, &fa); \ + res = fuse_bpf_simple_request(fm, &fa); \ fuse_unlock_inode(inode, locked); \ if (res < 0) { \ fer.result = ERR_PTR(res); \ diff --git a/include/linux/bpf_types.h b/include/linux/bpf_types.h index eaddb8d8365b..2d8b63574aa8 100644 --- a/include/linux/bpf_types.h +++ b/include/linux/bpf_types.h @@ -78,7 +78,7 @@ BPF_PROG_TYPE(BPF_PROG_TYPE_LSM, lsm, #endif /* CONFIG_BPF_LSM */ #endif #ifdef CONFIG_FUSE_BPF -BPF_PROG_TYPE(BPF_PROG_TYPE_FUSE, fuse, struct fuse_args, struct fuse_args) +BPF_PROG_TYPE(BPF_PROG_TYPE_FUSE, fuse, struct fuse_bpf_args, struct fuse_bpf_args) #endif BPF_MAP_TYPE(BPF_MAP_TYPE_ARRAY, array_map_ops) diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index a0c34ffc97fe..f0521a39a586 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -965,42 +965,44 @@ struct fuse_removemapping_one { #define FUSE_REMOVEMAPPING_MAX_ENTRY \ (PAGE_SIZE / sizeof(struct fuse_removemapping_one)) -struct fuse_mount; +/* + * Fuse BPF Args + * + * Used to communicate with bpf programs to allow checking or altering certain values. + * The end_offset allows the bpf verifier to check boundaries statically. This reflects + * the ends of the buffer. size shows the length that was actually used. + * + */ /** One input argument of a request */ -struct fuse_in_arg { - unsigned size; +struct fuse_bpf_in_arg { + uint32_t size; const void *value; + const void *end_offset; }; /** One output argument of a request */ -struct fuse_arg { - unsigned size; +struct fuse_bpf_arg { + uint32_t size; void *value; + void *end_offset; }; -struct fuse_args { +#define FUSE_MAX_IN_ARGS 5 +#define FUSE_MAX_OUT_ARGS 3 + +#define FUSE_BPF_FORCE (1 << 0) +#define FUSE_BPF_OUT_ARGVAR (1 << 6) + +struct fuse_bpf_args { uint64_t nodeid; uint32_t opcode; uint32_t error_in; - unsigned short in_numargs; - unsigned short out_numargs; - bool force:1; - bool noreply:1; - bool nocreds:1; - bool in_pages:1; - bool out_pages:1; - bool user_pages:1; - bool out_argvar:1; - bool page_zeroing:1; - bool page_replace:1; - bool may_block:1; - struct fuse_in_arg in_args[5]; - struct fuse_arg out_args[3]; - void (*end)(struct fuse_mount *fm, struct fuse_args *args, int error); - - /* Path used for completing d_canonical_path */ - struct path *canonical_path; + uint32_t in_numargs; + uint32_t out_numargs; + uint32_t flags; + struct fuse_bpf_in_arg in_args[FUSE_MAX_IN_ARGS]; + struct fuse_bpf_arg out_args[FUSE_MAX_OUT_ARGS]; }; #define FUSE_BPF_USER_FILTER 1 diff --git a/kernel/bpf/bpf_fuse.c b/kernel/bpf/bpf_fuse.c index 2a4db5ce06b4..00738c84b76e 100644 --- a/kernel/bpf/bpf_fuse.c +++ b/kernel/bpf/bpf_fuse.c @@ -36,12 +36,12 @@ static bool fuse_prog_is_valid_access(int off, int size, { int i; - if (off < 0 || off > offsetofend(struct fuse_args, out_args)) + if (off < 0 || off > offsetofend(struct fuse_bpf_args, out_args)) return false; /* TODO This is garbage. Do it properly */ for (i = 0; i < 5; i++) { - if (off == offsetof(struct fuse_args, in_args[i].value)) { + if (off == offsetof(struct fuse_bpf_args, in_args[i].value)) { info->reg_type = PTR_TO_RDONLY_BUF; info->ctx_field_size = 256; if (type != BPF_READ) @@ -50,7 +50,7 @@ static bool fuse_prog_is_valid_access(int off, int size, } } for (i = 0; i < 3; i++) { - if (off == offsetof(struct fuse_args, out_args[i].value)) { + if (off == offsetof(struct fuse_bpf_args, out_args[i].value)) { info->reg_type = PTR_TO_RDWR_BUF; info->ctx_field_size = 256; return true; From 3665b480d1bf22124739a8ea0976bf181a08de2a Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Thu, 21 Apr 2022 08:16:22 -0700 Subject: [PATCH 095/148] UPSTREAM: ANDROID: fuse-bpf: Fix misuse of args.out_args Test: fuse_test Bug: 202785178 Signed-off-by: Paul Lawrence Change-Id: I332d196329bba257a577d3ddc140136aa03bfdf1 --- fs/fuse/backing.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 2959a6944c1a..b68f4f534974 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -2439,7 +2439,7 @@ ssize_t fuse_bpf_simple_request(struct fuse_mount *fm, struct fuse_bpf_args *bpf }; if (args.force) bpf_args->flags |= FUSE_BPF_FORCE; - if (args.out_args) + if (args.out_argvar) bpf_args->flags |= FUSE_BPF_OUT_ARGVAR; for (i = 0; i < args.in_numargs; ++i) bpf_args->in_args[i] = (struct fuse_bpf_in_arg) { From 5387caaa4b1a6fa5371bd96614b328ae4b137bf2 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Wed, 20 Apr 2022 13:44:15 -0700 Subject: [PATCH 096/148] UPSTREAM: ANDROID: fuse-bpf: Fix non-fusebpf build Added #ifdefs around fuse-bpf init/cleanup code Bug: 202785178 Test: builds with and without CONFIG_FUSE_BPF Signed-off-by: Daniel Rosenberg Change-Id: Ie15bb04e439b496e4842303437b3f55c3da14f2c --- fs/fuse/inode.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 0e803d857f9f..8287dbe92967 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1911,17 +1911,21 @@ static int __init fuse_init(void) if (res) goto err_sysfs_cleanup; +#ifdef CONFIG_FUSE_BPF res = fuse_bpf_init(); if (res) goto err_ctl_cleanup; +#endif sanitize_global_limit(&max_user_bgreq); sanitize_global_limit(&max_user_congthresh); return 0; +#ifdef CONFIG_FUSE_BPF err_ctl_cleanup: fuse_ctl_cleanup(); +#endif err_sysfs_cleanup: fuse_sysfs_cleanup(); err_dev_cleanup: @@ -1939,7 +1943,9 @@ static void __exit fuse_exit(void) fuse_ctl_cleanup(); fuse_sysfs_cleanup(); fuse_fs_cleanup(); +#ifdef CONFIG_FUSE_BPF fuse_bpf_cleanup(); +#endif fuse_dev_cleanup(); } From 99f0eb0da8f616535c0a0d42434e8b81cef84b50 Mon Sep 17 00:00:00 2001 From: Dmitrii Merkurev Date: Mon, 30 May 2022 17:26:13 +0000 Subject: [PATCH 097/148] UPSTREAM: ANDROID: fuse-bpf: use target instead of parent inode to execute backing revalidate Usually as a result of initial fuse lookup with bpf enabled we have following dentry: ----------------------------------------------------------------- | dentry /storage/emulated/0/Android/data | | inode | | backing_inode: /pass_through/emulated/0/Android/data | ----------------------------------------------------------------- Every communication with this folder will have to go through fuse_dentry_revalidate(dentry, flags) which can move forward by: 1. If the timeout is not reached, just ignore it 2. If entry has backing_inode and bpf is not against it, execute revalidate on backing FS (inside kernel) 3. Move to userspace to revalidate But for some reason currently, we're checking parent inode (not one that we wanna revalidate) to have backing inode that we can use to execute operations on. Basically, the whole flow looks like this: 1. Receiving revalidate event for fuse_dentry_revalidate(/storage/emulated/0/Android/data, flags) 2. Checking .../0/Android/ inode to have backing inode <------------------------ Primary problem is HERE 3. Moving to the userspace with pf_lookup(/storage/emulated/0/Android, data) 4. Even though successfully handled lookup on the fuse daemon side, kernel cannot interpret the result due to fuse_simple_request and fuse_lookup_init logic changes <------- Secondary problem is HERE 5. Because of the problems I mentioned before, full lookup is triggered on the kernel side so we receive the second pf_lookup to the userspace Fixing primary problem by executing backing revalidate on the current inode (not the parent one). Bug: 234346312 Test: Manually made sure don't have any userspace calls for interactions inside directory with backing one. Test: Manually check youtube app is successfully saving exo cache into the external storage cache folder. Test: atest --test-mapping packages/providers/MediaProvider Signed-off-by: Dmitrii Merkurev Change-Id: Id57f1944302076d93ebef255533dfc53e8c30f20 --- fs/fuse/dir.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index ceda69466011..5a1579b62566 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -212,7 +212,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) { struct fuse_err_ret fer; - fer = fuse_bpf_backing(entry->d_parent->d_inode, + fer = fuse_bpf_backing(inode, struct fuse_lookup_io, fuse_lookup_initialize, fuse_revalidate_backing, From c01add08ee6832dfa1445754221476978baba746 Mon Sep 17 00:00:00 2001 From: Dmitrii Merkurev Date: Thu, 23 Jun 2022 18:31:07 +0000 Subject: [PATCH 098/148] UPSTREAM: Revert "ANDROID: fuse-bpf: use target instead of parent inode to execute backing revalidate" This reverts commit b610eff230f2ce92fd48502d71ad0993792f73d3. Reason for revert: I broke BPF calling logic with this one. Possible fix is here: https://android-review.googlesource.com/c/kernel/common/+/2132134 but we're still discussing possible way to go there: go/fuse-bpf-revalidate-problem Change-Id: I517941a2c341999dc8133b93cf045ec67bcf8a9e Signed-off-by: Dmitrii Merkurev --- fs/fuse/dir.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 5a1579b62566..ceda69466011 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -212,7 +212,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) { struct fuse_err_ret fer; - fer = fuse_bpf_backing(inode, + fer = fuse_bpf_backing(entry->d_parent->d_inode, struct fuse_lookup_io, fuse_lookup_initialize, fuse_revalidate_backing, From 95d31c5b4c587d5b2f0fc01bb8b64efdff9d7efb Mon Sep 17 00:00:00 2001 From: Dmitrii Merkurev Date: Wed, 29 Jun 2022 22:17:20 +0000 Subject: [PATCH 099/148] UPSTREAM: ANDROID: fuse-bpf: Make inodes with backing_fd reachable for regular FUSE fuse_iget MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently, when we’re trying to find inode based on their backing inode we strictly checking on nodeid == 0, so basically we’re not supporting nodeid != 0 for inode, which is backed by another one. Alongside with this, we’re using backing_inode as a hash for inode which make this inode not reachable for regular FUSE fuse_iget that as a result causing backing_inode losing because instead of getting existent one (with backing inode) we create a new one as a part of readdirplus. For more details please check: go/fuse-loosing-inode-with-backing Bug: 219958836 Test: Manually checked that /data and /obb inodes always have inode numbers configured. Co-developed-by: Paul Lawrence Change-Id: If6a5fb340561ac6320d3c4e86215f1bcd4c2c10c Signed-off-by: Dmitrii Merkurev --- fs/fuse/backing.c | 34 ++++++++++++++++++++++++---------- fs/fuse/dir.c | 2 +- fs/fuse/fuse_i.h | 3 ++- fs/fuse/inode.c | 31 ++++++++++++++++++++++++++++--- 4 files changed, 55 insertions(+), 15 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index b68f4f534974..0c39771a9527 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -262,6 +262,8 @@ int fuse_create_open_backing( struct dentry *newent; int err = 0; const struct fuse_create_in *fci = fa->in_args[0].value; + struct fuse_inode *fuse_inode = get_fuse_inode(entry->d_inode); + u64 target_nodeid = 0; if (!dir_fuse_inode || !dir_fuse_dentry) return -EIO; @@ -293,7 +295,10 @@ int fuse_create_open_backing( }; path_get(&get_fuse_dentry(entry)->backing_path); - inode = fuse_iget_backing(dir->i_sb, + if (fuse_inode) + target_nodeid = fuse_inode->nodeid; + + inode = fuse_iget_backing(dir->i_sb, target_nodeid, get_fuse_dentry(entry)->backing_path.dentry->d_inode); if (IS_ERR(inode)) { err = PTR_ERR(inode); @@ -1173,9 +1178,11 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct fuse_dentry *fd; struct dentry *bd; struct inode *inode, *backing_inode; + struct fuse_inode *fuse_inode = get_fuse_inode(entry->d_inode); struct fuse_entry_out *feo = fa->out_args[0].value; struct fuse_entry_bpf_out *febo = fa->out_args[1].value; struct fuse_entry_bpf *feb = container_of(febo, struct fuse_entry_bpf, out); + u64 target_nodeid = 0; fd = get_fuse_dentry(entry); if (!fd) @@ -1187,7 +1194,10 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, if (!backing_inode) return 0; - inode = fuse_iget_backing(dir->i_sb, backing_inode); + if (fuse_inode) + target_nodeid = fuse_inode->nodeid; + + inode = fuse_iget_backing(dir->i_sb, target_nodeid, backing_inode); if (IS_ERR(inode)) return ERR_PTR(PTR_ERR(inode)); @@ -1349,7 +1359,8 @@ int fuse_mknod_backing( { int err = 0; const struct fuse_mknod_in *fmi = fa->in_args[0].value; - struct inode *backing_inode = get_fuse_inode(dir)->backing_inode; + struct fuse_inode *fuse_inode = get_fuse_inode(dir); + struct inode *backing_inode = fuse_inode->backing_inode; struct path backing_path = {}; struct inode *inode = NULL; @@ -1376,7 +1387,7 @@ int fuse_mknod_backing( */ goto out; } - inode = fuse_iget_backing(dir->i_sb, backing_inode); + inode = fuse_iget_backing(dir->i_sb, fuse_inode->nodeid, backing_inode); if (IS_ERR(inode)) { err = PTR_ERR(inode); goto out; @@ -1425,7 +1436,8 @@ int fuse_mkdir_backing( { int err = 0; const struct fuse_mkdir_in *fmi = fa->in_args[0].value; - struct inode *backing_inode = get_fuse_inode(dir)->backing_inode; + struct fuse_inode *fuse_inode = get_fuse_inode(dir); + struct inode *backing_inode = fuse_inode->backing_inode; struct path backing_path = {}; struct inode *inode = NULL; struct dentry *d; @@ -1453,7 +1465,7 @@ int fuse_mkdir_backing( dput(backing_path.dentry); backing_path.dentry = d; } - inode = fuse_iget_backing(dir->i_sb, backing_inode); + inode = fuse_iget_backing(dir->i_sb, fuse_inode->nodeid, backing_inode); if (IS_ERR(inode)) { err = PTR_ERR(inode); goto out; @@ -1768,7 +1780,8 @@ int fuse_link_backing(struct fuse_bpf_args *fa, struct dentry *entry, struct path backing_new_path = {}; struct dentry *backing_dir_dentry; struct inode *fuse_new_inode = NULL; - struct inode *backing_dir_inode = get_fuse_inode(dir)->backing_inode; + struct fuse_inode *fuse_dir_inode = get_fuse_inode(dir); + struct inode *backing_dir_inode = fuse_dir_inode->backing_inode; get_fuse_backing_path(entry, &backing_old_path); if (!backing_old_path.dentry) @@ -1799,7 +1812,7 @@ int fuse_link_backing(struct fuse_bpf_args *fa, struct dentry *entry, goto out; } - fuse_new_inode = fuse_iget_backing(dir->i_sb, backing_dir_inode); + fuse_new_inode = fuse_iget_backing(dir->i_sb, fuse_dir_inode->nodeid, backing_dir_inode); if (IS_ERR(fuse_new_inode)) { err = PTR_ERR(fuse_new_inode); goto out; @@ -2163,7 +2176,8 @@ int fuse_symlink_backing( struct inode *dir, struct dentry *entry, const char *link, int len) { int err = 0; - struct inode *backing_inode = get_fuse_inode(dir)->backing_inode; + struct fuse_inode *fuse_inode = get_fuse_inode(dir); + struct inode *backing_inode = fuse_inode->backing_inode; struct path backing_path = {}; struct inode *inode = NULL; @@ -2186,7 +2200,7 @@ int fuse_symlink_backing( */ goto out; } - inode = fuse_iget_backing(dir->i_sb, backing_inode); + inode = fuse_iget_backing(dir->i_sb, fuse_inode->nodeid, backing_inode); if (IS_ERR(inode)) { err = PTR_ERR(inode); goto out; diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index ceda69466011..8a3e1fb9ab14 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -544,7 +544,7 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name goto out_queue_forget; backing_inode = backing_file->f_inode; - *inode = fuse_iget_backing(sb, backing_inode); + *inode = fuse_iget_backing(sb, outarg->nodeid, backing_inode); if (!*inode) goto bpf_arg_out; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 2f07d3a1ff28..22cf7ec88e04 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -975,9 +975,10 @@ extern const struct dentry_operations fuse_dentry_operations; extern const struct dentry_operations fuse_root_dentry_operations; /** - * Get a filled in inode + * Get a filled-in inode */ struct inode *fuse_iget_backing(struct super_block *sb, + u64 nodeid, struct inode *backing_inode); struct inode *fuse_iget(struct super_block *sb, u64 nodeid, int generation, struct fuse_attr *attr, diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 8287dbe92967..14bec19ac110 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -331,6 +331,15 @@ static int fuse_inode_eq(struct inode *inode, void *_nodeidp) (struct fuse_inode_identifier *) _nodeidp; struct fuse_inode *fi = get_fuse_inode(inode); + return fii->nodeid == fi->nodeid; +} + +static int fuse_inode_backing_eq(struct inode *inode, void *_nodeidp) +{ + struct fuse_inode_identifier *fii = + (struct fuse_inode_identifier *) _nodeidp; + struct fuse_inode *fi = get_fuse_inode(inode); + return fii->nodeid == fi->nodeid #ifdef CONFIG_FUSE_BPF && fii->backing_inode == fi->backing_inode @@ -345,6 +354,17 @@ static int fuse_inode_set(struct inode *inode, void *_nodeidp) struct fuse_inode *fi = get_fuse_inode(inode); fi->nodeid = fii->nodeid; + + return 0; +} + +static int fuse_inode_backing_set(struct inode *inode, void *_nodeidp) +{ + struct fuse_inode_identifier *fii = + (struct fuse_inode_identifier *) _nodeidp; + struct fuse_inode *fi = get_fuse_inode(inode); + + fi->nodeid = fii->nodeid; #ifdef CONFIG_FUSE_BPF fi->backing_inode = fii->backing_inode; if (fi->backing_inode) @@ -354,20 +374,25 @@ static int fuse_inode_set(struct inode *inode, void *_nodeidp) return 0; } -struct inode *fuse_iget_backing(struct super_block *sb, +struct inode *fuse_iget_backing(struct super_block *sb, u64 nodeid, struct inode *backing_inode) { struct inode *inode; struct fuse_inode *fi; struct fuse_conn *fc = get_fuse_conn_super(sb); struct fuse_inode_identifier fii = { + .nodeid = nodeid, .backing_inode = backing_inode, }; struct fuse_attr attr; + unsigned long hash = (unsigned long) backing_inode; + + if (nodeid) + hash = nodeid; fuse_fill_attr_from_inode(&attr, backing_inode); - inode = iget5_locked(sb, (unsigned long) backing_inode, fuse_inode_eq, - fuse_inode_set, &fii); + inode = iget5_locked(sb, hash, fuse_inode_backing_eq, + fuse_inode_backing_set, &fii); if (!inode) return NULL; From 3f557cd122e74dbf30dfc3a538880fbf2392b4e8 Mon Sep 17 00:00:00 2001 From: Dmitrii Merkurev Date: Mon, 27 Jun 2022 16:21:40 +0000 Subject: [PATCH 100/148] UPSTREAM: ANDROID: fuse-bpf: Make sure force_again flag is false by default MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Usage of uninitialized boolean is potentially can cause annoying and “hard to catch” types of problems. Currently we have 1 case where we use uninitialized boolean: int fuse_readdir(struct file *file, struct dir_context *ctx) And I constantly see that every userspace readdir operation causes an infinite cycle inside the Kernel for my QEMU tests (gcc). This problem isn’t reproducible inside cuttlefish, probably because we use clang toolchain. Bug: 219958836 Test: atest ScopedStorageDeviceTest Test: selftests Change-Id: I2c38056448cd2910e0cb20da5839d7db9ebd26b9 Signed-off-by: Dmitrii Merkurev --- fs/fuse/readdir.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index be0df513a2e0..06446ee6e9d0 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -581,7 +581,8 @@ int fuse_readdir(struct file *file, struct dir_context *ctx) #ifdef CONFIG_FUSE_BPF struct fuse_err_ret fer; - bool force_again, allow_force; + bool allow_force; + bool force_again = false; bool is_continued = false; again: From f72a971ce74b0e65b1b0f7a7c950c913010c1726 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Wed, 13 Jul 2022 15:50:52 -0700 Subject: [PATCH 101/148] UPSTREAM: ANDROID: fuse-bpf: Fix use of get_fuse_inode get_fuse_inode uses container_of, which results in a strange result if the inode is NULL. We should check if the inode is NULL instead. Fixes: 4ad093cae178 ("ANDROID: fuse-bpf: Make inodes with backing_fd reachable") Bug: 219958836 Change-Id: I386c4641edaa4dbc8d8e02f592c959c206851eda Signed-off-by: Daniel Rosenberg --- fs/fuse/backing.c | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 0c39771a9527..3fea31bec40a 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -262,7 +262,7 @@ int fuse_create_open_backing( struct dentry *newent; int err = 0; const struct fuse_create_in *fci = fa->in_args[0].value; - struct fuse_inode *fuse_inode = get_fuse_inode(entry->d_inode); + struct inode *d_inode = entry->d_inode; u64 target_nodeid = 0; if (!dir_fuse_inode || !dir_fuse_dentry) @@ -295,8 +295,8 @@ int fuse_create_open_backing( }; path_get(&get_fuse_dentry(entry)->backing_path); - if (fuse_inode) - target_nodeid = fuse_inode->nodeid; + if (d_inode) + target_nodeid = get_fuse_inode(d_inode)->nodeid; inode = fuse_iget_backing(dir->i_sb, target_nodeid, get_fuse_dentry(entry)->backing_path.dentry->d_inode); @@ -1178,7 +1178,7 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct fuse_dentry *fd; struct dentry *bd; struct inode *inode, *backing_inode; - struct fuse_inode *fuse_inode = get_fuse_inode(entry->d_inode); + struct inode *d_inode = entry->d_inode; struct fuse_entry_out *feo = fa->out_args[0].value; struct fuse_entry_bpf_out *febo = fa->out_args[1].value; struct fuse_entry_bpf *feb = container_of(febo, struct fuse_entry_bpf, out); @@ -1194,8 +1194,8 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, if (!backing_inode) return 0; - if (fuse_inode) - target_nodeid = fuse_inode->nodeid; + if (d_inode) + target_nodeid = get_fuse_inode(d_inode)->nodeid; inode = fuse_iget_backing(dir->i_sb, target_nodeid, backing_inode); From a82e300d0cd72f15bda20b3137ab8f9426c81212 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Wed, 6 Jul 2022 14:57:14 -0700 Subject: [PATCH 102/148] UPSTREAM: ANDROID: fuse: Don't use readdirplus w/ nodeid 0 If we have a nodeid of 0, we've probably got a backing inode, and a regular getattr will be fast. Otherwise, userspace is likely ill suited to properly handle a readdirplus anyways. Test: fuse_test Bug: 219958836 Signed-off-by: Daniel Rosenberg Change-Id: I02f031d87dcc5fcbe1e080e4f8ec92187b00fe2d --- fs/fuse/readdir.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index 06446ee6e9d0..fb795bc4b286 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -20,6 +20,8 @@ static bool fuse_use_readdirplus(struct inode *dir, struct dir_context *ctx) if (!fc->do_readdirplus) return false; + if (fi->nodeid == 0) + return false; if (!fc->readdirplus_auto) return true; if (test_and_clear_bit(FUSE_I_ADVISE_RDPLUS, &fi->state)) From 306c4b88cffebcdf6a69b0b16d79d3c66411c7bd Mon Sep 17 00:00:00 2001 From: Dmitrii Merkurev Date: Fri, 8 Jul 2022 23:38:18 +0000 Subject: [PATCH 103/148] UPSTREAM: ANDROID: fuse-bpf: Fix revalidate error path and backing handling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently we have 2 different problems 1. Every revalidate considered as a error because of added args->out_argvar = true; inside fuse_lookup_init which makes fuse_simple_request return out argument size which is considered as an error by revalidate code. 2. We’re ignoring backing_fd and bpf_program set by daemon lookup code called by revalidate. Problem 1 makes any revalidate (lookup to userspace) useless and any result lead us to the full lookup because it was interpreted as an error. This CL fixes both and introducing revalidate test case which makes sure: 1. We’re receiving only one lookup as a part of revalidate 2. We’re setting backing_fd as a part of revalidate’s lookup result Test is failed before the fix and passed after. Bug: 219958836 Test: Booted device 5 times to make sure we’re not receiving redundant lookups anymore. Test: selftests Signed-off-by: Dmitrii Merkurev Change-Id: Ifa62e56b42ca5580b25682eb5f16b5c91826cf49 --- fs/fuse/backing.c | 177 +++++++++++++++++++++++++++------------------- fs/fuse/dir.c | 67 ++++++++---------- fs/fuse/fuse_i.h | 7 ++ 3 files changed, 141 insertions(+), 110 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 3fea31bec40a..d21ab1a076ec 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1172,6 +1172,105 @@ int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, return 0; } +int handle_inode_backing_fd(struct inode *inode, struct dentry *entry, + struct fuse_entry_bpf_out *febo, + struct fuse_entry_bpf *feb) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_dentry *fd = get_fuse_dentry(entry); + int ret = 0; + + switch (febo->backing_action) { + case FUSE_ACTION_KEEP: + /* backing inode/path are added in fuse_lookup_backing */ + break; + + case FUSE_ACTION_REMOVE: + iput(fi->backing_inode); + fi->backing_inode = NULL; + path_put_init(&fd->backing_path); + break; + + case FUSE_ACTION_REPLACE: { + struct file *backing_file = feb->backing_file; + + if (!backing_file) + return -EINVAL; + if (IS_ERR(backing_file)) + return PTR_ERR(backing_file); + + if (fi->backing_inode) + iput(fi->backing_inode); + fi->backing_inode = backing_file->f_inode; + ihold(fi->backing_inode); + + path_put(&fd->backing_path); + fd->backing_path = backing_file->f_path; + path_get(&fd->backing_path); + + fput(backing_file); + break; + } + + default: + return -EINVAL; + } + + return ret; +} + +int handle_inode_bpf(struct inode *inode, struct inode *parent, + struct fuse_entry_bpf_out *febo, + struct fuse_entry_bpf *feb) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct fuse_inode *pi; + int ret = 0; + + // Parent isn't presented, but we want to keep + // Don't touch bpf program at all in this case + if (febo->bpf_action == FUSE_ACTION_KEEP && !parent) { + goto out; + } + + if (fi->bpf) { + bpf_prog_put(fi->bpf); + fi->bpf = NULL; + } + + switch (febo->bpf_action) { + case FUSE_ACTION_KEEP: + pi = get_fuse_inode(parent); + fi->bpf = pi->bpf; + if (fi->bpf) + bpf_prog_inc(fi->bpf); + break; + + case FUSE_ACTION_REMOVE: + break; + + case FUSE_ACTION_REPLACE: { + struct file *bpf_file = feb->bpf_file; + struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); + + if (bpf_file && !IS_ERR(bpf_file)) + bpf_prog = fuse_get_bpf_prog(bpf_file); + + if (IS_ERR(bpf_prog)) + return PTR_ERR(bpf_prog); + + fi->bpf = bpf_prog; + break; + } + + default: + return -EINVAL; + } + +out: + return ret; +} + struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags) { @@ -1182,6 +1281,7 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct fuse_entry_out *feo = fa->out_args[0].value; struct fuse_entry_bpf_out *febo = fa->out_args[1].value; struct fuse_entry_bpf *feb = container_of(febo, struct fuse_entry_bpf, out); + int error = -1; u64 target_nodeid = 0; fd = get_fuse_dentry(entry); @@ -1202,78 +1302,13 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, if (IS_ERR(inode)) return ERR_PTR(PTR_ERR(inode)); - /* TODO Make sure this handles invalid handles */ - /* TODO Do we need the same code in revalidate */ - if (get_fuse_inode(inode)->bpf) { - bpf_prog_put(get_fuse_inode(inode)->bpf); - get_fuse_inode(inode)->bpf = NULL; - } + error = handle_inode_bpf(inode, dir, febo, feb); + if (error) + return ERR_PTR(error); - switch (febo->bpf_action) { - case FUSE_ACTION_KEEP: - get_fuse_inode(inode)->bpf = get_fuse_inode(dir)->bpf; - if (get_fuse_inode(inode)->bpf) - bpf_prog_inc(get_fuse_inode(inode)->bpf); - break; - - case FUSE_ACTION_REMOVE: - get_fuse_inode(inode)->bpf = NULL; - break; - - case FUSE_ACTION_REPLACE: { - struct file *bpf_file = feb->bpf_file; - struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); - - if (bpf_file && !IS_ERR(bpf_file)) - bpf_prog = fuse_get_bpf_prog(bpf_file); - - if (IS_ERR(bpf_prog)) - return ERR_PTR(PTR_ERR(bpf_prog)); - - get_fuse_inode(inode)->bpf = bpf_prog; - break; - } - - default: - return ERR_PTR(-EIO); - } - - switch (febo->backing_action) { - case FUSE_ACTION_KEEP: - /* backing inode/path are added in fuse_lookup_backing */ - break; - - case FUSE_ACTION_REMOVE: - iput(get_fuse_inode(inode)->backing_inode); - get_fuse_inode(inode)->backing_inode = NULL; - path_put_init(&get_fuse_dentry(entry)->backing_path); - break; - - case FUSE_ACTION_REPLACE: { - struct fuse_conn *fc; - struct file *backing_file; - - fc = get_fuse_mount(dir)->fc; - backing_file = feb->backing_file; - if (!backing_file || IS_ERR(backing_file)) - return ERR_PTR(-EIO); - - iput(get_fuse_inode(inode)->backing_inode); - get_fuse_inode(inode)->backing_inode = - backing_file->f_inode; - ihold(get_fuse_inode(inode)->backing_inode); - - path_put(&get_fuse_dentry(entry)->backing_path); - get_fuse_dentry(entry)->backing_path = backing_file->f_path; - path_get(&get_fuse_dentry(entry)->backing_path); - - fput(backing_file); - break; - } - - default: - return ERR_PTR(-EIO); - } + error = handle_inode_backing_fd(inode, entry, febo, feb); + if (error) + return ERR_PTR(error); get_fuse_inode(inode)->nodeid = feo->nodeid; diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 8a3e1fb9ab14..ff1087671067 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -238,25 +238,23 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) ret = fuse_simple_request(fm, &args); dput(parent); - /* - * TODO This doesn't seem sufficient, though we don't plan to - * change the backing file ever, so not sure what is correct - * here yet, especially as we can't return an error to user - */ - if (bpf_arg.out.backing_action == FUSE_ACTION_REPLACE) { - struct file *file = bpf_arg.backing_file; +#ifdef CONFIG_FUSE_BPF + if (ret == sizeof(bpf_arg.out)) { + ret = -ENOENT; + if (!entry) + goto out; - if (file && !IS_ERR(file)) - fput(file); + ret = handle_inode_backing_fd(inode, entry, + &bpf_arg.out, &bpf_arg); + if (ret) + goto out; + + ret = handle_inode_bpf(inode, entry->d_parent->d_inode, + &bpf_arg.out, &bpf_arg); + if (ret) + goto out; } - - if (bpf_arg.out.bpf_action == FUSE_ACTION_REPLACE) { - struct file *file = bpf_arg.bpf_file; - - if (file && !IS_ERR(file)) - fput(file); - } - +#endif /* Zero nodeid is same as -ENOENT */ if (!ret && !outarg.nodeid) ret = -ENOENT; @@ -527,7 +525,6 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name #ifdef CONFIG_FUSE_BPF if (err == sizeof(bpf_arg.out)) { /* TODO Make sure this handles invalid handles */ - /* TODO Do we need the same code in revalidate */ struct file *backing_file; struct inode *backing_inode; @@ -536,37 +533,29 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name goto out_queue_forget; err = -EINVAL; - if (bpf_arg.out.backing_action != FUSE_ACTION_REPLACE) + backing_file = bpf_arg.backing_file; + if (!backing_file) goto out_queue_forget; - backing_file = bpf_arg.backing_file; - if (!backing_file || IS_ERR(backing_file)) + if (IS_ERR(backing_file)) { + err = PTR_ERR(backing_file); goto out_queue_forget; + } backing_inode = backing_file->f_inode; *inode = fuse_iget_backing(sb, outarg->nodeid, backing_inode); if (!*inode) goto bpf_arg_out; - if (bpf_arg.out.bpf_action == FUSE_ACTION_REPLACE) { - struct file *bpf_file = bpf_arg.bpf_file; - struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); - - if (bpf_file && !IS_ERR(bpf_file)) - bpf_prog = fuse_get_bpf_prog(bpf_file);; - - if (IS_ERR(bpf_prog)) { - iput(*inode); - *inode = NULL; - err = PTR_ERR(bpf_prog); - goto bpf_arg_out; - } - get_fuse_inode(*inode)->bpf = bpf_prog; - } - - get_fuse_dentry(entry)->backing_path = backing_file->f_path; - path_get(&get_fuse_dentry(entry)->backing_path); + err = handle_inode_backing_fd(*inode, entry, + &bpf_arg.out, &bpf_arg); + if (err) + goto out; + err = handle_inode_bpf(*inode, NULL, + &bpf_arg.out, &bpf_arg); + if (err) + goto out; bpf_arg_out: fput(backing_file); } else diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 22cf7ec88e04..aa4509e20235 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1595,6 +1595,13 @@ struct fuse_lookup_io { struct fuse_entry_bpf feb; }; +int handle_inode_backing_fd(struct inode *inode, struct dentry *entry, + struct fuse_entry_bpf_out *febo, + struct fuse_entry_bpf *feb); +int handle_inode_bpf(struct inode *inode, struct inode *parent, + struct fuse_entry_bpf_out *febo, + struct fuse_entry_bpf *feb); + int fuse_lookup_initialize(struct fuse_bpf_args *fa, struct fuse_lookup_io *feo, struct inode *dir, struct dentry *entry, unsigned int flags); int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, From 1874580176585759b28b52e8427dafd9168c067e Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Thu, 14 Jul 2022 18:16:16 -0700 Subject: [PATCH 104/148] UPSTREAM: ANDROID: fuse-bpf: Adjust backing handle funcs Cleaned up some args, and adjusted so we can choose to not update the actual values for the next patch. Test: fuse_test Bug: 219958836 Change-Id: I025b7026357b119e2cd588f25f0976f5d4b68090 Signed-off-by: Daniel Rosenberg --- fs/fuse/backing.c | 65 ++++++++++++++++++++--------------------------- fs/fuse/dir.c | 18 ++++++------- fs/fuse/fuse_i.h | 10 +++----- 3 files changed, 41 insertions(+), 52 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index d21ab1a076ec..db71e77f2bfb 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1172,23 +1172,17 @@ int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, return 0; } -int handle_inode_backing_fd(struct inode *inode, struct dentry *entry, - struct fuse_entry_bpf_out *febo, - struct fuse_entry_bpf *feb) -{ - struct fuse_inode *fi = get_fuse_inode(inode); - struct fuse_dentry *fd = get_fuse_dentry(entry); - int ret = 0; - - switch (febo->backing_action) { +int fuse_handle_backing(struct fuse_entry_bpf *feb, struct inode **backing_inode, + struct path *backing_path) { + switch (feb->out.backing_action) { case FUSE_ACTION_KEEP: /* backing inode/path are added in fuse_lookup_backing */ break; case FUSE_ACTION_REMOVE: - iput(fi->backing_inode); - fi->backing_inode = NULL; - path_put_init(&fd->backing_path); + iput(*backing_inode); + *backing_inode = NULL; + path_put_init(backing_path); break; case FUSE_ACTION_REPLACE: { @@ -1199,14 +1193,14 @@ int handle_inode_backing_fd(struct inode *inode, struct dentry *entry, if (IS_ERR(backing_file)) return PTR_ERR(backing_file); - if (fi->backing_inode) - iput(fi->backing_inode); - fi->backing_inode = backing_file->f_inode; - ihold(fi->backing_inode); + if (backing_inode) + iput(*backing_inode); + *backing_inode = backing_file->f_inode; + ihold(*backing_inode); - path_put(&fd->backing_path); - fd->backing_path = backing_file->f_path; - path_get(&fd->backing_path); + path_put(backing_path); + *backing_path = backing_file->f_path; + path_get(backing_path); fput(backing_file); break; @@ -1216,34 +1210,31 @@ int handle_inode_backing_fd(struct inode *inode, struct dentry *entry, return -EINVAL; } - return ret; + return 0; } -int handle_inode_bpf(struct inode *inode, struct inode *parent, - struct fuse_entry_bpf_out *febo, - struct fuse_entry_bpf *feb) +int fuse_handle_bpf_prog(struct fuse_entry_bpf *feb, struct inode *parent, + struct bpf_prog **bpf) { - struct fuse_inode *fi = get_fuse_inode(inode); struct fuse_inode *pi; - int ret = 0; // Parent isn't presented, but we want to keep // Don't touch bpf program at all in this case - if (febo->bpf_action == FUSE_ACTION_KEEP && !parent) { + if (feb->out.bpf_action == FUSE_ACTION_KEEP && !parent) { goto out; } - if (fi->bpf) { - bpf_prog_put(fi->bpf); - fi->bpf = NULL; + if (*bpf) { + bpf_prog_put(*bpf); + *bpf = NULL; } - switch (febo->bpf_action) { + switch (feb->out.bpf_action) { case FUSE_ACTION_KEEP: pi = get_fuse_inode(parent); - fi->bpf = pi->bpf; - if (fi->bpf) - bpf_prog_inc(fi->bpf); + *bpf = pi->bpf; + if (*bpf) + bpf_prog_inc(*bpf); break; case FUSE_ACTION_REMOVE: @@ -1259,7 +1250,7 @@ int handle_inode_bpf(struct inode *inode, struct inode *parent, if (IS_ERR(bpf_prog)) return PTR_ERR(bpf_prog); - fi->bpf = bpf_prog; + *bpf = bpf_prog; break; } @@ -1268,7 +1259,7 @@ int handle_inode_bpf(struct inode *inode, struct inode *parent, } out: - return ret; + return 0; } struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, @@ -1302,11 +1293,11 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, if (IS_ERR(inode)) return ERR_PTR(PTR_ERR(inode)); - error = handle_inode_bpf(inode, dir, febo, feb); + error = fuse_handle_bpf_prog(feb, dir, &get_fuse_inode(inode)->bpf); if (error) return ERR_PTR(error); - error = handle_inode_backing_fd(inode, entry, febo, feb); + error = fuse_handle_backing(feb, &get_fuse_inode(inode)->backing_inode, &fd->backing_path); if (error) return ERR_PTR(error); diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index ff1087671067..f288ac69b934 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -236,7 +236,6 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) fuse_lookup_init(fm->fc, &args, get_node_id(d_inode(parent)), &entry->d_name, &outarg, &bpf_arg.out); ret = fuse_simple_request(fm, &args); - dput(parent); #ifdef CONFIG_FUSE_BPF if (ret == sizeof(bpf_arg.out)) { @@ -244,17 +243,18 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) if (!entry) goto out; - ret = handle_inode_backing_fd(inode, entry, - &bpf_arg.out, &bpf_arg); + ret = fuse_handle_backing(&bpf_arg, &get_fuse_inode(inode)->backing_inode, + &get_fuse_dentry(entry)->backing_path); if (ret) goto out; - ret = handle_inode_bpf(inode, entry->d_parent->d_inode, - &bpf_arg.out, &bpf_arg); + ret = fuse_handle_bpf_prog(&bpf_arg, parent->d_inode, + &get_fuse_inode(inode)->bpf); if (ret) goto out; } #endif + dput(parent); /* Zero nodeid is same as -ENOENT */ if (!ret && !outarg.nodeid) ret = -ENOENT; @@ -547,13 +547,13 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name if (!*inode) goto bpf_arg_out; - err = handle_inode_backing_fd(*inode, entry, - &bpf_arg.out, &bpf_arg); + err = fuse_handle_backing(&bpf_arg, + &get_fuse_inode(*inode)->backing_inode, + &get_fuse_dentry(entry)->backing_path); if (err) goto out; - err = handle_inode_bpf(*inode, NULL, - &bpf_arg.out, &bpf_arg); + err = fuse_handle_bpf_prog(&bpf_arg, NULL, &get_fuse_inode(*inode)->bpf); if (err) goto out; bpf_arg_out: diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index aa4509e20235..172721f2676b 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1595,12 +1595,10 @@ struct fuse_lookup_io { struct fuse_entry_bpf feb; }; -int handle_inode_backing_fd(struct inode *inode, struct dentry *entry, - struct fuse_entry_bpf_out *febo, - struct fuse_entry_bpf *feb); -int handle_inode_bpf(struct inode *inode, struct inode *parent, - struct fuse_entry_bpf_out *febo, - struct fuse_entry_bpf *feb); +int fuse_handle_backing(struct fuse_entry_bpf *feb, struct inode **backing_inode, + struct path *backing_path); +int fuse_handle_bpf_prog(struct fuse_entry_bpf *feb, struct inode *parent, + struct bpf_prog **bpf); int fuse_lookup_initialize(struct fuse_bpf_args *fa, struct fuse_lookup_io *feo, struct inode *dir, struct dentry *entry, unsigned int flags); From 4e5cea6bec20e45fddbd639f62f2fbbc18ca2705 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Mon, 11 Jul 2022 15:05:15 -0700 Subject: [PATCH 105/148] UPSTREAM: ANDROID: fuse-bpf: Always call revalidate for backing If we have a backing dentry, we must call it's revalidate always, or we may end up using an invalid lower dentry. Revalidate is called for dentries, not inodes, and the dentry may be negative. This goes against all of the macro conventions, so we're just calling the backing function directly. Signed-off-by: Daniel Rosenberg Bug: 219958836 Test: fuse_test Change-Id: Ia28da5bd2ce42b40466c577137d5440d9f2f6600 --- fs/fuse/backing.c | 9 +--- fs/fuse/dir.c | 108 +++++++++++++++++++++++++++++++--------------- fs/fuse/fuse_i.h | 5 +-- 3 files changed, 76 insertions(+), 46 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index db71e77f2bfb..b536fd9e99b7 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1306,8 +1306,7 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, return d_splice_alias(inode, entry); } -int fuse_revalidate_backing(struct fuse_bpf_args *fa, struct inode *dir, - struct dentry *entry, unsigned int flags) +int fuse_revalidate_backing(struct dentry *entry, unsigned int flags) { struct fuse_dentry *fuse_dentry = get_fuse_dentry(entry); struct dentry *backing_entry = fuse_dentry->backing_path.dentry; @@ -1324,12 +1323,6 @@ int fuse_revalidate_backing(struct fuse_bpf_args *fa, struct inode *dir, return 1; } -void *fuse_revalidate_finalize(struct fuse_bpf_args *fa, struct inode *dir, - struct dentry *entry, unsigned int flags) -{ - return 0; -} - int fuse_canonical_path_initialize(struct fuse_bpf_args *fa, struct fuse_dummy_io *fdi, const struct path *path, diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index f288ac69b934..8b0e22a85c49 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -173,6 +173,44 @@ static void fuse_lookup_init(struct fuse_conn *fc, struct fuse_args *args, args->out_args[1].value = bpf_outarg; } +#ifdef CONFIG_FUSE_BPF +static bool backing_data_changed(struct fuse_inode *fi, struct dentry *entry, + struct fuse_entry_bpf *bpf_arg) +{ + struct path new_backing_path; + struct inode *new_backing_inode; + struct bpf_prog *bpf = NULL; + int err; + bool ret = true; + + if (!entry) + return false; + + get_fuse_backing_path(entry, &new_backing_path); + new_backing_inode = fi->backing_inode; + ihold(new_backing_inode); + + err = fuse_handle_backing(bpf_arg, &new_backing_inode, &new_backing_path); + + if (err) + goto put_inode; + + err = fuse_handle_bpf_prog(bpf_arg, entry->d_parent->d_inode, &bpf); + if (err) + goto put_bpf; + + ret = (bpf != fi->bpf || fi->backing_inode != new_backing_inode || + !path_equal(&get_fuse_dentry(entry)->backing_path, &new_backing_path)); +put_bpf: + if (bpf) + bpf_prog_put(bpf); +put_inode: + iput(new_backing_inode); + path_put(&new_backing_path); + return ret; +} +#endif + /* * Check whether the dentry is still valid * @@ -193,7 +231,28 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) inode = d_inode_rcu(entry); if (inode && fuse_is_bad(inode)) goto invalid; - else if (time_before64(fuse_dentry_time(entry), get_jiffies_64()) || + +#ifdef CONFIG_FUSE_BPF + /* TODO: Do we need bpf support for revalidate? + * If the lower filesystem says the entry is invalid, FUSE probably shouldn't + * try to fix that without going through the normal lookup path... + */ + if (get_fuse_dentry(entry)->backing_path.dentry) { + ret = fuse_revalidate_backing(entry, flags); + if (ret <= 0) { + goto out; + } + } + /* TODO: Respect timeouts for lookups with backing inodes */ + parent = dget_parent(entry); + if (get_fuse_inode(d_inode_rcu(parent))->backing_inode) { + dput(parent); + ret = 1; + goto out; + } + dput(parent); +#endif + if (time_before64(fuse_dentry_time(entry), get_jiffies_64()) || (flags & LOOKUP_REVAL)) { struct fuse_entry_out outarg; struct fuse_entry_bpf bpf_arg; @@ -208,20 +267,6 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) ret = -ECHILD; if (flags & LOOKUP_RCU) goto out; -#ifdef CONFIG_FUSE_BPF - { - struct fuse_err_ret fer; - - fer = fuse_bpf_backing(entry->d_parent->d_inode, - struct fuse_lookup_io, - fuse_lookup_initialize, - fuse_revalidate_backing, - fuse_revalidate_finalize, - d_inode(entry->d_parent), entry, flags); - if (fer.ret) - return PTR_ERR(fer.result); - } -#endif fm = get_fuse_mount(inode); forget = fuse_alloc_forget(); @@ -233,34 +278,29 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) parent = dget_parent(entry); + /* TODO: Once we're handling timeouts for backing inodes, do a + * bpf based lookup_revalidate here. + */ + if (get_fuse_inode(parent->d_inode)->backing_inode) { + ret = 1; + goto out; + } + fuse_lookup_init(fm->fc, &args, get_node_id(d_inode(parent)), &entry->d_name, &outarg, &bpf_arg.out); ret = fuse_simple_request(fm, &args); - -#ifdef CONFIG_FUSE_BPF - if (ret == sizeof(bpf_arg.out)) { - ret = -ENOENT; - if (!entry) - goto out; - - ret = fuse_handle_backing(&bpf_arg, &get_fuse_inode(inode)->backing_inode, - &get_fuse_dentry(entry)->backing_path); - if (ret) - goto out; - - ret = fuse_handle_bpf_prog(&bpf_arg, parent->d_inode, - &get_fuse_inode(inode)->bpf); - if (ret) - goto out; - } -#endif dput(parent); + /* Zero nodeid is same as -ENOENT */ if (!ret && !outarg.nodeid) ret = -ENOENT; - if (!ret) { + if (!ret || ret == sizeof(bpf_arg.out)) { fi = get_fuse_inode(inode); if (outarg.nodeid != get_node_id(inode) || +#ifdef CONFIG_FUSE_BPF + (ret == sizeof(bpf_arg.out) && + backing_data_changed(fi, entry, &bpf_arg)) || +#endif (bool) IS_AUTOMOUNT(inode) != (bool) (outarg.attr.flags & FUSE_ATTR_SUBMOUNT)) { fuse_queue_forget(fm->fc, forget, outarg.nodeid, 1); diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 172721f2676b..71d01a96bced 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1606,10 +1606,7 @@ int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags); struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags); -int fuse_revalidate_backing(struct fuse_bpf_args *fa, struct inode *dir, - struct dentry *entry, unsigned int flags); -void *fuse_revalidate_finalize(struct fuse_bpf_args *fa, struct inode *dir, - struct dentry *entry, unsigned int flags); +int fuse_revalidate_backing(struct dentry *entry, unsigned int flags); int fuse_canonical_path_initialize(struct fuse_bpf_args *fa, struct fuse_dummy_io *fdi, From e36a3e56c927f03d1a491c6744480a0d00f8c896 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Mon, 25 Jul 2022 14:28:33 -0700 Subject: [PATCH 106/148] UPSTREAM: ANDROID: fuse-bpf: Fix RCU/reference issue 094905c87771 ("ANDROID: fuse-bpf: Always call revalidate for backing") called dget_parent/dput improperly within an RCU context. Additionally, it failed to free/put some references. Fixes: 094905c87771 ("ANDROID: fuse-bpf: Always call revalidate for backing") Signed-off-by: Daniel Rosenberg Change-Id: Iedeee7550ff88366bc5310eedece285019336814 --- fs/fuse/dir.c | 25 ++++++++++--------------- 1 file changed, 10 insertions(+), 15 deletions(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 8b0e22a85c49..d16cab30342c 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -243,14 +243,6 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) goto out; } } - /* TODO: Respect timeouts for lookups with backing inodes */ - parent = dget_parent(entry); - if (get_fuse_inode(d_inode_rcu(parent))->backing_inode) { - dput(parent); - ret = 1; - goto out; - } - dput(parent); #endif if (time_before64(fuse_dentry_time(entry), get_jiffies_64()) || (flags & LOOKUP_REVAL)) { @@ -269,23 +261,26 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) goto out; fm = get_fuse_mount(inode); - forget = fuse_alloc_forget(); - ret = -ENOMEM; - if (!forget) - goto out; - - attr_version = fuse_get_attr_version(fm->fc); - parent = dget_parent(entry); /* TODO: Once we're handling timeouts for backing inodes, do a * bpf based lookup_revalidate here. */ if (get_fuse_inode(parent->d_inode)->backing_inode) { + dput(parent); ret = 1; goto out; } + forget = fuse_alloc_forget(); + ret = -ENOMEM; + if (!forget) { + dput(parent); + goto out; + } + + attr_version = fuse_get_attr_version(fm->fc); + fuse_lookup_init(fm->fc, &args, get_node_id(d_inode(parent)), &entry->d_name, &outarg, &bpf_arg.out); ret = fuse_simple_request(fm, &args); From defb79fa30672f6eeee765147bb2e5d86beab825 Mon Sep 17 00:00:00 2001 From: Todd Kjos Date: Fri, 29 Jul 2022 00:59:18 +0000 Subject: [PATCH 107/148] UPSTREAM: ANDROID: fix kernelci error in fs/fuse/dir.c kernelci reported the following error when !CONFIG_FUSE_BPF: fs/fuse/dir.c:269:40: error: no member named 'backing_inode' in 'struct fuse_inode' Fixed by wrapping that new code in #ifdef CONFIG_FUSE_BPF. Fixes: 094905c87771 ("ANDROID: fuse-bpf: Always call revalidate for backing") Signed-off-by: Todd Kjos Change-Id: I08760f340cebf21d3084ca84ac1792bc70f311ff --- fs/fuse/dir.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index d16cab30342c..6071e12dbd6c 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -263,6 +263,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) parent = dget_parent(entry); +#ifdef CONFIG_FUSE_BPF /* TODO: Once we're handling timeouts for backing inodes, do a * bpf based lookup_revalidate here. */ @@ -271,7 +272,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) ret = 1; goto out; } - +#endif forget = fuse_alloc_forget(); ret = -ENOMEM; if (!forget) { From 3937d62267251d5fe0d4f1345dfd97d9aec20b21 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Mon, 15 Aug 2022 10:12:59 -0700 Subject: [PATCH 108/148] UPSTREAM: ANDROID: fuse-bpf: readddir postfilter fixes Test: fuse_test passes, with appropriate user code directories can be hidden Bug: 219958836 Signed-off-by: Paul Lawrence Change-Id: Ia77d4bf9e63d0dd4535e53e9e07c3216fd4b8430 --- fs/fuse/backing.c | 25 ++++++++++++++++++++++--- 1 file changed, 22 insertions(+), 3 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index b536fd9e99b7..49b848e4462d 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -23,6 +23,9 @@ struct fuse_bpf_aio_req { static struct kmem_cache *fuse_bpf_aio_request_cachep; +static void fuse_stat_to_attr(struct fuse_conn *fc, struct inode *inode, + struct kstat *stat, struct fuse_attr *attr); + static void fuse_file_accessed(struct file *dst_file, struct file *src_file) { struct inode *dst_inode; @@ -181,8 +184,10 @@ void *fuse_open_finalize(struct fuse_bpf_args *fa, struct fuse_file *ff = file->private_data; struct fuse_open_out *foo = fa->out_args[0].value; - if (ff) + if (ff) { ff->fh = foo->fh; + ff->nodeid = get_fuse_inode(inode)->nodeid; + } return 0; } @@ -1153,6 +1158,9 @@ int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *dir_backing_entry = dir_fuse_entry->backing_path.dentry; struct inode *dir_backing_inode = dir_backing_entry->d_inode; struct dentry *backing_entry; + struct fuse_entry_out *feo = (void *)fa->out_args[0].value; + struct kstat stat; + int err; /* TODO this will not handle lookups over mount points */ inode_lock_nested(dir_backing_inode, I_MUTEX_PARENT); @@ -1165,10 +1173,21 @@ int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, fuse_entry->backing_path = (struct path) { .dentry = backing_entry, - .mnt = dir_fuse_entry->backing_path.mnt, + .mnt = mntget(dir_fuse_entry->backing_path.mnt), }; - mntget(fuse_entry->backing_path.mnt); + if (d_is_negative(backing_entry)) + return 0; + + err = vfs_getattr(&fuse_entry->backing_path, &stat, + STATX_BASIC_STATS, 0); + if (err) { + path_put_init(&fuse_entry->backing_path); + return err; + } + + fuse_stat_to_attr(get_fuse_conn(dir), + backing_entry->d_inode, &stat, &feo->attr); return 0; } From 9769804365a054861ba2a547d20d38473ca0c973 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Mon, 3 Oct 2022 09:56:21 -0700 Subject: [PATCH 109/148] UPSTREAM: ANDROID: fuse-bpf: set error_in to ENOENT in negative lookup Bug: 250617797 Test: fuse-test passes, app installs Signed-off-by: Paul Lawrence Change-Id: I44544f8989d0ddf00a98eaddce5751df6598130f --- fs/fuse/backing.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 49b848e4462d..9f78384b021f 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1176,8 +1176,10 @@ int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, .mnt = mntget(dir_fuse_entry->backing_path.mnt), }; - if (d_is_negative(backing_entry)) + if (d_is_negative(backing_entry)) { + fa->error_in = -ENOENT; return 0; + } err = vfs_getattr(&fuse_entry->backing_path, &stat, STATX_BASIC_STATS, 0); From ac8fa186fc9293bb02eb28e56d5d1cca9ee89d6c Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Thu, 20 Oct 2022 17:18:58 +0200 Subject: [PATCH 110/148] UPSTREAM: fuse: fix readdir cache race [ Upstream commit 9fa248c65bdbf5af0a2f74dd38575acfc8dfd2bf ] There's a race in fuse's readdir cache that can result in an uninitilized page being read. The page lock is supposed to prevent this from happening but in the following case it doesn't: Two fuse_add_dirent_to_cache() start out and get the same parameters (size=0,offset=0). One of them wins the race to create and lock the page, after which it fills in data, sets rdc.size and unlocks the page. In the meantime the page gets evicted from the cache before the other instance gets to run. That one also creates the page, but finds the size to be mismatched, bails out and leaves the uninitialized page in the cache. Fix by marking a filled page uptodate and ignoring non-uptodate pages. Reported-by: Frank Sorenson Fixes: 5d7bc7e8680c ("fuse: allow using readdir cache") Cc: # v4.20 Change-Id: I1dea0531dd4d2cb0b89b8e1ee6796339fc2dc624 Signed-off-by: Miklos Szeredi Signed-off-by: Sasha Levin --- fs/fuse/readdir.c | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c index fb795bc4b286..f9d0be56d34a 100644 --- a/fs/fuse/readdir.c +++ b/fs/fuse/readdir.c @@ -79,8 +79,10 @@ static void fuse_add_dirent_to_cache(struct file *file, goto unlock; addr = kmap_atomic(page); - if (!offset) + if (!offset) { clear_page(addr); + SetPageUptodate(page); + } memcpy(addr + offset, dirent, reclen); kunmap_atomic(addr); fi->rdc.size = (index << PAGE_SHIFT) + offset + reclen; @@ -526,6 +528,12 @@ retry_locked: page = find_get_page_flags(file->f_mapping, index, FGP_ACCESSED | FGP_LOCK); + /* Page gone missing, then re-added to cache, but not initialized? */ + if (page && !PageUptodate(page)) { + unlock_page(page); + put_page(page); + page = NULL; + } spin_lock(&fi->rdc.lock); if (!page) { /* From 010f25aa35aeeb1b26b7f56bea054978263dac55 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Mon, 5 Dec 2022 15:22:09 -0800 Subject: [PATCH 111/148] UPSTREAM: ANDROID: fuse-bpf: Add /sys/fs flags for fuse-bpf version Note that this is specific for the non-upstreamed version Bug: 202785178 Test: cat /sys/fs/fuse/fuse_bpf_major_version Change-Id: I68f9ca56778874975428839dfc1fd8f48b11bd75 Signed-off-by: Paul Lawrence --- Documentation/ABI/testing/sysfs-fs-fuse | 19 +++++++++++++ fs/fuse/inode.c | 36 +++++++++++++++++++++++-- 2 files changed, 53 insertions(+), 2 deletions(-) create mode 100644 Documentation/ABI/testing/sysfs-fs-fuse diff --git a/Documentation/ABI/testing/sysfs-fs-fuse b/Documentation/ABI/testing/sysfs-fs-fuse new file mode 100644 index 000000000000..b9956842b36f --- /dev/null +++ b/Documentation/ABI/testing/sysfs-fs-fuse @@ -0,0 +1,19 @@ +What: /sys/fs/fuse/features/fuse_bpf +Date: December 2022 +Contact: Paul Lawrence +Description: + Read-only file that contains the word 'supported' if fuse-bpf is + supported, does not exist otherwise + +What: /sys/fs/fuse/bpf_prog_type_fuse +Date: December 2022 +Contact: Paul Lawrence +Description: + bpf_prog_type_fuse defines the program type of bpf programs that + may be passed to fuse-bpf. For upstream bpf program types, this + is a constant defined in a contiguous array of constants. + bpf_prog_type_fuse is appended to the end of the list, so it may + change and therefore its value must be read from this file. + + Contents is ASCII decimal representation of bpf_prog_type_fuse + diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 14bec19ac110..c0d9ca6ea074 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1857,7 +1857,31 @@ static void fuse_fs_cleanup(void) static struct kobject *fuse_kobj; -/* TODO Remove this once BPF_PROG_TYPE_FUSE is upstreamed */ +static ssize_t fuse_bpf_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buff) +{ + return sysfs_emit(buff, "supported\n"); +} + +static struct kobj_attribute fuse_bpf_attr = + __ATTR_RO(fuse_bpf); + +static struct attribute *bpf_features[] = { + &fuse_bpf_attr.attr, + NULL, +}; + +static const struct attribute_group bpf_features_group = { + .name = "features", + .attrs = bpf_features, +}; + +/* + * TODO Remove this once fuse-bpf is upstreamed + * + * bpf_prog_type_fuse exports the bpf_prog_type_fuse 'constant', which cannot be + * constant until the code is upstreamed + */ static ssize_t bpf_prog_type_fuse_show(struct kobject *kobj, struct kobj_attribute *attr, char *buff) { @@ -1875,6 +1899,13 @@ static struct attribute *bpf_attributes[] = { static const struct attribute_group bpf_attr_group = { .attrs = bpf_attributes, }; + +static const struct attribute_group *attribute_groups[] = { + &bpf_features_group, + &bpf_attr_group, + NULL +}; + /* TODO remove to here */ static int fuse_sysfs_init(void) @@ -1892,7 +1923,7 @@ static int fuse_sysfs_init(void) goto out_fuse_unregister; /* TODO Remove when BPF_PROG_TYPE_FUSE is upstreamed */ - err = sysfs_create_group(fuse_kobj, &bpf_attr_group); + err = sysfs_create_groups(fuse_kobj, attribute_groups); if (err) goto out_fuse_remove_mount_point; @@ -1908,6 +1939,7 @@ static int fuse_sysfs_init(void) static void fuse_sysfs_cleanup(void) { + sysfs_remove_groups(fuse_kobj, attribute_groups); sysfs_remove_mount_point(fuse_kobj, "connections"); kobject_put(fuse_kobj); } From 3a772729dddbfd427dff48a44b0d7dd0bb4addc3 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Wed, 11 Nov 2020 17:22:32 +0100 Subject: [PATCH 112/148] BACKPORT: fuse: always revalidate if exclusive create commit df8629af293493757beccac2d3168fe5a315636e upstream. Failure to do so may result in EEXIST even if the file only exists in the cache and not in the filesystem. The atomic nature of O_EXCL mandates that the cached state should be ignored and existence verified anew. Reported-by: Ken Schalk Change-Id: I7bc67048363004f94dfe7b74d0873c019f06eaef Signed-off-by: Miklos Szeredi Signed-off-by: Wu Bo Signed-off-by: Greg Kroah-Hartman --- fs/fuse/dir.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 6071e12dbd6c..59c36841e86a 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -231,7 +231,6 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) inode = d_inode_rcu(entry); if (inode && fuse_is_bad(inode)) goto invalid; - #ifdef CONFIG_FUSE_BPF /* TODO: Do we need bpf support for revalidate? * If the lower filesystem says the entry is invalid, FUSE probably shouldn't @@ -245,7 +244,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) } #endif if (time_before64(fuse_dentry_time(entry), get_jiffies_64()) || - (flags & LOOKUP_REVAL)) { + (flags & (LOOKUP_EXCL | LOOKUP_REVAL))) { struct fuse_entry_out outarg; struct fuse_entry_bpf bpf_arg; FUSE_ARGS(args); From 7734b6d01d75a63a81690bee1555fbc9a5cba1c6 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Fri, 10 Feb 2023 11:11:33 -0800 Subject: [PATCH 113/148] UPSTREAM: ANDROID: fuse-bpf: Do not change bpf program in lookups If a lookup finds an existing inode, it must not change the existing bpf program since it may be in use. Bug: 267095363 Test: fuse_test, atest CtsScopedStorageHostTest Change-Id: Icb00681fbcd51fdd4b0764906509093d98caeec4 Signed-off-by: Paul Lawrence --- fs/fuse/backing.c | 40 ++++++++++++++++++++++------------------ 1 file changed, 22 insertions(+), 18 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 9f78384b021f..8ab36055c967 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1237,28 +1237,26 @@ int fuse_handle_backing(struct fuse_entry_bpf *feb, struct inode **backing_inode int fuse_handle_bpf_prog(struct fuse_entry_bpf *feb, struct inode *parent, struct bpf_prog **bpf) { - struct fuse_inode *pi; + struct bpf_prog *new_bpf; - // Parent isn't presented, but we want to keep - // Don't touch bpf program at all in this case - if (feb->out.bpf_action == FUSE_ACTION_KEEP && !parent) { - goto out; - } - - if (*bpf) { - bpf_prog_put(*bpf); - *bpf = NULL; - } + /* Parent isn't presented, but we want to keep + * Don't touch bpf program at all in this case + */ + if (feb->out.bpf_action == FUSE_ACTION_KEEP && !parent) + return 0; switch (feb->out.bpf_action) { - case FUSE_ACTION_KEEP: - pi = get_fuse_inode(parent); - *bpf = pi->bpf; - if (*bpf) - bpf_prog_inc(*bpf); + case FUSE_ACTION_KEEP: { + struct fuse_inode *pi = get_fuse_inode(parent); + + new_bpf = pi->bpf; + if (new_bpf) + bpf_prog_inc(new_bpf); break; + } case FUSE_ACTION_REMOVE: + new_bpf = NULL; break; case FUSE_ACTION_REPLACE: { @@ -1271,7 +1269,7 @@ int fuse_handle_bpf_prog(struct fuse_entry_bpf *feb, struct inode *parent, if (IS_ERR(bpf_prog)) return PTR_ERR(bpf_prog); - *bpf = bpf_prog; + new_bpf = bpf_prog; break; } @@ -1279,7 +1277,13 @@ int fuse_handle_bpf_prog(struct fuse_entry_bpf *feb, struct inode *parent, return -EINVAL; } -out: + /* Cannot change existing program */ + if (*bpf) { + bpf_prog_put(new_bpf); + return new_bpf == *bpf ? 0 : -EINVAL; + } + + *bpf = new_bpf; return 0; } From 0db2ea97e463a003185c2305dcc7407871691a7e Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Mon, 20 Mar 2023 09:36:15 -0700 Subject: [PATCH 114/148] UPSTREAM: ANDROID: fuse-bpf: Correctly put backing files Backing files were sometimes put twice before, this fixes it so backing files sent in response to lookups are closed exactly once always Test: fuse_test pases, Android no longer throws a double close Bug: 273737310 Change-Id: Ifa75ffd846185cfabfd1f5bad504078d955c99ed Signed-off-by: Paul Lawrence Signed-off-by: Alistair Delva --- fs/fuse/backing.c | 47 ++++++++++++++++++++++++++++++++--------------- fs/fuse/dir.c | 15 ++++++++++----- 2 files changed, 42 insertions(+), 20 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 8ab36055c967..f544286dbdaa 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1222,8 +1222,6 @@ int fuse_handle_backing(struct fuse_entry_bpf *feb, struct inode **backing_inode path_put(backing_path); *backing_path = backing_file->f_path; path_get(backing_path); - - fput(backing_file); break; } @@ -1299,36 +1297,55 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct fuse_entry_bpf *feb = container_of(febo, struct fuse_entry_bpf, out); int error = -1; u64 target_nodeid = 0; + struct dentry *ret; fd = get_fuse_dentry(entry); - if (!fd) - return ERR_PTR(-EIO); + if (!fd) { + ret = ERR_PTR(-EIO); + goto out; + } + bd = fd->backing_path.dentry; - if (!bd) - return ERR_PTR(-ENOENT); + if (!bd) { + ret = ERR_PTR(-ENOENT); + goto out; + } + backing_inode = bd->d_inode; - if (!backing_inode) - return 0; + if (!backing_inode) { + ret = 0; + goto out; + } if (d_inode) target_nodeid = get_fuse_inode(d_inode)->nodeid; inode = fuse_iget_backing(dir->i_sb, target_nodeid, backing_inode); - if (IS_ERR(inode)) - return ERR_PTR(PTR_ERR(inode)); + if (IS_ERR(inode)) { + ret = ERR_PTR(PTR_ERR(inode)); + goto out; + } error = fuse_handle_bpf_prog(feb, dir, &get_fuse_inode(inode)->bpf); - if (error) - return ERR_PTR(error); + if (error) { + ret = ERR_PTR(error); + goto out; + } error = fuse_handle_backing(feb, &get_fuse_inode(inode)->backing_inode, &fd->backing_path); - if (error) - return ERR_PTR(error); + if (error) { + ret = ERR_PTR(error); + goto out; + } get_fuse_inode(inode)->nodeid = feo->nodeid; - return d_splice_alias(inode, entry); + ret = d_splice_alias(inode, entry); +out: + if (feb->backing_file) + fput(feb->backing_file); + return ret; } int fuse_revalidate_backing(struct dentry *entry, unsigned int flags) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 59c36841e86a..ce007c7836ba 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -183,8 +183,10 @@ static bool backing_data_changed(struct fuse_inode *fi, struct dentry *entry, int err; bool ret = true; - if (!entry) - return false; + if (!entry) { + ret = false; + goto put_backing_file; + } get_fuse_backing_path(entry, &new_backing_path); new_backing_inode = fi->backing_inode; @@ -207,6 +209,9 @@ put_bpf: put_inode: iput(new_backing_inode); path_put(&new_backing_path); +put_backing_file: + if (bpf_arg->backing_file) + fput(bpf_arg->backing_file); return ret; } #endif @@ -580,7 +585,7 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name backing_inode = backing_file->f_inode; *inode = fuse_iget_backing(sb, outarg->nodeid, backing_inode); if (!*inode) - goto bpf_arg_out; + goto out; err = fuse_handle_backing(&bpf_arg, &get_fuse_inode(*inode)->backing_inode, @@ -591,8 +596,6 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name err = fuse_handle_bpf_prog(&bpf_arg, NULL, &get_fuse_inode(*inode)->bpf); if (err) goto out; -bpf_arg_out: - fput(backing_file); } else #endif { @@ -624,6 +627,8 @@ out_queue_forget: out_put_forget: kfree(forget); out: + if (bpf_arg.backing_file) + fput(bpf_arg.backing_file); return err; } From 69bb0347bf128ef1632535d8f18ad2b09666b17f Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Thu, 30 Mar 2023 13:47:33 -0700 Subject: [PATCH 115/148] UPSTREAM: ANDROID: fuse-bpf: Run bpf with migration disabled To avoid a BUG_ON, we must disable migration before running any bpf program. Bug: 273620140 Test: fuse-test passes, no bug with CONFIG_DEBUG_ATOMIC_SLEEP Change-Id: I9ed69ae93fc4b922782fccde293cb02b3eae3f06 Signed-off-by: Paul Lawrence (cherry picked from commit 0411f8d9dfa88e441456b5bcf684c5ec28ef08a7) --- fs/fuse/fuse_i.h | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 71d01a96bced..128fcbd77caf 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1830,6 +1830,16 @@ void __exit fuse_bpf_cleanup(void); ssize_t fuse_bpf_simple_request(struct fuse_mount *fm, struct fuse_bpf_args *args); +static inline int fuse_bpf_run(struct bpf_prog *prog, struct fuse_bpf_args *fba) +{ + int ret; + + migrate_disable(); + ret = BPF_PROG_RUN(prog, fba); + migrate_enable(); + return ret; +} + /* * expression statement to wrap the backing filter logic * struct inode *inode: inode with bpf and backing inode @@ -1881,7 +1891,7 @@ ssize_t fuse_bpf_simple_request(struct fuse_mount *fm, struct fuse_bpf_args *arg fa.out_numargs = fa.in_numargs; \ \ ext_flags = fuse_inode->bpf ? \ - BPF_PROG_RUN(fuse_inode->bpf, &fa) : \ + fuse_bpf_run(fuse_inode->bpf, &fa) : \ FUSE_BPF_BACKING; \ if (ext_flags < 0) { \ fer = (struct fuse_err_ret) { \ @@ -1936,7 +1946,7 @@ ssize_t fuse_bpf_simple_request(struct fuse_mount *fm, struct fuse_bpf_args *arg .size = fa.out_args[i].size, \ .value = fa.out_args[i].value, \ }; \ - ext_flags = BPF_PROG_RUN(fuse_inode->bpf, &fa); \ + ext_flags = fuse_bpf_run(fuse_inode->bpf, &fa); \ if (ext_flags < 0) { \ fer = (struct fuse_err_ret) { \ ERR_PTR(ext_flags), \ From d6738a5c66b4eaba070719c08173a284f0b5918e Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Wed, 11 Nov 2020 17:22:32 +0100 Subject: [PATCH 116/148] UPSTREAM: virtiofs: clean up error handling in virtio_fs_get_tree() commit 833c5a42e28beeefa1f9bd476a63fe8050c1e8ca upstream. Avoid duplicating error cleanup. Change-Id: I42491cd5c48689662186027b4ed123aa8d0d7908 Signed-off-by: Miklos Szeredi Signed-off-by: Yang Bo Signed-off-by: Greg Kroah-Hartman --- fs/fuse/virtio_fs.c | 25 ++++++++++++------------- 1 file changed, 12 insertions(+), 13 deletions(-) diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index f8341da273f8..7398790b73df 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1457,22 +1457,14 @@ static int virtio_fs_get_tree(struct fs_context *fsc) return -EINVAL; } + err = -ENOMEM; fc = kzalloc(sizeof(struct fuse_conn), GFP_KERNEL); - if (!fc) { - mutex_lock(&virtio_fs_mutex); - virtio_fs_put(fs); - mutex_unlock(&virtio_fs_mutex); - return -ENOMEM; - } + if (!fc) + goto out_err; fm = kzalloc(sizeof(struct fuse_mount), GFP_KERNEL); - if (!fm) { - mutex_lock(&virtio_fs_mutex); - virtio_fs_put(fs); - mutex_unlock(&virtio_fs_mutex); - kfree(fc); - return -ENOMEM; - } + if (!fm) + goto out_err; fuse_conn_init(fc, fm, fsc->user_ns, &virtio_fs_fiq_ops, fs); fc->release = fuse_free_conn; @@ -1500,6 +1492,13 @@ static int virtio_fs_get_tree(struct fs_context *fsc) WARN_ON(fsc->root); fsc->root = dget(sb->s_root); return 0; + +out_err: + kfree(fc); + mutex_lock(&virtio_fs_mutex); + virtio_fs_put(fs); + mutex_unlock(&virtio_fs_mutex); + return err; } static const struct fs_context_operations virtio_fs_context_ops = { From 7a2f1138f958df69fe296f491157da86b0cc9915 Mon Sep 17 00:00:00 2001 From: Connor Kuehl Date: Thu, 18 Mar 2021 08:52:22 -0500 Subject: [PATCH 117/148] UPSTREAM: virtiofs: split requests that exceed virtqueue size commit a7f0d7aab0b4f3f0780b1f77356e2fe7202ac0cb upstream. If an incoming FUSE request can't fit on the virtqueue, the request is placed onto a workqueue so a worker can try to resubmit it later where there will (hopefully) be space for it next time. This is fine for requests that aren't larger than a virtqueue's maximum capacity. However, if a request's size exceeds the maximum capacity of the virtqueue (even if the virtqueue is empty), it will be doomed to a life of being placed on the workqueue, removed, discovered it won't fit, and placed on the workqueue yet again. Furthermore, from section 2.6.5.3.1 (Driver Requirements: Indirect Descriptors) of the virtio spec: "A driver MUST NOT create a descriptor chain longer than the Queue Size of the device." To fix this, limit the number of pages FUSE will use for an overall request. This way, each request can realistically fit on the virtqueue when it is decomposed into a scattergather list and avoid violating section 2.6.5.3.1 of the virtio spec. Change-Id: I78a1abfe6cb5484ec0f59c3aa841f119b6a5ff16 Signed-off-by: Connor Kuehl Reviewed-by: Vivek Goyal Signed-off-by: Miklos Szeredi Signed-off-by: Yang Bo Signed-off-by: Greg Kroah-Hartman --- fs/fuse/fuse_i.h | 3 +++ fs/fuse/inode.c | 3 ++- fs/fuse/virtio_fs.c | 19 +++++++++++++++++-- 3 files changed, 22 insertions(+), 3 deletions(-) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 128fcbd77caf..ec46be93c6b5 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -634,6 +634,9 @@ struct fuse_conn { /** Maxmum number of pages that can be used in a single request */ unsigned int max_pages; + /** Constrain ->max_pages to this value during feature negotiation */ + unsigned int max_pages_limit; + /** Input queue */ struct fuse_iqueue iq; diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index c0d9ca6ea074..de88c8727a52 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -857,6 +857,7 @@ void fuse_conn_init(struct fuse_conn *fc, struct fuse_mount *fm, fc->pid_ns = get_pid_ns(task_active_pid_ns(current)); fc->user_ns = get_user_ns(user_ns); fc->max_pages = FUSE_DEFAULT_MAX_PAGES_PER_REQ; + fc->max_pages_limit = FUSE_MAX_MAX_PAGES; INIT_LIST_HEAD(&fc->mounts); list_add(&fm->fc_entry, &fc->mounts); @@ -1225,7 +1226,7 @@ static void process_init_reply(struct fuse_mount *fm, struct fuse_args *args, fc->abort_err = 1; if (arg->flags & FUSE_MAX_PAGES) { fc->max_pages = - min_t(unsigned int, FUSE_MAX_MAX_PAGES, + min_t(unsigned int, fc->max_pages_limit, max_t(unsigned int, arg->max_pages, 1)); } if (IS_ENABLED(CONFIG_FUSE_DAX) && diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 7398790b73df..092a3a979c68 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -18,6 +18,12 @@ #include #include "fuse_i.h" +/* Used to help calculate the FUSE connection's max_pages limit for a request's + * size. Parts of the struct fuse_req are sliced into scattergather lists in + * addition to the pages used, so this can help account for that overhead. + */ +#define FUSE_HEADER_OVERHEAD 4 + /* List of virtio-fs device instances and a lock for the list. Also provides * mutual exclusion in device removal and mounting path */ @@ -1440,9 +1446,10 @@ static int virtio_fs_get_tree(struct fs_context *fsc) { struct virtio_fs *fs; struct super_block *sb; - struct fuse_conn *fc; + struct fuse_conn *fc = NULL; struct fuse_mount *fm; - int err; + unsigned int virtqueue_size; + int err = -EIO; if (!fsc->source) return invalf(fsc, "No source specified"); @@ -1457,6 +1464,10 @@ static int virtio_fs_get_tree(struct fs_context *fsc) return -EINVAL; } + virtqueue_size = virtqueue_get_vring_size(fs->vqs[VQ_REQUEST].vq); + if (WARN_ON(virtqueue_size <= FUSE_HEADER_OVERHEAD)) + goto out_err; + err = -ENOMEM; fc = kzalloc(sizeof(struct fuse_conn), GFP_KERNEL); if (!fc) @@ -1471,6 +1482,10 @@ static int virtio_fs_get_tree(struct fs_context *fsc) fc->delete_stale = true; fc->auto_submounts = true; + /* Tell FUSE to split requests that exceed the virtqueue's size */ + fc->max_pages_limit = min_t(unsigned int, fc->max_pages_limit, + virtqueue_size - FUSE_HEADER_OVERHEAD); + fsc->s_fs_info = fm; sb = sget_fc(fsc, virtio_fs_test_super, virtio_fs_set_super); fuse_mount_put(fm); From 9085f3d5d51b348ef57217b100bdb62e722fda04 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Thu, 21 Oct 2021 10:01:38 +0200 Subject: [PATCH 118/148] UPSTREAM: fuse: check s_root when destroying sb commit d534d31d6a45d71de61db22090b4820afb68fddc upstream. Checking "fm" works because currently sb->s_fs_info is cleared on error paths; however, sb->s_root is what generic_shutdown_super() checks to determine whether the sb was fully initialized or not. This change will allow cleanup of sb setup error paths. Change-Id: Icc78860e8a64e279fb2717b1cc08d1b83160d733 Signed-off-by: Miklos Szeredi Signed-off-by: Yang Bo Signed-off-by: Greg Kroah-Hartman --- fs/fuse/inode.c | 2 +- fs/fuse/virtio_fs.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index de88c8727a52..65f35eeb858b 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1768,7 +1768,7 @@ static void fuse_kill_sb_blk(struct super_block *sb) struct fuse_mount *fm = get_fuse_mount_super(sb); bool last; - if (fm) { + if (sb->s_root) { last = fuse_mount_remove(fm); if (last) fuse_conn_destroy(fm); diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 092a3a979c68..cc43a773b32e 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -1413,7 +1413,7 @@ static void virtio_kill_sb(struct super_block *sb) bool last; /* If mount failed, we can still be called without any fc */ - if (fm) { + if (sb->s_root) { last = fuse_mount_remove(fm); if (last) virtio_fs_conn_destroy(fm); From f6598346ec3928620b4f94bcd91dfe0d3fe6c336 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Fri, 22 Oct 2021 17:03:03 +0200 Subject: [PATCH 119/148] UPSTREAM: fuse: fix attr version comparison in fuse_read_update_size() commit 484ce65715b06aead8c4901f01ca32c5a240bc71 upstream. A READ request returning a short count is taken as indication of EOF, and the cached file size is modified accordingly. Fix the attribute version checking to allow for changes to fc->attr_version on other inodes. Change-Id: I03143ae182ed264aea806a02c5ff4edf0ae28239 Signed-off-by: Miklos Szeredi Signed-off-by: Yang Bo Signed-off-by: Greg Kroah-Hartman --- fs/fuse/file.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 04fd62f4562f..fea39e0b037c 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -836,7 +836,7 @@ static void fuse_read_update_size(struct inode *inode, loff_t size, struct fuse_inode *fi = get_fuse_inode(inode); spin_lock(&fi->lock); - if (attr_ver == fi->attr_version && size < inode->i_size && + if (attr_ver >= fi->attr_version && size < inode->i_size && !test_bit(FUSE_I_SIZE_UNSTABLE, &fi->state)) { fi->attr_version = atomic64_inc_return(&fc->attr_version); i_size_write(inode, size); From 4056baf3cd11e2053cbb3b7d3dfc80021b65be0f Mon Sep 17 00:00:00 2001 From: Jiachen Zhang Date: Wed, 28 Sep 2022 20:19:34 +0800 Subject: [PATCH 120/148] UPSTREAM: fuse: always revalidate rename target dentry commit ccc031e26afe60d2a5a3d93dabd9c978210825fb upstream. The previous commit df8629af2934 ("fuse: always revalidate if exclusive create") ensures that the dentries are revalidated on O_EXCL creates. This commit complements it by also performing revalidation for rename target dentries. Otherwise, a rename target file that only exists in kernel dentry cache but not in the filesystem will result in EEXIST if RENAME_NOREPLACE flag is used. Change-Id: Ie26ba53d008737ff709081dda0ef3e07ade449a5 Signed-off-by: Jiachen Zhang Signed-off-by: Zhang Tianci Signed-off-by: Miklos Szeredi Signed-off-by: Yang Bo Signed-off-by: Greg Kroah-Hartman --- fs/fuse/dir.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index ce007c7836ba..4a336096b153 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -249,7 +249,7 @@ static int fuse_dentry_revalidate(struct dentry *entry, unsigned int flags) } #endif if (time_before64(fuse_dentry_time(entry), get_jiffies_64()) || - (flags & (LOOKUP_EXCL | LOOKUP_REVAL))) { + (flags & (LOOKUP_EXCL | LOOKUP_REVAL | LOOKUP_RENAME_TARGET))) { struct fuse_entry_out outarg; struct fuse_entry_bpf bpf_arg; FUSE_ARGS(args); From 4000339dc01d99d959b5fcbeaff9715c9c53100d Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Fri, 22 Apr 2022 15:48:53 +0200 Subject: [PATCH 121/148] UPSTREAM: fuse: fix deadlock between atomic O_TRUNC and page invalidation commit 2fdbb8dd01556e1501132b5ad3826e8f71e24a8b upstream. fuse_finish_open() will be called with FUSE_NOWRITE set in case of atomic O_TRUNC open(), so commit 76224355db75 ("fuse: truncate pagecache on atomic_o_trunc") replaced invalidate_inode_pages2() by truncate_pagecache() in such a case to avoid the A-A deadlock. However, we found another A-B-B-A deadlock related to the case above, which will cause the xfstests generic/464 testcase hung in our virtio-fs test environment. For example, consider two processes concurrently open one same file, one with O_TRUNC and another without O_TRUNC. The deadlock case is described below, if open(O_TRUNC) is already set_nowrite(acquired A), and is trying to lock a page (acquiring B), open() could have held the page lock (acquired B), and waiting on the page writeback (acquiring A). This would lead to deadlocks. open(O_TRUNC) ---------------------------------------------------------------- fuse_open_common inode_lock [C acquire] fuse_set_nowrite [A acquire] fuse_finish_open truncate_pagecache lock_page [B acquire] truncate_inode_page unlock_page [B release] fuse_release_nowrite [A release] inode_unlock [C release] ---------------------------------------------------------------- open() ---------------------------------------------------------------- fuse_open_common fuse_finish_open invalidate_inode_pages2 lock_page [B acquire] fuse_launder_page fuse_wait_on_page_writeback [A acquire & release] unlock_page [B release] ---------------------------------------------------------------- Besides this case, all calls of invalidate_inode_pages2() and invalidate_inode_pages2_range() in fuse code also can deadlock with open(O_TRUNC). Fix by moving the truncate_pagecache() call outside the nowrite protected region. The nowrite protection is only for delayed writeback (writeback_cache) case, where inode lock does not protect against truncation racing with writes on the server. Write syscalls racing with page cache truncation still get the inode lock protection. This patch also changes the order of filemap_invalidate_lock() vs. fuse_set_nowrite() in fuse_open_common(). This new order matches the order found in fuse_file_fallocate() and fuse_do_setattr(). Reported-by: Jiachen Zhang Tested-by: Jiachen Zhang Fixes: e4648309b85a ("fuse: truncate pending writes on O_TRUNC") Cc: Change-Id: I7e01c787503114b4cd671ef7283fb0a1143887a1 Signed-off-by: Miklos Szeredi Signed-off-by: Yang Bo Signed-off-by: Greg Kroah-Hartman --- fs/fuse/dir.c | 5 +++++ fs/fuse/file.c | 29 +++++++++++++++++------------ 2 files changed, 22 insertions(+), 12 deletions(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 4a336096b153..b1ee3a6b5b83 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -713,6 +713,7 @@ static int fuse_create_open(struct inode *dir, struct dentry *entry, struct fuse_entry_out outentry; struct fuse_inode *fi; struct fuse_file *ff; + bool trunc = flags & O_TRUNC; /* Userspace expects S_IFREG in create mode */ BUG_ON((mode & S_IFMT) != S_IFREG); @@ -795,6 +796,10 @@ static int fuse_create_open(struct inode *dir, struct dentry *entry, } else { file->private_data = ff; fuse_finish_open(inode, file); + if (fm->fc->atomic_o_trunc && trunc) + truncate_pagecache(inode, 0); + else if (!(ff->open_flags & FOPEN_KEEP_CACHE)) + invalidate_inode_pages2(inode->i_mapping); } return err; diff --git a/fs/fuse/file.c b/fs/fuse/file.c index fea39e0b037c..bb76ba48ef32 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -211,14 +211,10 @@ void fuse_finish_open(struct inode *inode, struct file *file) fi->attr_version = atomic64_inc_return(&fc->attr_version); i_size_write(inode, 0); spin_unlock(&fi->lock); - truncate_pagecache(inode, 0); fuse_invalidate_attr(inode); if (fc->writeback_cache) file_update_time(file); - } else if (!(ff->open_flags & FOPEN_KEEP_CACHE)) { - invalidate_inode_pages2(inode->i_mapping); } - if ((file->f_mode & FMODE_WRITE) && fc->writeback_cache) fuse_link_write_file(file); } @@ -255,30 +251,39 @@ int fuse_open_common(struct inode *inode, struct file *file, bool isdir) } #endif - if (is_wb_truncate || dax_truncate) { + if (is_wb_truncate || dax_truncate) inode_lock(inode); - fuse_set_nowrite(inode); - } if (dax_truncate) { down_write(&get_fuse_inode(inode)->i_mmap_sem); err = fuse_dax_break_layouts(inode, 0, 0); if (err) - goto out; + goto out_inode_unlock; } + if (is_wb_truncate || dax_truncate) + fuse_set_nowrite(inode); + err = fuse_do_open(fm, get_node_id(inode), file, isdir); if (!err) fuse_finish_open(inode, file); -out: + if (is_wb_truncate || dax_truncate) + fuse_release_nowrite(inode); + if (!err) { + struct fuse_file *ff = file->private_data; + + if (fc->atomic_o_trunc && (file->f_flags & O_TRUNC)) + truncate_pagecache(inode, 0); + else if (!(ff->open_flags & FOPEN_KEEP_CACHE)) + invalidate_inode_pages2(inode->i_mapping); + } if (dax_truncate) up_write(&get_fuse_inode(inode)->i_mmap_sem); - if (is_wb_truncate | dax_truncate) { - fuse_release_nowrite(inode); +out_inode_unlock: + if (is_wb_truncate || dax_truncate) inode_unlock(inode); - } return err; } From b0fc361eb2de6ba3f61120611f27bb4bf3c566e1 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Wed, 26 Apr 2023 14:11:58 -0700 Subject: [PATCH 122/148] UPSTREAM: ANDROID: fuse-bpf: Simplify and fix setting bpf program Fix case when an existing bpf prog is being removed Tidy up code Bug: 279363668 Test: Boots, can copy file to /sdcardfs/Android/data, fuse_test passes Change-Id: If0e682f43cbeb62764a7a2be543b90cb974b0aa0 Signed-off-by: Paul Lawrence --- fs/fuse/backing.c | 36 ++++++++++++++++++------------------ 1 file changed, 18 insertions(+), 18 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index f544286dbdaa..e292c557a850 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1235,39 +1235,36 @@ int fuse_handle_backing(struct fuse_entry_bpf *feb, struct inode **backing_inode int fuse_handle_bpf_prog(struct fuse_entry_bpf *feb, struct inode *parent, struct bpf_prog **bpf) { - struct bpf_prog *new_bpf; - - /* Parent isn't presented, but we want to keep - * Don't touch bpf program at all in this case - */ - if (feb->out.bpf_action == FUSE_ACTION_KEEP && !parent) - return 0; + struct bpf_prog *new_bpf = NULL; switch (feb->out.bpf_action) { case FUSE_ACTION_KEEP: { - struct fuse_inode *pi = get_fuse_inode(parent); + /* Parent isn't presented, but we want to keep + * Don't touch bpf program at all in this case + */ + if (!parent) + return 0; - new_bpf = pi->bpf; + new_bpf = get_fuse_inode(parent)->bpf; if (new_bpf) bpf_prog_inc(new_bpf); break; } case FUSE_ACTION_REMOVE: - new_bpf = NULL; break; case FUSE_ACTION_REPLACE: { struct file *bpf_file = feb->bpf_file; - struct bpf_prog *bpf_prog = ERR_PTR(-EINVAL); - if (bpf_file && !IS_ERR(bpf_file)) - bpf_prog = fuse_get_bpf_prog(bpf_file); + if (!bpf_file) + return -EINVAL; + if (IS_ERR(bpf_file)) + return PTR_ERR(bpf_file); - if (IS_ERR(bpf_prog)) - return PTR_ERR(bpf_prog); - - new_bpf = bpf_prog; + new_bpf = fuse_get_bpf_prog(bpf_file); + if (IS_ERR(new_bpf)) + return PTR_ERR(new_bpf); break; } @@ -1276,11 +1273,14 @@ int fuse_handle_bpf_prog(struct fuse_entry_bpf *feb, struct inode *parent, } /* Cannot change existing program */ - if (*bpf) { + if (*bpf && new_bpf) { bpf_prog_put(new_bpf); return new_bpf == *bpf ? 0 : -EINVAL; } + if (*bpf) + bpf_prog_put(*bpf); + *bpf = new_bpf; return 0; } From 8f058f1cedc6745cf147b22442f3ded14a0e45c6 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Wed, 7 Jun 2023 10:19:01 -0700 Subject: [PATCH 123/148] UPSTREAM: ANDROID: fuse-bpf: Ensure bpf field can never be nulled By putting and nulling fuse_inode's bpf field in fuse_evict_inode, we left a race condition - this inode can still be active. Do not put the bpf program until we are doing the final free in fuse_free_inode. This was the root cause of the reported bug. The backing inode cannot be put in fuse_free_inode, since put_inode can sleep and this is called from an RCU handler. But the backing inode cannot be freed until an RCU interval, so move the put_inode to the same location as in overlayfs, which is destroy_inode. Remove a path in fuse_handle_bpf_prog whereby bpf can be nulled out. When we want to be able to null/change the bpf_prog in the future, we will have to use a mutex or maybe RCU to protect existing users. But until this time, ban this path. Bug: 284450048 Test: fuse_test passes, Pixel 6 passes basic tests Change-Id: Ie6844242f279a5b202eb021eac5a2dd3d08bf09d Signed-off-by: Paul Lawrence --- fs/fuse/backing.c | 8 +++----- fs/fuse/inode.c | 22 ++++++++++++++++------ 2 files changed, 19 insertions(+), 11 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index e292c557a850..19e690aaffb1 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1273,14 +1273,12 @@ int fuse_handle_bpf_prog(struct fuse_entry_bpf *feb, struct inode *parent, } /* Cannot change existing program */ - if (*bpf && new_bpf) { - bpf_prog_put(new_bpf); + if (*bpf) { + if (new_bpf) + bpf_prog_put(new_bpf); return new_bpf == *bpf ? 0 : -EINVAL; } - if (*bpf) - bpf_prog_put(*bpf); - *bpf = new_bpf; return 0; } diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 65f35eeb858b..98b2371a8712 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -113,6 +113,10 @@ static void fuse_free_inode(struct inode *inode) kfree(fi->forget); #ifdef CONFIG_FUSE_DAX kfree(fi->dax); +#endif +#ifdef CONFIG_FUSE_BPF + if (fi->bpf) + bpf_prog_put(fi->bpf); #endif kmem_cache_free(fuse_inode_cachep, fi); } @@ -121,12 +125,6 @@ static void fuse_evict_inode(struct inode *inode) { struct fuse_inode *fi = get_fuse_inode(inode); -#ifdef CONFIG_FUSE_BPF - iput(fi->backing_inode); - if (fi->bpf) - bpf_prog_put(fi->bpf); - fi->bpf = NULL; -#endif truncate_inode_pages_final(&inode->i_data); clear_inode(inode); if (inode->i_sb->s_flags & SB_ACTIVE) { @@ -146,6 +144,15 @@ static void fuse_evict_inode(struct inode *inode) } } +#ifdef CONFIG_FUSE_BPF +static void fuse_destroy_inode(struct inode *inode) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + + iput(fi->backing_inode); +} +#endif + static int fuse_reconfigure(struct fs_context *fc) { struct super_block *sb = fc->root->d_sb; @@ -1088,6 +1095,9 @@ static const struct export_operations fuse_export_operations = { static const struct super_operations fuse_super_operations = { .alloc_inode = fuse_alloc_inode, +#ifdef CONFIG_FUSE_BPF + .destroy_inode = fuse_destroy_inode, +#endif .free_inode = fuse_free_inode, .evict_inode = fuse_evict_inode, .write_inode = fuse_write_inode, From dfc8380342c0d60f4e0adf77d0cfdddd6542dc03 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Wed, 14 Jun 2023 11:55:53 -0700 Subject: [PATCH 124/148] UPSTREAM: ANDROID: fuse-bpf: Move FUSE_RELEASE to correct place The existing fuse-bpf freeing logic would free the fuse_file struct immediately. However, this would break readahead. Move freeing logic to the same place as done in classic fuse. Bug: 286287652 Test: fuse_test passes, android boots, cts tests run Change-Id: If13519f0e956a8da0dc98e7ac4aed2036070e969 Signed-off-by: Paul Lawrence --- fs/fuse/backing.c | 40 +++++-------------------- fs/fuse/dir.c | 11 ------- fs/fuse/file.c | 74 +++++++++++++++++++++++++---------------------- fs/fuse/fuse_i.h | 9 ++---- 4 files changed, 50 insertions(+), 84 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 19e690aaffb1..aad46ab272f7 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -348,44 +348,19 @@ void *fuse_create_open_finalize( } int fuse_release_initialize(struct fuse_bpf_args *fa, struct fuse_release_in *fri, - struct inode *inode, struct file *file) + struct inode *inode, struct fuse_file *ff) { - struct fuse_file *fuse_file = file->private_data; - /* Always put backing file whatever bpf/userspace says */ - fput(fuse_file->backing_file); + fput(ff->backing_file); *fri = (struct fuse_release_in) { - .fh = ((struct fuse_file *)(file->private_data))->fh, + .fh = ff->fh, }; *fa = (struct fuse_bpf_args) { .nodeid = get_fuse_inode(inode)->nodeid, - .opcode = FUSE_RELEASE, - .in_numargs = 1, - .in_args[0].size = sizeof(*fri), - .in_args[0].value = fri, - }; - - return 0; -} - -int fuse_releasedir_initialize(struct fuse_bpf_args *fa, - struct fuse_release_in *fri, - struct inode *inode, struct file *file) -{ - struct fuse_file *fuse_file = file->private_data; - - /* Always put backing file whatever bpf/userspace says */ - fput(fuse_file->backing_file); - - *fri = (struct fuse_release_in) { - .fh = ((struct fuse_file *)(file->private_data))->fh, - }; - - *fa = (struct fuse_bpf_args) { - .nodeid = get_fuse_inode(inode)->nodeid, - .opcode = FUSE_RELEASEDIR, + .opcode = S_ISDIR(inode->i_mode) ? FUSE_RELEASEDIR + : FUSE_RELEASE, .in_numargs = 1, .in_args[0].size = sizeof(*fri), .in_args[0].value = fri, @@ -395,15 +370,14 @@ int fuse_releasedir_initialize(struct fuse_bpf_args *fa, } int fuse_release_backing(struct fuse_bpf_args *fa, - struct inode *inode, struct file *file) + struct inode *inode, struct fuse_file *ff) { return 0; } void *fuse_release_finalize(struct fuse_bpf_args *fa, - struct inode *inode, struct file *file) + struct inode *inode, struct fuse_file *ff) { - fuse_file_free(file->private_data); return NULL; } diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index b1ee3a6b5b83..e6fcb5c55f9e 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -1726,17 +1726,6 @@ static int fuse_dir_open(struct inode *inode, struct file *file) static int fuse_dir_release(struct inode *inode, struct file *file) { -#ifdef CONFIG_FUSE_BPF - struct fuse_err_ret fer; - - fer = fuse_bpf_backing(inode, struct fuse_release_in, - fuse_releasedir_initialize, fuse_release_backing, - fuse_release_finalize, - inode, file); - if (fer.ret) - return PTR_ERR(fer.result); -#endif - fuse_release_common(file, true); return 0; } diff --git a/fs/fuse/file.c b/fs/fuse/file.c index bb76ba48ef32..26dd29c54cfb 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -110,25 +110,39 @@ static void fuse_release_end(struct fuse_mount *fm, struct fuse_args *args, kfree(ra); } -static void fuse_file_put(struct fuse_file *ff, bool sync, bool isdir) +static void fuse_file_put(struct inode *inode, struct fuse_file *ff, + bool sync, bool isdir) { - if (refcount_dec_and_test(&ff->count)) { - struct fuse_args *args = &ff->release_args->args; + struct fuse_args *args = &ff->release_args->args; +#ifdef CONFIG_FUSE_BPF + struct fuse_err_ret fer; +#endif - if (isdir ? ff->fm->fc->no_opendir : ff->fm->fc->no_open) { - /* Do nothing when client does not implement 'open' */ - fuse_release_end(ff->fm, args, 0); - } else if (sync) { - fuse_simple_request(ff->fm, args); - fuse_release_end(ff->fm, args, 0); - } else { - args->end = fuse_release_end; - if (fuse_simple_background(ff->fm, args, - GFP_KERNEL | __GFP_NOFAIL)) - fuse_release_end(ff->fm, args, -ENOTCONN); - } - kfree(ff); + if (!refcount_dec_and_test(&ff->count)) + return; + +#ifdef CONFIG_FUSE_BPF + fer = fuse_bpf_backing(inode, struct fuse_release_in, + fuse_release_initialize, fuse_release_backing, + fuse_release_finalize, + inode, ff); + if (fer.ret) { + fuse_release_end(ff->fm, args, 0); + } else +#endif + if (isdir ? ff->fm->fc->no_opendir : ff->fm->fc->no_open) { + /* Do nothing when client does not implement 'open' */ + fuse_release_end(ff->fm, args, 0); + } else if (sync) { + fuse_simple_request(ff->fm, args); + fuse_release_end(ff->fm, args, 0); + } else { + args->end = fuse_release_end; + if (fuse_simple_background(ff->fm, args, + GFP_KERNEL | __GFP_NOFAIL)) + fuse_release_end(ff->fm, args, -ENOTCONN); } + kfree(ff); } int fuse_do_open(struct fuse_mount *fm, u64 nodeid, struct file *file, @@ -346,7 +360,7 @@ void fuse_release_common(struct file *file, bool isdir) * synchronous RELEASE is allowed (and desirable) in this case * because the server can be trusted not to screw up. */ - fuse_file_put(ff, ff->fm->fc->destroy, isdir); + fuse_file_put(ra->inode, ff, ff->fm->fc->destroy, isdir); } static int fuse_open(struct inode *inode, struct file *file) @@ -358,17 +372,6 @@ static int fuse_release(struct inode *inode, struct file *file) { struct fuse_conn *fc = get_fuse_conn(inode); -#ifdef CONFIG_FUSE_BPF - struct fuse_err_ret fer; - - fer = fuse_bpf_backing(inode, struct fuse_release_in, - fuse_release_initialize, fuse_release_backing, - fuse_release_finalize, - inode, file); - if (fer.ret) - return PTR_ERR(fer.result); -#endif - /* see fuse_vma_close() for !writeback_cache case */ if (fc->writeback_cache) write_inode_now(inode, 1); @@ -387,7 +390,7 @@ void fuse_sync_release(struct fuse_inode *fi, struct fuse_file *ff, int flags) * iput(NULL) is a no-op and since the refcount is 1 and everything's * synchronous, we are fine with not doing igrab() here" */ - fuse_file_put(ff, true, false); + fuse_file_put(&fi->inode, ff, true, false); } EXPORT_SYMBOL_GPL(fuse_sync_release); @@ -969,8 +972,11 @@ static void fuse_readpages_end(struct fuse_mount *fm, struct fuse_args *args, unlock_page(page); put_page(page); } - if (ia->ff) - fuse_file_put(ia->ff, false, false); + if (ia->ff) { + WARN_ON(!mapping); + fuse_file_put(mapping ? mapping->host : NULL, ia->ff, + false, false); + } fuse_io_free(ia); } @@ -1753,7 +1759,7 @@ static void fuse_writepage_free(struct fuse_writepage_args *wpa) __free_page(ap->pages[i]); if (wpa->ia.ff) - fuse_file_put(wpa->ia.ff, false, false); + fuse_file_put(wpa->inode, wpa->ia.ff, false, false); kfree(ap->pages); kfree(wpa); @@ -2007,7 +2013,7 @@ int fuse_write_inode(struct inode *inode, struct writeback_control *wbc) ff = __fuse_write_file_get(fc, fi); err = fuse_flush_times(inode, ff); if (ff) - fuse_file_put(ff, false, false); + fuse_file_put(inode, ff, false, false); return err; } @@ -2379,7 +2385,7 @@ static int fuse_writepages(struct address_space *mapping, fuse_writepages_send(&data); } if (data.ff) - fuse_file_put(data.ff, false, false); + fuse_file_put(inode, data.ff, false, false); kfree(data.orig_pages); out: diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index ec46be93c6b5..7e4a24b2d5f7 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1452,14 +1452,11 @@ void *fuse_link_finalize(struct fuse_bpf_args *fa, struct dentry *entry, struct inode *dir, struct dentry *newent); int fuse_release_initialize(struct fuse_bpf_args *fa, struct fuse_release_in *fri, - struct inode *inode, struct file *file); -int fuse_releasedir_initialize(struct fuse_bpf_args *fa, - struct fuse_release_in *fri, - struct inode *inode, struct file *file); + struct inode *inode, struct fuse_file *ff); int fuse_release_backing(struct fuse_bpf_args *fa, - struct inode *inode, struct file *file); + struct inode *inode, struct fuse_file *ff); void *fuse_release_finalize(struct fuse_bpf_args *fa, - struct inode *inode, struct file *file); + struct inode *inode, struct fuse_file *ff); int fuse_flush_initialize(struct fuse_bpf_args *fa, struct fuse_flush_in *ffi, struct file *file, fl_owner_t id); From 293c259e5e1719ab20eb7d9976f1daf16a7a779a Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Fri, 7 Jul 2023 16:07:09 -0700 Subject: [PATCH 125/148] UPSTREAM: ANDROID: fuse-bpf: Add partial flock support This adds passthrough support for flock on fuse-bpf files. It does not give any control via a bpf filter. The flock will act as though it was taken on the lower file. Bug: 289882899 Test: fuse_test -t32 (flock_test) Change-Id: Iba0b9630766cedbd3195532c5e929891593cfe30 Signed-off-by: Daniel Rosenberg --- fs/fuse/backing.c | 14 ++++++++++++++ fs/fuse/file.c | 8 +++++++- fs/fuse/fuse_i.h | 1 + 3 files changed, 22 insertions(+), 1 deletion(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index aad46ab272f7..8db2a0c87f75 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1016,6 +1016,20 @@ void *fuse_file_write_iter_finalize(struct fuse_bpf_args *fa, return ERR_PTR(fwio->ret); } +int fuse_file_flock_backing(struct file *file, int cmd, struct file_lock *fl) +{ + struct fuse_file *ff = file->private_data; + struct file *backing_file = ff->backing_file; + int error; + + fl->fl_file = backing_file; + if (backing_file->f_op->flock) + error = backing_file->f_op->flock(backing_file, cmd, fl); + else + error = locks_lock_file_wait(backing_file, fl); + return error; +} + ssize_t fuse_backing_mmap(struct file *file, struct vm_area_struct *vma) { int ret; diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 26dd29c54cfb..781f5fe3ab0e 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -2697,12 +2697,18 @@ static int fuse_file_flock(struct file *file, int cmd, struct file_lock *fl) { struct inode *inode = file_inode(file); struct fuse_conn *fc = get_fuse_conn(inode); + struct fuse_file *ff = file->private_data; int err; +#ifdef CONFIG_FUSE_BPF + /* TODO - this is simply passthrough, not a proper BPF filter */ + if (ff->backing_file) + return fuse_file_flock_backing(file, cmd, fl); +#endif + if (fc->no_flock) { err = locks_lock_file_wait(file, fl); } else { - struct fuse_file *ff = file->private_data; /* emulate flock with POSIX locks */ ff->flock = true; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 7e4a24b2d5f7..05e43ad2215e 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1580,6 +1580,7 @@ int fuse_file_write_iter_backing(struct fuse_bpf_args *fa, void *fuse_file_write_iter_finalize(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *from); +int fuse_file_flock_backing(struct file *file, int cmd, struct file_lock *fl); ssize_t fuse_backing_mmap(struct file *file, struct vm_area_struct *vma); int fuse_file_fallocate_initialize(struct fuse_bpf_args *fa, From 5c6d3f261827a67df84a7c55a0d204b882dcb6a8 Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Thu, 4 May 2023 15:43:42 -0700 Subject: [PATCH 126/148] UPSTREAM: ANDROID: fuse-bpf: Add partial ioctl support This adds passthrough only support for ioctls with fuse-bpf. compat_ioctls will return -ENOTTY. Bug: 279519292 Test: F2fsMiscTest#testAtomicWrite Change-Id: Ia3052e465d87dc1d15ae13955fba8a7f93bc387b Signed-off-by: Daniel Rosenberg --- fs/fuse/backing.c | 13 +++++++++++++ fs/fuse/file.c | 9 +++++++++ fs/fuse/fuse_i.h | 2 ++ 3 files changed, 24 insertions(+) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 8db2a0c87f75..eac91793e7a5 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1016,6 +1016,19 @@ void *fuse_file_write_iter_finalize(struct fuse_bpf_args *fa, return ERR_PTR(fwio->ret); } +long fuse_backing_ioctl(struct file *file, unsigned int command, unsigned long arg, int flags) +{ + struct fuse_file *ff = file->private_data; + long ret; + + if (flags & FUSE_IOCTL_COMPAT) + ret = -ENOTTY; + else + ret = vfs_ioctl(ff->backing_file, command, arg); + + return ret; +} + int fuse_file_flock_backing(struct file *file, int cmd, struct file_lock *fl) { struct fuse_file *ff = file->private_data; diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 781f5fe3ab0e..ac3b4e9f5668 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -3176,6 +3176,15 @@ long fuse_ioctl_common(struct file *file, unsigned int cmd, if (fuse_is_bad(inode)) return -EIO; +#ifdef CONFIG_FUSE_BPF + { + struct fuse_file *ff = file->private_data; + + /* TODO - this is simply passthrough, not a proper BPF filter */ + if (ff->backing_file) + return fuse_backing_ioctl(file, cmd, arg, flags); + } +#endif return fuse_do_ioctl(file, cmd, arg, flags); } diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 05e43ad2215e..f758c09f77fe 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1580,6 +1580,8 @@ int fuse_file_write_iter_backing(struct fuse_bpf_args *fa, void *fuse_file_write_iter_finalize(struct fuse_bpf_args *fa, struct kiocb *iocb, struct iov_iter *from); +long fuse_backing_ioctl(struct file *file, unsigned int command, unsigned long arg, int flags); + int fuse_file_flock_backing(struct file *file, int cmd, struct file_lock *fl); ssize_t fuse_backing_mmap(struct file *file, struct vm_area_struct *vma); From b0e557bef5d7c251c27e9ab01b97eb52225a10c7 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Wed, 26 Jul 2023 15:03:41 -0700 Subject: [PATCH 127/148] UPSTREAM: ANDROID: fuse-bpf: Check inode not null fuse_iget_backing returns an inode or null, not a ERR_PTR. So check it's not NULL Also make sure we put the inode if d_splice_alias fails Bug: 293349757 Test: fuse_test runs Signed_off_by: Paul Lawrence Change-Id: I1eadad32f80bab6730e461412b4b7ab4d6c56bf2 --- fs/fuse/backing.c | 17 ++++++++++------- fs/fuse/dir.c | 26 ++++++++++++-------------- 2 files changed, 22 insertions(+), 21 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index eac91793e7a5..4372443ff1b4 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -305,8 +305,8 @@ int fuse_create_open_backing( inode = fuse_iget_backing(dir->i_sb, target_nodeid, get_fuse_dentry(entry)->backing_path.dentry->d_inode); - if (IS_ERR(inode)) { - err = PTR_ERR(inode); + if (!inode) { + err = -EIO; goto out; } @@ -322,10 +322,12 @@ int fuse_create_open_backing( goto out; } + inode = NULL; entry = newent ? newent : entry; err = finish_open(file, entry, fuse_open_file_backing); out: + iput(inode); dput(backing_dentry); return err; } @@ -1289,7 +1291,7 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, { struct fuse_dentry *fd; struct dentry *bd; - struct inode *inode, *backing_inode; + struct inode *inode = NULL, *backing_inode; struct inode *d_inode = entry->d_inode; struct fuse_entry_out *feo = fa->out_args[0].value; struct fuse_entry_bpf_out *febo = fa->out_args[1].value; @@ -1320,9 +1322,8 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, target_nodeid = get_fuse_inode(d_inode)->nodeid; inode = fuse_iget_backing(dir->i_sb, target_nodeid, backing_inode); - - if (IS_ERR(inode)) { - ret = ERR_PTR(PTR_ERR(inode)); + if (!inode) { + ret = ERR_PTR(-EIO); goto out; } @@ -1339,9 +1340,11 @@ struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, } get_fuse_inode(inode)->nodeid = feo->nodeid; - ret = d_splice_alias(inode, entry); + if (!IS_ERR(ret)) + inode = NULL; out: + iput(inode); if (feb->backing_file) fput(feb->backing_file); return ret; diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index e6fcb5c55f9e..07cc413638c4 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -551,7 +551,6 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name if (name->len > FUSE_NAME_MAX) goto out; - forget = fuse_alloc_forget(); err = -ENOMEM; if (!forget) @@ -570,32 +569,34 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name err = -ENOENT; if (!entry) - goto out_queue_forget; + goto out_put_forget; err = -EINVAL; backing_file = bpf_arg.backing_file; if (!backing_file) - goto out_queue_forget; + goto out_put_forget; if (IS_ERR(backing_file)) { err = PTR_ERR(backing_file); - goto out_queue_forget; + goto out_put_forget; } backing_inode = backing_file->f_inode; *inode = fuse_iget_backing(sb, outarg->nodeid, backing_inode); if (!*inode) - goto out; + goto out_put_forget; err = fuse_handle_backing(&bpf_arg, &get_fuse_inode(*inode)->backing_inode, &get_fuse_dentry(entry)->backing_path); - if (err) - goto out; - - err = fuse_handle_bpf_prog(&bpf_arg, NULL, &get_fuse_inode(*inode)->bpf); - if (err) - goto out; + if (!err) + err = fuse_handle_bpf_prog(&bpf_arg, NULL, + &get_fuse_inode(*inode)->bpf); + if (err) { + iput(*inode); + *inode = NULL; + goto out_put_forget; + } } else #endif { @@ -615,9 +616,6 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name } err = -ENOMEM; -#ifdef CONFIG_FUSE_BPF -out_queue_forget: -#endif if (!*inode && outarg->nodeid) { fuse_queue_forget(fm->fc, forget, outarg->nodeid, 1); goto out; From 35003c2f72dbb67010757f70c740ede3d62b76cb Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Tue, 25 Jul 2023 11:18:38 -0700 Subject: [PATCH 128/148] UPSTREAM: ANDROID: fuse-bpf: Add bpf to negative fuse_dentry Store the results of a negative lookup in the fuse_dentry so later opcodes can use them to create files Bug: 291705489 Test: fuse_test passes Signed-off-by: Paul Lawrence Change-Id: I725e714a1d6ce43f24431d07c24e96349ef1a55c --- fs/fuse/backing.c | 64 +++++++++++++++++++++++------------------------ fs/fuse/dir.c | 5 ++++ fs/fuse/fuse_i.h | 6 +++++ 3 files changed, 43 insertions(+), 32 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 4372443ff1b4..770b6f07f443 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1289,60 +1289,60 @@ int fuse_handle_bpf_prog(struct fuse_entry_bpf *feb, struct inode *parent, struct dentry *fuse_lookup_finalize(struct fuse_bpf_args *fa, struct inode *dir, struct dentry *entry, unsigned int flags) { - struct fuse_dentry *fd; - struct dentry *bd; + struct fuse_dentry *fuse_entry; + struct dentry *backing_entry; struct inode *inode = NULL, *backing_inode; - struct inode *d_inode = entry->d_inode; + struct inode *entry_inode = entry->d_inode; struct fuse_entry_out *feo = fa->out_args[0].value; struct fuse_entry_bpf_out *febo = fa->out_args[1].value; - struct fuse_entry_bpf *feb = container_of(febo, struct fuse_entry_bpf, out); + struct fuse_entry_bpf *feb = container_of(febo, struct fuse_entry_bpf, + out); int error = -1; u64 target_nodeid = 0; - struct dentry *ret; + struct dentry *ret = NULL; - fd = get_fuse_dentry(entry); - if (!fd) { + fuse_entry = get_fuse_dentry(entry); + if (!fuse_entry) { ret = ERR_PTR(-EIO); goto out; } - bd = fd->backing_path.dentry; - if (!bd) { + backing_entry = fuse_entry->backing_path.dentry; + if (!backing_entry) { ret = ERR_PTR(-ENOENT); goto out; } - backing_inode = bd->d_inode; - if (!backing_inode) { - ret = 0; - goto out; - } + if (entry_inode) + target_nodeid = get_fuse_inode(entry_inode)->nodeid; - if (d_inode) - target_nodeid = get_fuse_inode(d_inode)->nodeid; + backing_inode = backing_entry->d_inode; + if (backing_inode) + inode = fuse_iget_backing(dir->i_sb, target_nodeid, + backing_inode); - inode = fuse_iget_backing(dir->i_sb, target_nodeid, backing_inode); - if (!inode) { - ret = ERR_PTR(-EIO); - goto out; - } - - error = fuse_handle_bpf_prog(feb, dir, &get_fuse_inode(inode)->bpf); + error = inode ? + fuse_handle_bpf_prog(feb, dir, &get_fuse_inode(inode)->bpf) : + fuse_handle_bpf_prog(feb, dir, &fuse_entry->bpf); if (error) { ret = ERR_PTR(error); goto out; } - error = fuse_handle_backing(feb, &get_fuse_inode(inode)->backing_inode, &fd->backing_path); - if (error) { - ret = ERR_PTR(error); - goto out; - } + if (inode) { + error = fuse_handle_backing(feb, + &get_fuse_inode(inode)->backing_inode, + &fuse_entry->backing_path); + if (error) { + ret = ERR_PTR(error); + goto out; + } - get_fuse_inode(inode)->nodeid = feo->nodeid; - ret = d_splice_alias(inode, entry); - if (!IS_ERR(ret)) - inode = NULL; + get_fuse_inode(inode)->nodeid = feo->nodeid; + ret = d_splice_alias(inode, entry); + if (!IS_ERR(ret)) + inode = NULL; + } out: iput(inode); if (feb->backing_file) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 07cc413638c4..ea116997dee2 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -354,9 +354,14 @@ static void fuse_dentry_release(struct dentry *dentry) { struct fuse_dentry *fd = dentry->d_fsdata; +#ifdef CONFIG_FUSE_BPF if (fd && fd->backing_path.dentry) path_put(&fd->backing_path); + if (fd && fd->bpf) + bpf_prog_put(fd->bpf); +#endif + kfree_rcu(fd, rcu); } #endif diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index f758c09f77fe..a76019d9403f 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -75,7 +75,13 @@ struct fuse_dentry { u64 time; struct rcu_head rcu; }; + +#ifdef CONFIG_FUSE_BPF struct path backing_path; + + /* bpf program *only* set for negative dentries */ + struct bpf_prog *bpf; +#endif }; static inline struct fuse_dentry *get_fuse_dentry(const struct dentry *entry) From 65d60dffad0bffba938260552d0f9c5341a7c130 Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Tue, 25 Jul 2023 11:20:11 -0700 Subject: [PATCH 129/148] UPSTREAM: ANDROID: fuse-bpf: Use stored bpf for create_open create_open would always take its parent directory's bpf for the created object. Modify to use the bpf stored in fuse_dentry which is set by lookup. Bug: 291705489 Test: fuse_test passes, adb push file /sdcard/Android/data works Signed-off-by: Paul Lawrence Change-Id: I0a1ea2a291a8fdf67923f1827176b2ea96bd4c2d --- fs/fuse/backing.c | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 770b6f07f443..731d0336baab 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -261,6 +261,7 @@ int fuse_create_open_backing( struct file *file, unsigned int flags, umode_t mode) { struct fuse_inode *dir_fuse_inode = get_fuse_inode(dir); + struct fuse_dentry *fuse_entry = get_fuse_dentry(entry); struct fuse_dentry *dir_fuse_dentry = get_fuse_dentry(entry->d_parent); struct dentry *backing_dentry = NULL; struct inode *inode = NULL; @@ -292,19 +293,19 @@ int fuse_create_open_backing( if (err) goto out; - if (get_fuse_dentry(entry)->backing_path.dentry) - path_put(&get_fuse_dentry(entry)->backing_path); - get_fuse_dentry(entry)->backing_path = (struct path) { + if (fuse_entry->backing_path.dentry) + path_put(&fuse_entry->backing_path); + fuse_entry->backing_path = (struct path) { .mnt = dir_fuse_dentry->backing_path.mnt, .dentry = backing_dentry, }; - path_get(&get_fuse_dentry(entry)->backing_path); + path_get(&fuse_entry->backing_path); if (d_inode) target_nodeid = get_fuse_inode(d_inode)->nodeid; inode = fuse_iget_backing(dir->i_sb, target_nodeid, - get_fuse_dentry(entry)->backing_path.dentry->d_inode); + fuse_entry->backing_path.dentry->d_inode); if (!inode) { err = -EIO; goto out; @@ -312,9 +313,8 @@ int fuse_create_open_backing( if (get_fuse_inode(inode)->bpf) bpf_prog_put(get_fuse_inode(inode)->bpf); - get_fuse_inode(inode)->bpf = dir_fuse_inode->bpf; - if (get_fuse_inode(inode)->bpf) - bpf_prog_inc(dir_fuse_inode->bpf); + get_fuse_inode(inode)->bpf = fuse_entry->bpf; + fuse_entry->bpf = NULL; newent = d_splice_alias(inode, entry); if (IS_ERR(newent)) { From 7b999dfdbb8232aa30aaae32282126e42810416e Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Tue, 29 Aug 2023 11:17:44 -0700 Subject: [PATCH 130/148] UPSTREAM: ANDROID: fuse-bpf: Get correct inode in mkdir We were getting the inode with the parent inode info Also change variable names to remove confusion Also set bpf correctly in new inode Bug: 293838958 Test: fuse_test, atest ScopedStorageDeviceTest, atest CtsScopedStorageHostTest Signed-off-by: Paul Lawrence (cherry picked from https://android-review.googlesource.com/q/commit:12eb00420cad77d7b7fcfbd1088c921ae629e69d) Merged-In: I0b6a6951599e0d211afd2243daacb98679503448 Change-Id: I0b6a6951599e0d211afd2243daacb98679503448 --- fs/fuse/backing.c | 28 +++++++++++++++++----------- 1 file changed, 17 insertions(+), 11 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 731d0336baab..f4f529ff8580 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1495,32 +1495,33 @@ int fuse_mkdir_initialize( int fuse_mkdir_backing( struct fuse_bpf_args *fa, - struct inode *dir, struct dentry *entry, umode_t mode) + struct inode *dir_inode, struct dentry *entry, umode_t mode) { int err = 0; const struct fuse_mkdir_in *fmi = fa->in_args[0].value; - struct fuse_inode *fuse_inode = get_fuse_inode(dir); - struct inode *backing_inode = fuse_inode->backing_inode; + struct fuse_inode *dir_fuse_inode = get_fuse_inode(dir_inode); + struct inode *dir_backing_inode = dir_fuse_inode->backing_inode; struct path backing_path = {}; struct inode *inode = NULL; - struct dentry *d; //TODO Actually deal with changing the backing entry in mkdir get_fuse_backing_path(entry, &backing_path); if (!backing_path.dentry) return -EBADF; - inode_lock_nested(backing_inode, I_MUTEX_PARENT); + inode_lock_nested(dir_backing_inode, I_MUTEX_PARENT); mode = fmi->mode; - if (!IS_POSIXACL(backing_inode)) + if (!IS_POSIXACL(dir_backing_inode)) mode &= ~fmi->umask; - err = vfs_mkdir(backing_inode, backing_path.dentry, mode); + err = vfs_mkdir(dir_backing_inode, backing_path.dentry, mode); if (err) goto out; if (d_really_is_negative(backing_path.dentry) || unlikely(d_unhashed(backing_path.dentry))) { - d = lookup_one_len(entry->d_name.name, backing_path.dentry->d_parent, - entry->d_name.len); + struct dentry *d = lookup_one_len(entry->d_name.name, + backing_path.dentry->d_parent, + entry->d_name.len); + if (IS_ERR(d)) { err = PTR_ERR(d); goto out; @@ -1528,14 +1529,19 @@ int fuse_mkdir_backing( dput(backing_path.dentry); backing_path.dentry = d; } - inode = fuse_iget_backing(dir->i_sb, fuse_inode->nodeid, backing_inode); + inode = fuse_iget_backing(dir_inode->i_sb, 0, + backing_path.dentry->d_inode); if (IS_ERR(inode)) { err = PTR_ERR(inode); goto out; } d_instantiate(entry, inode); + if (get_fuse_inode(inode)->bpf) + bpf_prog_put(get_fuse_inode(inode)->bpf); + get_fuse_inode(inode)->bpf = get_fuse_dentry(entry)->bpf; + get_fuse_dentry(entry)->bpf = NULL; out: - inode_unlock(backing_inode); + inode_unlock(dir_backing_inode); path_put(&backing_path); return err; } From 9545dda436c3fb536e1c7adc853902fd8c5049e2 Mon Sep 17 00:00:00 2001 From: liujinbao1 Date: Thu, 12 Oct 2023 12:28:06 +0800 Subject: [PATCH 131/148] UPSTREAM: ANDROID: fuse-bpf: Add NULL pointer check in fuse_entry_revalidate If userspace tried to add a backing file in a fuse_dentry_revalidate where there wasn't one originally, this would trigger a crash. Disallow this operation for now. Bug: 296013218 Fixes: 57f3ff964899 ("ANDROID: fuse-bpf v1.1") Test: fuse_test passes, following script no longer crashes: adb shell su root setenforce 0 adb shell su root chmod ug+w /data/media adb shell su root rm /data/media/Android -rf adb shell su root mkdir -p /storage/emulated/Android/data/test adb shell su root ls -l /storage/emulated/Android/data/test Change-Id: Id8a67c43d1edfa010403d5f17e31109b796998cf Signed-off-by: liujinbao1 (cherry picked from commit e89b1266f784b2271af2e72a5d04e3e39d0afcdc) --- fs/fuse/dir.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index ea116997dee2..0fd334a50c53 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -183,7 +183,7 @@ static bool backing_data_changed(struct fuse_inode *fi, struct dentry *entry, int err; bool ret = true; - if (!entry) { + if (!entry || !fi->backing_inode) { ret = false; goto put_backing_file; } From 35e153203115b59ea3f4120c073afca30176373f Mon Sep 17 00:00:00 2001 From: Lyon Wang Date: Mon, 30 Oct 2023 19:22:31 +0800 Subject: [PATCH 132/148] UPSTREAM: ANDROID: fs/passthrough: Fix compatibility with R/O file system Look at the passthrough file and check to see if its mount is read-only. Bug: 297482438 Signed-off-by: Lyon Wang Change-Id: I0f483c6bcb1effe395eee07b6d721f343840d115 (cherry picked from commit dbeed231965e1a01bf0d70ea5439fbfacc7c4aae) --- fs/fuse/passthrough.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c index 6031e72f3b5a..1845e05db2da 100644 --- a/fs/fuse/passthrough.c +++ b/fs/fuse/passthrough.c @@ -213,7 +213,8 @@ int fuse_passthrough_open(struct fuse_dev *fud, u32 lower_fd) } if (!passthrough_filp->f_op->read_iter || - !passthrough_filp->f_op->write_iter) { + !((passthrough_filp->f_path.mnt->mnt_flags | MNT_READONLY) || + passthrough_filp->f_op->write_iter)) { pr_err("FUSE: passthrough file misses file operations.\n"); res = -EBADF; goto err_free_file; From 3ab55eae015d00fbc871572aadd1dcc1da74ce7f Mon Sep 17 00:00:00 2001 From: liujing40 Date: Wed, 25 Oct 2023 15:26:27 +0800 Subject: [PATCH 133/148] UPSTREAM: ANDROID: fuse-bpf: Add NULL pointer check in fuse_release_in If open request sent to classic fuse, backing_file is null. In fuse_release_initialize, fput will trigger a crash. Bug: 297831741 Signed-off-by: liujing40 (cherry picked from https://android-review.googlesource.com/q/commit:4d2ff573981f06ba09e1ddda8726bb73ff6a2c3f) Merged-In: I2d54d99d62b54c39a6dc9064f8f62488433aff6f Change-Id: I2d54d99d62b54c39a6dc9064f8f62488433aff6f --- fs/fuse/backing.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index f4f529ff8580..3b2b2a470ba4 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -353,7 +353,9 @@ int fuse_release_initialize(struct fuse_bpf_args *fa, struct fuse_release_in *fr struct inode *inode, struct fuse_file *ff) { /* Always put backing file whatever bpf/userspace says */ - fput(ff->backing_file); + if (ff->backing_file) { + fput(ff->backing_file); + } *fri = (struct fuse_release_in) { .fh = ff->fh, From fa625f8622651a9acc9cf48a796153717c027e7a Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Wed, 15 Nov 2023 14:29:23 -0800 Subject: [PATCH 134/148] BACKPORT: ANDROID: fuse-bpf: Ignore readaheads unless they go to the daemon readpages will be triggered on the fuse fs in passthrough mode though system calls like fadvise. If the daemon isn't aware of the file, this will likely cause a hang. For the moment, simply ignore fadvise in this situation Bug: 301201239 Test: fuse_test, atest ScopedStorageDeviceTest both pass Signed-off-by: Paul Lawrence (cherry picked from https://android-review.googlesource.com/q/commit:ac9071df3ba6715219a16a44d4711f041b0c25de) Merged-In: I524a84aeeb1b1593e51264fcc37f7cfa66757168 Change-Id: I524a84aeeb1b1593e51264fcc37f7cfa66757168 --- fs/fuse/file.c | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index ac3b4e9f5668..cad73bad9c90 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1033,6 +1033,19 @@ static int fuse_readpages_fill(void *_data, struct page *page) struct inode *inode = data->inode; struct fuse_conn *fc = get_fuse_conn(inode); +#ifdef CONFIG_FUSE_BPF + /* + * Currently no meaningful readahead is possible with fuse-bpf within + * the kernel, so unless the daemon is aware of this file, ignore this + * call. + */ + if (!get_fuse_inode(inode)->nodeid) + return -EIO; +#endif + + if (fuse_is_bad(inode)) + return -EIO; + fuse_wait_on_page_writeback(inode, page->index); if (ap->num_pages && From 2068b53c6a01c582a065884c5f3280f53e2e1071 Mon Sep 17 00:00:00 2001 From: Hangyu Hua Date: Thu, 16 Nov 2023 15:57:26 +0800 Subject: [PATCH 135/148] UPSTREAM: fuse: dax: set fc->dax to NULL in fuse_dax_conn_free() commit 7f8ed28d1401320bcb02dda81b3c23ab2dc5a6d8 upstream. fuse_dax_conn_free() will be called when fuse_fill_super_common() fails after fuse_dax_conn_alloc(). Then deactivate_locked_super() in virtio_fs_get_tree() will call virtio_kill_sb() to release the discarded superblock. This will call fuse_dax_conn_free() again in fuse_conn_put(), resulting in a possible double free. Fixes: 1dd539577c42 ("virtiofs: add a mount option to enable dax") Change-Id: I5aabc81ad4f96078c612f2ec1c9077f21f4ffae4 Signed-off-by: Hangyu Hua Acked-by: Vivek Goyal Reviewed-by: Jingbo Xu Cc: # v5.10 Signed-off-by: Miklos Szeredi Signed-off-by: Greg Kroah-Hartman --- fs/fuse/dax.c | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index ff99ab2a3c43..2739e218df64 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -1228,6 +1228,7 @@ void fuse_dax_conn_free(struct fuse_conn *fc) if (fc->dax) { fuse_free_dax_mem_ranges(&fc->dax->free_ranges); kfree(fc->dax); + fc->dax = NULL; } } From aeeb2c9cb34609af522634fffdbc462d95d38e9d Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Wed, 2 Aug 2023 12:23:44 -0700 Subject: [PATCH 136/148] UPSTREAM: ANDROID: fuse-bpf: Follow mounts in lookups Bug: 292925770 Test: fuse_test run. The following steps on Android also now pass: Create /data/123 and /data/media/0/Android/data/45 directories Mount /data/123 directory to /data/media/0/Android/data/45 directory Create 1.txt under the /data/123 directory File 1.txt should appear in /storage/emulated/0/Android/data/45 Signed-off-by: Paul Lawrence (cherry picked from https://android-review.googlesource.com/q/commit:9323938705b42cb4dd863d5cf8022ba8f2282952) Merged-In: I1fe27d743ca2981e624a9aa87d9ab6deb313aadc Change-Id: I1fe27d743ca2981e624a9aa87d9ab6deb313aadc --- fs/fuse/backing.c | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 3b2b2a470ba4..859114b13508 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1167,7 +1167,6 @@ int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, struct kstat stat; int err; - /* TODO this will not handle lookups over mount points */ inode_lock_nested(dir_backing_inode, I_MUTEX_PARENT); backing_entry = lookup_one_len(entry->d_name.name, dir_backing_entry, strlen(entry->d_name.name)); @@ -1186,16 +1185,22 @@ int fuse_lookup_backing(struct fuse_bpf_args *fa, struct inode *dir, return 0; } + err = follow_down(&fuse_entry->backing_path); + if (err) + goto err_out; + err = vfs_getattr(&fuse_entry->backing_path, &stat, STATX_BASIC_STATS, 0); - if (err) { - path_put_init(&fuse_entry->backing_path); - return err; - } + if (err) + goto err_out; fuse_stat_to_attr(get_fuse_conn(dir), backing_entry->d_inode, &stat, &feo->attr); return 0; + +err_out: + path_put_init(&fuse_entry->backing_path); + return err; } int fuse_handle_backing(struct fuse_entry_bpf *feb, struct inode **backing_inode, From 84ab1889e1f32cf7423a3e893e1bcddec44653eb Mon Sep 17 00:00:00 2001 From: Paul Lawrence Date: Tue, 9 Jan 2024 10:02:00 -0800 Subject: [PATCH 137/148] UPSTREAM: ANDROID: fuse-bpf: Fix the issue of abnormal lseek system calls fuse_lseek_backing was returning the offset as an int, which would then be treated as an ERR if in the range 4G-4096 and 4G. Although the call would appear to work correctly, the file position would be incorrect according to a subsequent fseek with SEEK_CUR. Based on a change by chenyuwen who found and fixed this issue. Bug: 319219307 Change-Id: I3aef5fb22751a72ce2bd7674ee081956a89fc752 Signed-off-by: chenyuwen Signed-off-by: Paul Lawrence --- fs/fuse/backing.c | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 859114b13508..e9533afd5961 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -454,23 +454,26 @@ int fuse_lseek_backing(struct fuse_bpf_args *fa, struct file *file, loff_t offse struct file *backing_file = fuse_file->backing_file; loff_t ret; - /* TODO: Handle changing of the file handle */ if (offset == 0) { if (whence == SEEK_CUR) { flo->offset = file->f_pos; - return flo->offset; + return 0; } if (whence == SEEK_SET) { flo->offset = vfs_setpos(file, 0, 0); - return flo->offset; + return 0; } } inode_lock(file->f_inode); backing_file->f_pos = file->f_pos; ret = vfs_llseek(backing_file, fli->offset, fli->whence); - flo->offset = ret; + + if (!IS_ERR(ERR_PTR(ret))) { + flo->offset = ret; + ret = 0; + } inode_unlock(file->f_inode); return ret; } From 5477ccc442c796ef048557769454641b18b8e32e Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Wed, 28 Feb 2024 16:50:49 +0100 Subject: [PATCH 138/148] UPSTREAM: fuse: fix root lookup with nonzero generation [ Upstream commit 68ca1b49e430f6534d0774a94147a823e3b8b26e ] The root inode has a fixed nodeid and generation (1, 0). Prior to the commit 15db16837a35 ("fuse: fix illegal access to inode with reused nodeid") generation number on lookup was ignored. After this commit lookup with the wrong generation number resulted in the inode being unhashed. This is correct for non-root inodes, but replacing the root inode is wrong and results in weird behavior. Fix by reverting to the old behavior if ignoring the generation for the root inode, but issuing a warning in dmesg. Reported-by: Antonio SJ Musumeci Closes: https://lore.kernel.org/all/CAOQ4uxhek5ytdN8Yz2tNEOg5ea4NkBb4nk0FGPjPk_9nz-VG3g@mail.gmail.com/ Fixes: 15db16837a35 ("fuse: fix illegal access to inode with reused nodeid") Cc: # v5.14 Change-Id: I83231b08b813b2acead0c3af078b6933b6895d6e Signed-off-by: Miklos Szeredi Signed-off-by: Sasha Levin --- fs/fuse/dir.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 0fd334a50c53..02c00217b385 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -614,6 +614,10 @@ int fuse_lookup_name(struct super_block *sb, u64 nodeid, const struct qstr *name goto out_put_forget; if (fuse_invalid_attr(&outarg->attr)) goto out_put_forget; + if (outarg->nodeid == FUSE_ROOT_ID && outarg->generation != 0) { + pr_warn_once("root generation should be zero\n"); + outarg->generation = 0; + } *inode = fuse_iget(sb, outarg->nodeid, outarg->generation, &outarg->attr, entry_attr_timeout(outarg), From c8c8c5680d7f1743b44fb4db95e5413a2e63fed0 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Wed, 4 Aug 2021 13:22:58 +0200 Subject: [PATCH 139/148] BACKPORT: fuse: name fs_context consistently [ Upstream commit 84c215075b5723ab946708a6c74c26bd3c51114c ] Naming convention under fs/fuse/: struct fuse_conn *fc; struct fs_context *fsc; Change-Id: I80e624089f822b1480b766825265a77f30e10787 Signed-off-by: Miklos Szeredi Stable-dep-of: 525bd65aa759 ("fuse: verify {g,u}id mount options correctly") Signed-off-by: Sasha Levin --- fs/fuse/control.c | 10 ++++---- fs/fuse/inode.c | 60 ++++++++++++++++++++++----------------------- fs/fuse/virtio_fs.c | 12 ++++----- 3 files changed, 41 insertions(+), 41 deletions(-) diff --git a/fs/fuse/control.c b/fs/fuse/control.c index 9b46669f8f85..4aae0e606ba4 100644 --- a/fs/fuse/control.c +++ b/fs/fuse/control.c @@ -328,7 +328,7 @@ void fuse_ctl_remove_conn(struct fuse_conn *fc) drop_nlink(d_inode(fuse_control_sb->s_root)); } -static int fuse_ctl_fill_super(struct super_block *sb, struct fs_context *fctx) +static int fuse_ctl_fill_super(struct super_block *sb, struct fs_context *fsc) { static const struct tree_descr empty_descr = {""}; struct fuse_conn *fc; @@ -354,18 +354,18 @@ static int fuse_ctl_fill_super(struct super_block *sb, struct fs_context *fctx) return 0; } -static int fuse_ctl_get_tree(struct fs_context *fc) +static int fuse_ctl_get_tree(struct fs_context *fsc) { - return get_tree_single(fc, fuse_ctl_fill_super); + return get_tree_single(fsc, fuse_ctl_fill_super); } static const struct fs_context_operations fuse_ctl_context_ops = { .get_tree = fuse_ctl_get_tree, }; -static int fuse_ctl_init_fs_context(struct fs_context *fc) +static int fuse_ctl_init_fs_context(struct fs_context *fsc) { - fc->ops = &fuse_ctl_context_ops; + fsc->ops = &fuse_ctl_context_ops; return 0; } diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 98b2371a8712..dbf377ea61de 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -153,12 +153,12 @@ static void fuse_destroy_inode(struct inode *inode) } #endif -static int fuse_reconfigure(struct fs_context *fc) +static int fuse_reconfigure(struct fs_context *fsc) { - struct super_block *sb = fc->root->d_sb; + struct super_block *sb = fsc->root->d_sb; sync_filesystem(sb); - if (fc->sb_flags & SB_MANDLOCK) + if (fsc->sb_flags & SB_MANDLOCK) return -EINVAL; return 0; @@ -660,38 +660,38 @@ static const struct fs_parameter_description fuse_fs_parameters = { .specs = fuse_param_specs, }; -static int fuse_parse_param(struct fs_context *fc, struct fs_parameter *param) +static int fuse_parse_param(struct fs_context *fsc, struct fs_parameter *param) { struct fs_parse_result result; - struct fuse_fs_context *ctx = fc->fs_private; + struct fuse_fs_context *ctx = fsc->fs_private; int opt; - if (fc->purpose == FS_CONTEXT_FOR_RECONFIGURE) { + if (fsc->purpose == FS_CONTEXT_FOR_RECONFIGURE) { /* * Ignore options coming from mount(MS_REMOUNT) for backward * compatibility. */ - if (fc->oldapi) + if (fsc->oldapi) return 0; - return invalf(fc, "No changes allowed in reconfigure"); + return invalf(fsc, "No changes allowed in reconfigure"); } - opt = fs_parse(fc, &fuse_fs_parameters, param, &result); + opt = fs_parse(fsc, &fuse_fs_parameters, param, &result); if (opt < 0) return opt; switch (opt) { case OPT_SOURCE: - if (fc->source) - return invalf(fc, "fuse: Multiple sources specified"); - fc->source = param->string; + if (fsc->source) + return invalf(fsc, "Multiple sources specified"); + fsc->source = param->string; param->string = NULL; break; case OPT_SUBTYPE: if (ctx->subtype) - return invalf(fc, "fuse: Multiple subtypes specified"); + return invalf(fsc, "Multiple subtypes specified"); ctx->subtype = param->string; param->string = NULL; return 0; @@ -703,22 +703,22 @@ static int fuse_parse_param(struct fs_context *fc, struct fs_parameter *param) case OPT_ROOTMODE: if (!fuse_valid_type(result.uint_32)) - return invalf(fc, "fuse: Invalid rootmode"); + return invalf(fsc, "Invalid rootmode"); ctx->rootmode = result.uint_32; ctx->rootmode_present = true; break; case OPT_USER_ID: - ctx->user_id = make_kuid(fc->user_ns, result.uint_32); + ctx->user_id = make_kuid(fsc->user_ns, result.uint_32); if (!uid_valid(ctx->user_id)) - return invalf(fc, "fuse: Invalid user_id"); + return invalf(fsc, "Invalid user_id"); ctx->user_id_present = true; break; case OPT_GROUP_ID: - ctx->group_id = make_kgid(fc->user_ns, result.uint_32); + ctx->group_id = make_kgid(fsc->user_ns, result.uint_32); if (!gid_valid(ctx->group_id)) - return invalf(fc, "fuse: Invalid group_id"); + return invalf(fsc, "Invalid group_id"); ctx->group_id_present = true; break; @@ -736,7 +736,7 @@ static int fuse_parse_param(struct fs_context *fc, struct fs_parameter *param) case OPT_BLKSIZE: if (!ctx->is_bdev) - return invalf(fc, "fuse: blksize only supported for fuseblk"); + return invalf(fsc, "blksize only supported for fuseblk"); ctx->blksize = result.uint_32; break; @@ -767,9 +767,9 @@ static int fuse_parse_param(struct fs_context *fc, struct fs_parameter *param) return 0; } -static void fuse_free_fc(struct fs_context *fc) +static void fuse_free_fsc(struct fs_context *fsc) { - struct fuse_fs_context *ctx = fc->fs_private; + struct fuse_fs_context *ctx = fsc->fs_private; if (ctx) { if (ctx->root_dir) @@ -1665,9 +1665,9 @@ static int fuse_fill_super(struct super_block *sb, struct fs_context *fsc) return err; } -static int fuse_get_tree(struct fs_context *fc) +static int fuse_get_tree(struct fs_context *fsc) { - struct fuse_fs_context *ctx = fc->fs_private; + struct fuse_fs_context *ctx = fsc->fs_private; if (!ctx->fd_present || !ctx->rootmode_present || !ctx->user_id_present || !ctx->group_id_present) @@ -1675,14 +1675,14 @@ static int fuse_get_tree(struct fs_context *fc) #ifdef CONFIG_BLOCK if (ctx->is_bdev) - return get_tree_bdev(fc, fuse_fill_super); + return get_tree_bdev(fsc, fuse_fill_super); #endif - return get_tree_nodev(fc, fuse_fill_super); + return get_tree_nodev(fsc, fuse_fill_super); } static const struct fs_context_operations fuse_context_ops = { - .free = fuse_free_fc, + .free = fuse_free_fsc, .parse_param = fuse_parse_param, .reconfigure = fuse_reconfigure, .get_tree = fuse_get_tree, @@ -1691,7 +1691,7 @@ static const struct fs_context_operations fuse_context_ops = { /* * Set up the filesystem mount context. */ -static int fuse_init_fs_context(struct fs_context *fc) +static int fuse_init_fs_context(struct fs_context *fsc) { struct fuse_fs_context *ctx; @@ -1704,14 +1704,14 @@ static int fuse_init_fs_context(struct fs_context *fc) ctx->legacy_opts_show = true; #ifdef CONFIG_BLOCK - if (fc->fs_type == &fuseblk_fs_type) { + if (fsc->fs_type == &fuseblk_fs_type) { ctx->is_bdev = true; ctx->destroy = true; } #endif - fc->fs_private = ctx; - fc->ops = &fuse_context_ops; + fsc->fs_private = ctx; + fsc->ops = &fuse_context_ops; return 0; } diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index cc43a773b32e..c23a94d179fd 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -101,14 +101,14 @@ static const struct fs_parameter_description virtio_fs_parameters = { .specs = virtio_param_specs, }; -static int virtio_fs_parse_param(struct fs_context *fc, +static int virtio_fs_parse_param(struct fs_context *fsc, struct fs_parameter *param) { struct fs_parse_result result; - struct fuse_fs_context *ctx = fc->fs_private; + struct fuse_fs_context *ctx = fsc->fs_private; int opt; - opt = fs_parse(fc, &virtio_fs_parameters, param, &result); + opt = fs_parse(fsc, &virtio_fs_parameters, param, &result); if (opt < 0) return opt; @@ -123,9 +123,9 @@ static int virtio_fs_parse_param(struct fs_context *fc, return 0; } -static void virtio_fs_free_fc(struct fs_context *fc) +static void virtio_fs_free_fsc(struct fs_context *fsc) { - struct fuse_fs_context *ctx = fc->fs_private; + struct fuse_fs_context *ctx = fsc->fs_private; kfree(ctx); } @@ -1517,7 +1517,7 @@ out_err: } static const struct fs_context_operations virtio_fs_context_ops = { - .free = virtio_fs_free_fc, + .free = virtio_fs_free_fsc, .parse_param = virtio_fs_parse_param, .get_tree = virtio_fs_get_tree, }; From 8c6efff53a404ef19a84be320caa40cbc95a21f6 Mon Sep 17 00:00:00 2001 From: Tashfin Shakeer Rhythm Date: Fri, 3 Apr 2026 02:17:42 +0600 Subject: [PATCH 140/148] fixup! BACKPORT: fuse: name fs_context consistently This is a follow-up to commit a20fb465b5b0cb0be63c5e552ad58017d7d1818c that fixes missing instances where s/fc/fsc wasn't performed which breaks compilation with CONFIG_FUSE_BPF enabled. Change-Id: I1d5543efc56c180e0416d09a393139fe8e9667b5 Signed-off-by: Tashfin Shakeer Rhythm --- fs/fuse/inode.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index dbf377ea61de..4679cf592225 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -745,14 +745,14 @@ static int fuse_parse_param(struct fs_context *fsc, struct fs_parameter *param) BPF_PROG_TYPE_FUSE, false); if (IS_ERR(ctx->root_bpf)) { ctx->root_bpf = NULL; - return invalf(fc, "Unable to open bpf program"); + return invalf(fsc, "Unable to open bpf program"); } break; case OPT_ROOT_DIR: ctx->root_dir = fget(result.uint_32); if (!ctx->root_dir) - return invalf(fc, "Unable to open root directory"); + return invalf(fsc, "Unable to open root directory"); break; case OPT_NO_DAEMON: From ec0df7ca6d30e0a673e180a2d00499a3be443fe6 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Tue, 2 Jul 2024 17:22:41 -0500 Subject: [PATCH 141/148] UPSTREAM: fuse: verify {g,u}id mount options correctly [ Upstream commit 525bd65aa759ec320af1dc06e114ed69733e9e23 ] As was done in 0200679fc795 ("tmpfs: verify {g,u}id mount options correctly") we need to validate that the requested uid and/or gid is representable in the filesystem's idmapping. Cribbing from the above commit log, The contract for {g,u}id mount options and {g,u}id values in general set from userspace has always been that they are translated according to the caller's idmapping. In so far, fuse has been doing the correct thing. But since fuse is mountable in unprivileged contexts it is also necessary to verify that the resulting {k,g}uid is representable in the namespace of the superblock. Fixes: c30da2e981a7 ("fuse: convert to use the new mount API") Cc: stable@vger.kernel.org # 5.4+ Change-Id: Ic3734ff9039e40ae029872bf334f6d3448ef8845 Signed-off-by: Eric Sandeen Link: https://lore.kernel.org/r/8f07d45d-c806-484d-a2e3-7a2199df1cd2@redhat.com Reviewed-by: Christian Brauner Reviewed-by: Josef Bacik Signed-off-by: Christian Brauner Signed-off-by: Sasha Levin --- fs/fuse/inode.c | 24 ++++++++++++++++++++---- 1 file changed, 20 insertions(+), 4 deletions(-) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 4679cf592225..8488a862f4da 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -665,6 +665,8 @@ static int fuse_parse_param(struct fs_context *fsc, struct fs_parameter *param) struct fs_parse_result result; struct fuse_fs_context *ctx = fsc->fs_private; int opt; + kuid_t kuid; + kgid_t kgid; if (fsc->purpose == FS_CONTEXT_FOR_RECONFIGURE) { /* @@ -709,16 +711,30 @@ static int fuse_parse_param(struct fs_context *fsc, struct fs_parameter *param) break; case OPT_USER_ID: - ctx->user_id = make_kuid(fsc->user_ns, result.uint_32); - if (!uid_valid(ctx->user_id)) + kuid = make_kuid(fsc->user_ns, result.uint_32); + if (!uid_valid(kuid)) return invalf(fsc, "Invalid user_id"); + /* + * The requested uid must be representable in the + * filesystem's idmapping. + */ + if (!kuid_has_mapping(fsc->user_ns, kuid)) + return invalf(fsc, "Invalid user_id"); + ctx->user_id = kuid; ctx->user_id_present = true; break; case OPT_GROUP_ID: - ctx->group_id = make_kgid(fsc->user_ns, result.uint_32); - if (!gid_valid(ctx->group_id)) + kgid = make_kgid(fsc->user_ns, result.uint_32);; + if (!gid_valid(kgid)) return invalf(fsc, "Invalid group_id"); + /* + * The requested gid must be representable in the + * filesystem's idmapping. + */ + if (!kgid_has_mapping(fsc->user_ns, kgid)) + return invalf(fsc, "Invalid group_id"); + ctx->group_id = kgid; ctx->group_id_present = true; break; From 1c9bc1da8e3dbf68601aeefa3a73ba803099b97b Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Tue, 11 Feb 2025 13:36:52 -0800 Subject: [PATCH 142/148] UPSTREAM: ANDROID: fuse-bpf: Correct fuse bpf feature flag The feature flag should only advertise fuse-bpf if fuse-bpf is a supported feature Bug: 372951405 Test: Compile with CONFIG_FUSE_BPF unset Change-Id: I0049a3075f78576499168b8ebb6e833ccd18db0f Signed-off-by: Daniel Rosenberg --- fs/fuse/inode.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 8488a862f4da..39d9d531c965 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1884,6 +1884,7 @@ static void fuse_fs_cleanup(void) static struct kobject *fuse_kobj; +#ifdef CONFIG_FUSE_BPF static ssize_t fuse_bpf_show(struct kobject *kobj, struct kobj_attribute *attr, char *buff) { @@ -1892,9 +1893,12 @@ static ssize_t fuse_bpf_show(struct kobject *kobj, static struct kobj_attribute fuse_bpf_attr = __ATTR_RO(fuse_bpf); +#endif static struct attribute *bpf_features[] = { +#ifdef CONFIG_FUSE_BPF &fuse_bpf_attr.attr, +#endif NULL, }; From 389625e5d729aae318768f66cd315fd507861966 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Fri, 22 Oct 2021 17:03:01 +0200 Subject: [PATCH 143/148] UPSTREAM: fuse: make sure reclaim doesn't write the inode In writeback cache mode mtime/ctime updates are cached, and flushed to the server using the ->write_inode() callback. Closing the file will result in a dirty inode being immediately written, but in other cases the inode can remain dirty after all references are dropped. This result in the inode being written back from reclaim, which can deadlock on a regular allocation while the request is being served. The usual mechanisms (GFP_NOFS/PF_MEMALLOC*) don't work for FUSE, because serving a request involves unrelated userspace process(es). Instead do the same as for dirty pages: make sure the inode is written before the last reference is gone. - fallocate(2)/copy_file_range(2): these call file_update_time() or file_modified(), so flush the inode before returning from the call - unlink(2), link(2) and rename(2): these call fuse_update_ctime(), so flush the ctime directly from this helper fuse_flush_time_update(inode) was skipped to call in __fuse_copy_file_range() because of huge dependent changes. Change-Id: I102dab1992c9ed2b5e89606265b3d3aa9c1cdb8a Reported-by: chenguanyou Signed-off-by: Miklos Szeredi Git-commit: 5c791fe1e2a4f401f819065ea4fc0450849f1818 Git-repo: https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git Signed-off-by: Pradeep P V K --- fs/fuse/dir.c | 8 ++++++++ fs/fuse/file.c | 12 ++++++++++++ fs/fuse/fuse_i.h | 1 + fs/fuse/inode.c | 3 +++ 4 files changed, 24 insertions(+) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 02c00217b385..a55375ae2797 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -1027,11 +1027,19 @@ static int fuse_symlink(struct inode *dir, struct dentry *entry, return create_new_entry(fm, &args, dir, entry, S_IFLNK); } +void fuse_flush_time_update(struct inode *inode) +{ + int err = sync_inode_metadata(inode, 1); + + mapping_set_error(inode->i_mapping, err); +} + void fuse_update_ctime(struct inode *inode) { if (!IS_NOCMTIME(inode)) { inode->i_ctime = current_time(inode); mark_inode_dirty_sync(inode); + fuse_flush_time_update(inode); } } diff --git a/fs/fuse/file.c b/fs/fuse/file.c index cad73bad9c90..d96daeaa807b 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -2022,6 +2022,16 @@ int fuse_write_inode(struct inode *inode, struct writeback_control *wbc) */ if (!S_ISREG(inode->i_mode)) return 0; + /* + * Inode is always written before the last reference is dropped and + * hence this should not be reached from reclaim. + * + * Writing back the inode from reclaim can deadlock if the request + * processing itself needs an allocation. Allocations triggering + * reclaim while serving a request can't be prevented, because it can + * involve any number of unrelated userspace processes. + */ + WARN_ON(wbc->for_reclaim); ff = __fuse_write_file_get(fc, fi); err = fuse_flush_times(inode, ff); @@ -3559,6 +3569,8 @@ out: inode_unlock(inode); + fuse_flush_time_update(inode); + return err; } diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index a76019d9403f..97b44bbf5274 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1216,6 +1216,7 @@ int fuse_allow_current_process(struct fuse_conn *fc); u64 fuse_lock_owner_id(struct fuse_conn *fc, fl_owner_t id); +void fuse_flush_time_update(struct inode *inode); void fuse_update_ctime(struct inode *inode); int fuse_update_attributes(struct inode *inode, struct file *file); diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 39d9d531c965..e37fba257c8b 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -125,6 +125,9 @@ static void fuse_evict_inode(struct inode *inode) { struct fuse_inode *fi = get_fuse_inode(inode); + /* Will write inode on close/munmap and in all other dirtiers */ + WARN_ON(inode->i_state & I_DIRTY_INODE); + truncate_inode_pages_final(&inode->i_data); clear_inode(inode); if (inode->i_sb->s_flags & SB_ACTIVE) { From 1ec19d7588bab7adeab4ef48b57eaa5b2ea72670 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Tue, 11 Sep 2018 21:59:08 +0200 Subject: [PATCH 144/148] BACKPORT: compat_ioctl: move more drivers to compat_ptr_ioctl The .ioctl and .compat_ioctl file operations have the same prototype so they can both point to the same function, which works great almost all the time when all the commands are compatible. One exception is the s390 architecture, where a compat pointer is only 31 bit wide, and converting it into a 64-bit pointer requires calling compat_ptr(). Most drivers here will never run in s390, but since we now have a generic helper for it, it's easy enough to use it consistently. I double-checked all these drivers to ensure that all ioctl arguments are used as pointers or are ignored, but are not interpreted as integer values. Acked-by: Jason Gunthorpe Acked-by: Daniel Vetter Acked-by: Mauro Carvalho Chehab Acked-by: Greg Kroah-Hartman Acked-by: David Sterba Acked-by: Darren Hart (VMware) Acked-by: Jonathan Cameron Acked-by: Bjorn Andersson Acked-by: Dan Williams Change-Id: I7c19d15ea2d2e48a30e726ca0a5b8ee915e6bbe7 Signed-off-by: Arnd Bergmann --- drivers/android/binder.c | 2 +- drivers/crypto/qat/qat_common/adf_ctl_drv.c | 2 +- drivers/dma-buf/dma-buf.c | 4 +--- drivers/dma-buf/sw_sync.c | 2 +- drivers/dma-buf/sync_file.c | 2 +- drivers/gpu/drm/amd/amdkfd/kfd_chardev.c | 2 +- drivers/hid/hidraw.c | 4 +--- drivers/iio/industrialio-core.c | 2 +- drivers/infiniband/core/uverbs_main.c | 4 ++-- drivers/media/rc/lirc_dev.c | 4 +--- drivers/misc/vmw_vmci/vmci_host.c | 2 +- drivers/nvdimm/bus.c | 4 ++-- drivers/nvme/host/core.c | 2 +- drivers/pci/switch/switchtec.c | 2 +- drivers/platform/x86/wmi.c | 2 +- drivers/rpmsg/rpmsg_char.c | 4 ++-- drivers/sbus/char/display7seg.c | 2 +- drivers/sbus/char/envctrl.c | 4 +--- drivers/scsi/3w-xxxx.c | 4 +--- drivers/scsi/cxlflash/main.c | 2 +- drivers/scsi/esas2r/esas2r_main.c | 2 +- drivers/scsi/pmcraid.c | 4 +--- drivers/staging/android/ion/ion.c | 4 +--- drivers/staging/vme/devices/vme_user.c | 2 +- drivers/tee/tee_core.c | 2 +- drivers/usb/class/cdc-wdm.c | 2 +- drivers/usb/class/usbtmc.c | 4 +--- drivers/virt/fsl_hypervisor.c | 2 +- fs/btrfs/super.c | 2 +- fs/fuse/dev.c | 2 +- fs/notify/fanotify/fanotify_user.c | 2 +- fs/userfaultfd.c | 2 +- net/rfkill/core.c | 2 +- 33 files changed, 36 insertions(+), 52 deletions(-) diff --git a/drivers/android/binder.c b/drivers/android/binder.c index 5419d6e0e560..cfe7a2aa026c 100644 --- a/drivers/android/binder.c +++ b/drivers/android/binder.c @@ -6923,7 +6923,7 @@ const struct file_operations binder_fops = { .owner = THIS_MODULE, .poll = binder_poll, .unlocked_ioctl = binder_ioctl, - .compat_ioctl = binder_ioctl, + .compat_ioctl = compat_ptr_ioctl, .mmap = binder_mmap, .open = binder_open, .flush = binder_flush, diff --git a/drivers/crypto/qat/qat_common/adf_ctl_drv.c b/drivers/crypto/qat/qat_common/adf_ctl_drv.c index abc7a7f64d64..ef0e482ee04f 100644 --- a/drivers/crypto/qat/qat_common/adf_ctl_drv.c +++ b/drivers/crypto/qat/qat_common/adf_ctl_drv.c @@ -68,7 +68,7 @@ static long adf_ctl_ioctl(struct file *fp, unsigned int cmd, unsigned long arg); static const struct file_operations adf_ctl_ops = { .owner = THIS_MODULE, .unlocked_ioctl = adf_ctl_ioctl, - .compat_ioctl = adf_ctl_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; struct adf_ctl_drv_info { diff --git a/drivers/dma-buf/dma-buf.c b/drivers/dma-buf/dma-buf.c index d60b7ced21cb..d0d4fdd38a2c 100644 --- a/drivers/dma-buf/dma-buf.c +++ b/drivers/dma-buf/dma-buf.c @@ -463,9 +463,7 @@ static const struct file_operations dma_buf_fops = { .llseek = dma_buf_llseek, .poll = dma_buf_poll, .unlocked_ioctl = dma_buf_ioctl, -#ifdef CONFIG_COMPAT - .compat_ioctl = dma_buf_ioctl, -#endif + .compat_ioctl = compat_ptr_ioctl, .show_fdinfo = dma_buf_show_fdinfo, }; diff --git a/drivers/dma-buf/sw_sync.c b/drivers/dma-buf/sw_sync.c index 7e7356970d5f..7f5ed1aa7a9f 100644 --- a/drivers/dma-buf/sw_sync.c +++ b/drivers/dma-buf/sw_sync.c @@ -408,5 +408,5 @@ const struct file_operations sw_sync_debugfs_fops = { .open = sw_sync_debugfs_open, .release = sw_sync_debugfs_release, .unlocked_ioctl = sw_sync_ioctl, - .compat_ioctl = sw_sync_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; diff --git a/drivers/dma-buf/sync_file.c b/drivers/dma-buf/sync_file.c index 992c15c016dd..f0c822952c20 100644 --- a/drivers/dma-buf/sync_file.c +++ b/drivers/dma-buf/sync_file.c @@ -481,5 +481,5 @@ static const struct file_operations sync_file_fops = { .release = sync_file_release, .poll = sync_file_poll, .unlocked_ioctl = sync_file_ioctl, - .compat_ioctl = sync_file_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_chardev.c b/drivers/gpu/drm/amd/amdkfd/kfd_chardev.c index 548cd062b9cd..fb5fca3d2149 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_chardev.c +++ b/drivers/gpu/drm/amd/amdkfd/kfd_chardev.c @@ -49,7 +49,7 @@ static const char kfd_dev_name[] = "kfd"; static const struct file_operations kfd_fops = { .owner = THIS_MODULE, .unlocked_ioctl = kfd_ioctl, - .compat_ioctl = kfd_ioctl, + .compat_ioctl = compat_ptr_ioctl, .open = kfd_open, .mmap = kfd_mmap, }; diff --git a/drivers/hid/hidraw.c b/drivers/hid/hidraw.c index f8ef6268f3f2..97a4eb4ca1fb 100644 --- a/drivers/hid/hidraw.c +++ b/drivers/hid/hidraw.c @@ -472,9 +472,7 @@ static const struct file_operations hidraw_ops = { .release = hidraw_release, .unlocked_ioctl = hidraw_ioctl, .fasync = hidraw_fasync, -#ifdef CONFIG_COMPAT - .compat_ioctl = hidraw_ioctl, -#endif + .compat_ioctl = compat_ptr_ioctl, .llseek = noop_llseek, }; diff --git a/drivers/iio/industrialio-core.c b/drivers/iio/industrialio-core.c index 485b2e6748c5..243a1600096a 100644 --- a/drivers/iio/industrialio-core.c +++ b/drivers/iio/industrialio-core.c @@ -1612,7 +1612,7 @@ static const struct file_operations iio_buffer_fileops = { .owner = THIS_MODULE, .llseek = noop_llseek, .unlocked_ioctl = iio_ioctl, - .compat_ioctl = iio_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static int iio_check_unique_scan_index(struct iio_dev *indio_dev) diff --git a/drivers/infiniband/core/uverbs_main.c b/drivers/infiniband/core/uverbs_main.c index af6dedabd6a5..d46ea235c216 100644 --- a/drivers/infiniband/core/uverbs_main.c +++ b/drivers/infiniband/core/uverbs_main.c @@ -1135,7 +1135,7 @@ static const struct file_operations uverbs_fops = { .release = ib_uverbs_close, .llseek = no_llseek, .unlocked_ioctl = ib_uverbs_ioctl, - .compat_ioctl = ib_uverbs_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static const struct file_operations uverbs_mmap_fops = { @@ -1146,7 +1146,7 @@ static const struct file_operations uverbs_mmap_fops = { .release = ib_uverbs_close, .llseek = no_llseek, .unlocked_ioctl = ib_uverbs_ioctl, - .compat_ioctl = ib_uverbs_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static int ib_uverbs_get_nl_info(struct ib_device *ibdev, void *client_data, diff --git a/drivers/media/rc/lirc_dev.c b/drivers/media/rc/lirc_dev.c index aeff7d7155fd..1d554bba567e 100644 --- a/drivers/media/rc/lirc_dev.c +++ b/drivers/media/rc/lirc_dev.c @@ -724,9 +724,7 @@ static const struct file_operations lirc_fops = { .owner = THIS_MODULE, .write = ir_lirc_transmit_ir, .unlocked_ioctl = ir_lirc_ioctl, -#ifdef CONFIG_COMPAT - .compat_ioctl = ir_lirc_ioctl, -#endif + .compat_ioctl = compat_ptr_ioctl, .read = ir_lirc_read, .poll = ir_lirc_poll, .open = ir_lirc_open, diff --git a/drivers/misc/vmw_vmci/vmci_host.c b/drivers/misc/vmw_vmci/vmci_host.c index 6eddb805642e..c7c98d50d9fe 100644 --- a/drivers/misc/vmw_vmci/vmci_host.c +++ b/drivers/misc/vmw_vmci/vmci_host.c @@ -968,7 +968,7 @@ static const struct file_operations vmuser_fops = { .release = vmci_host_close, .poll = vmci_host_poll, .unlocked_ioctl = vmci_host_unlocked_ioctl, - .compat_ioctl = vmci_host_unlocked_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static struct miscdevice vmci_host_miscdev = { diff --git a/drivers/nvdimm/bus.c b/drivers/nvdimm/bus.c index 56f189c3129a..4600f80fb250 100644 --- a/drivers/nvdimm/bus.c +++ b/drivers/nvdimm/bus.c @@ -1229,7 +1229,7 @@ static const struct file_operations nvdimm_bus_fops = { .owner = THIS_MODULE, .open = nd_open, .unlocked_ioctl = bus_ioctl, - .compat_ioctl = bus_ioctl, + .compat_ioctl = compat_ptr_ioctl, .llseek = noop_llseek, }; @@ -1237,7 +1237,7 @@ static const struct file_operations nvdimm_fops = { .owner = THIS_MODULE, .open = nd_open, .unlocked_ioctl = dimm_ioctl, - .compat_ioctl = dimm_ioctl, + .compat_ioctl = compat_ptr_ioctl, .llseek = noop_llseek, }; diff --git a/drivers/nvme/host/core.c b/drivers/nvme/host/core.c index 94c8ef4a54d3..d1a369c619dc 100644 --- a/drivers/nvme/host/core.c +++ b/drivers/nvme/host/core.c @@ -3062,7 +3062,7 @@ static const struct file_operations nvme_dev_fops = { .open = nvme_dev_open, .release = nvme_dev_release, .unlocked_ioctl = nvme_dev_ioctl, - .compat_ioctl = nvme_dev_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static ssize_t nvme_sysfs_reset(struct device *dev, diff --git a/drivers/pci/switch/switchtec.c b/drivers/pci/switch/switchtec.c index 0b8325235852..1b30d08e1867 100644 --- a/drivers/pci/switch/switchtec.c +++ b/drivers/pci/switch/switchtec.c @@ -1025,7 +1025,7 @@ static const struct file_operations switchtec_fops = { .read = switchtec_dev_read, .poll = switchtec_dev_poll, .unlocked_ioctl = switchtec_dev_ioctl, - .compat_ioctl = switchtec_dev_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static void link_event_work(struct work_struct *work) diff --git a/drivers/platform/x86/wmi.c b/drivers/platform/x86/wmi.c index 66cfc35e4e3d..2acdbf93088d 100644 --- a/drivers/platform/x86/wmi.c +++ b/drivers/platform/x86/wmi.c @@ -912,7 +912,7 @@ static const struct file_operations wmi_fops = { .read = wmi_char_read, .open = wmi_char_open, .unlocked_ioctl = wmi_ioctl, - .compat_ioctl = wmi_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static int wmi_dev_probe(struct device *dev) diff --git a/drivers/rpmsg/rpmsg_char.c b/drivers/rpmsg/rpmsg_char.c index ac50ed757765..c503829fb307 100644 --- a/drivers/rpmsg/rpmsg_char.c +++ b/drivers/rpmsg/rpmsg_char.c @@ -292,7 +292,7 @@ static const struct file_operations rpmsg_eptdev_fops = { .write_iter = rpmsg_eptdev_write_iter, .poll = rpmsg_eptdev_poll, .unlocked_ioctl = rpmsg_eptdev_ioctl, - .compat_ioctl = rpmsg_eptdev_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static ssize_t name_show(struct device *dev, struct device_attribute *attr, @@ -446,7 +446,7 @@ static const struct file_operations rpmsg_ctrldev_fops = { .open = rpmsg_ctrldev_open, .release = rpmsg_ctrldev_release, .unlocked_ioctl = rpmsg_ctrldev_ioctl, - .compat_ioctl = rpmsg_ctrldev_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static void rpmsg_ctrldev_release_device(struct device *dev) diff --git a/drivers/sbus/char/display7seg.c b/drivers/sbus/char/display7seg.c index 971fe074d7c9..fad936eb845f 100644 --- a/drivers/sbus/char/display7seg.c +++ b/drivers/sbus/char/display7seg.c @@ -156,7 +156,7 @@ static long d7s_ioctl(struct file *file, unsigned int cmd, unsigned long arg) static const struct file_operations d7s_fops = { .owner = THIS_MODULE, .unlocked_ioctl = d7s_ioctl, - .compat_ioctl = d7s_ioctl, + .compat_ioctl = compat_ptr_ioctl, .open = d7s_open, .release = d7s_release, .llseek = noop_llseek, diff --git a/drivers/sbus/char/envctrl.c b/drivers/sbus/char/envctrl.c index a63d5e402ff2..12d66aa61ede 100644 --- a/drivers/sbus/char/envctrl.c +++ b/drivers/sbus/char/envctrl.c @@ -715,9 +715,7 @@ static const struct file_operations envctrl_fops = { .owner = THIS_MODULE, .read = envctrl_read, .unlocked_ioctl = envctrl_ioctl, -#ifdef CONFIG_COMPAT - .compat_ioctl = envctrl_ioctl, -#endif + .compat_ioctl = compat_ptr_ioctl, .open = envctrl_open, .release = envctrl_release, .llseek = noop_llseek, diff --git a/drivers/scsi/3w-xxxx.c b/drivers/scsi/3w-xxxx.c index 75290aabd543..211a25351e7d 100644 --- a/drivers/scsi/3w-xxxx.c +++ b/drivers/scsi/3w-xxxx.c @@ -1049,9 +1049,7 @@ static int tw_chrdev_open(struct inode *inode, struct file *file) static const struct file_operations tw_fops = { .owner = THIS_MODULE, .unlocked_ioctl = tw_chrdev_ioctl, -#ifdef CONFIG_COMPAT - .compat_ioctl = tw_chrdev_ioctl, -#endif + .compat_ioctl = compat_ptr_ioctl, .open = tw_chrdev_open, .release = NULL, .llseek = noop_llseek, diff --git a/drivers/scsi/cxlflash/main.c b/drivers/scsi/cxlflash/main.c index 67d681c53c29..6f45074ee895 100644 --- a/drivers/scsi/cxlflash/main.c +++ b/drivers/scsi/cxlflash/main.c @@ -3593,7 +3593,7 @@ static const struct file_operations cxlflash_chr_fops = { .owner = THIS_MODULE, .open = cxlflash_chr_open, .unlocked_ioctl = cxlflash_chr_ioctl, - .compat_ioctl = cxlflash_chr_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; /** diff --git a/drivers/scsi/esas2r/esas2r_main.c b/drivers/scsi/esas2r/esas2r_main.c index fdbda5c05aa0..80c5a235d193 100644 --- a/drivers/scsi/esas2r/esas2r_main.c +++ b/drivers/scsi/esas2r/esas2r_main.c @@ -613,7 +613,7 @@ static int __init esas2r_init(void) /* Handle ioctl calls to "/proc/scsi/esas2r/ATTOnode" */ static const struct file_operations esas2r_proc_fops = { - .compat_ioctl = esas2r_proc_ioctl, + .compat_ioctl = compat_ptr_ioctl, .unlocked_ioctl = esas2r_proc_ioctl, }; diff --git a/drivers/scsi/pmcraid.c b/drivers/scsi/pmcraid.c index f95a970db8fd..4dc316ca5d28 100644 --- a/drivers/scsi/pmcraid.c +++ b/drivers/scsi/pmcraid.c @@ -3973,9 +3973,7 @@ static const struct file_operations pmcraid_fops = { .open = pmcraid_chr_open, .fasync = pmcraid_chr_fasync, .unlocked_ioctl = pmcraid_chr_ioctl, -#ifdef CONFIG_COMPAT - .compat_ioctl = pmcraid_chr_ioctl, -#endif + .compat_ioctl = compat_ptr_ioctl, .llseek = noop_llseek, }; diff --git a/drivers/staging/android/ion/ion.c b/drivers/staging/android/ion/ion.c index 61d2a8cbf894..268c4610c25a 100644 --- a/drivers/staging/android/ion/ion.c +++ b/drivers/staging/android/ion/ion.c @@ -221,9 +221,7 @@ static long ion_ioctl(struct file *filp, unsigned int cmd, unsigned long arg) static const struct file_operations ion_fops = { .owner = THIS_MODULE, .unlocked_ioctl = ion_ioctl, -#ifdef CONFIG_COMPAT - .compat_ioctl = ion_ioctl, -#endif + .compat_ioctl = compat_ptr_ioctl, }; static int debug_shrink_set(void *data, u64 val) diff --git a/drivers/staging/vme/devices/vme_user.c b/drivers/staging/vme/devices/vme_user.c index 6a33aaa1a49f..fd0ea4dbcb91 100644 --- a/drivers/staging/vme/devices/vme_user.c +++ b/drivers/staging/vme/devices/vme_user.c @@ -494,7 +494,7 @@ static const struct file_operations vme_user_fops = { .write = vme_user_write, .llseek = vme_user_llseek, .unlocked_ioctl = vme_user_unlocked_ioctl, - .compat_ioctl = vme_user_unlocked_ioctl, + .compat_ioctl = compat_ptr_ioctl, .mmap = vme_user_mmap, }; diff --git a/drivers/tee/tee_core.c b/drivers/tee/tee_core.c index 28cbe4613ed9..faa6ac8719cc 100644 --- a/drivers/tee/tee_core.c +++ b/drivers/tee/tee_core.c @@ -683,7 +683,7 @@ static const struct file_operations tee_fops = { .open = tee_open, .release = tee_release, .unlocked_ioctl = tee_ioctl, - .compat_ioctl = tee_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static void tee_release_device(struct device *dev) diff --git a/drivers/usb/class/cdc-wdm.c b/drivers/usb/class/cdc-wdm.c index 6afb941dd267..d2bbc9a43e7b 100644 --- a/drivers/usb/class/cdc-wdm.c +++ b/drivers/usb/class/cdc-wdm.c @@ -790,7 +790,7 @@ static const struct file_operations wdm_fops = { .release = wdm_release, .poll = wdm_poll, .unlocked_ioctl = wdm_ioctl, - .compat_ioctl = wdm_ioctl, + .compat_ioctl = compat_ptr_ioctl, .llseek = noop_llseek, }; diff --git a/drivers/usb/class/usbtmc.c b/drivers/usb/class/usbtmc.c index d8ed205e6b43..7ad47637f2a7 100644 --- a/drivers/usb/class/usbtmc.c +++ b/drivers/usb/class/usbtmc.c @@ -2241,9 +2241,7 @@ static const struct file_operations fops = { .release = usbtmc_release, .flush = usbtmc_flush, .unlocked_ioctl = usbtmc_ioctl, -#ifdef CONFIG_COMPAT - .compat_ioctl = usbtmc_ioctl, -#endif + .compat_ioctl = compat_ptr_ioctl, .fasync = usbtmc_fasync, .poll = usbtmc_poll, .llseek = default_llseek, diff --git a/drivers/virt/fsl_hypervisor.c b/drivers/virt/fsl_hypervisor.c index fb292f9cf29d..46ee0a0998b6 100644 --- a/drivers/virt/fsl_hypervisor.c +++ b/drivers/virt/fsl_hypervisor.c @@ -705,7 +705,7 @@ static const struct file_operations fsl_hv_fops = { .poll = fsl_hv_poll, .read = fsl_hv_read, .unlocked_ioctl = fsl_hv_ioctl, - .compat_ioctl = fsl_hv_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; static struct miscdevice fsl_hv_misc_dev = { diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c index e271f1f8c601..daa0c0467991 100644 --- a/fs/btrfs/super.c +++ b/fs/btrfs/super.c @@ -2322,7 +2322,7 @@ static const struct super_operations btrfs_super_ops = { static const struct file_operations btrfs_ctl_fops = { .open = btrfs_control_open, .unlocked_ioctl = btrfs_control_ioctl, - .compat_ioctl = btrfs_control_ioctl, + .compat_ioctl = compat_ptr_ioctl, .owner = THIS_MODULE, .llseek = noop_llseek, }; diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index 4b0e5a97f03c..c908eb46cbd0 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -2353,7 +2353,7 @@ const struct file_operations fuse_dev_operations = { .release = fuse_dev_release, .fasync = fuse_dev_fasync, .unlocked_ioctl = fuse_dev_ioctl, - .compat_ioctl = fuse_dev_ioctl, + .compat_ioctl = compat_ptr_ioctl, }; EXPORT_SYMBOL_GPL(fuse_dev_operations); diff --git a/fs/notify/fanotify/fanotify_user.c b/fs/notify/fanotify/fanotify_user.c index ec4eadf459ae..e774f7851bc0 100644 --- a/fs/notify/fanotify/fanotify_user.c +++ b/fs/notify/fanotify/fanotify_user.c @@ -523,7 +523,7 @@ static const struct file_operations fanotify_fops = { .fasync = NULL, .release = fanotify_release, .unlocked_ioctl = fanotify_ioctl, - .compat_ioctl = fanotify_ioctl, + .compat_ioctl = compat_ptr_ioctl, .llseek = noop_llseek, }; diff --git a/fs/userfaultfd.c b/fs/userfaultfd.c index de1203b48bb9..69321d466379 100644 --- a/fs/userfaultfd.c +++ b/fs/userfaultfd.c @@ -2017,7 +2017,7 @@ static const struct file_operations userfaultfd_fops = { .poll = userfaultfd_poll, .read = userfaultfd_read, .unlocked_ioctl = userfaultfd_ioctl, - .compat_ioctl = userfaultfd_ioctl, + .compat_ioctl = compat_ptr_ioctl, .llseek = noop_llseek, }; diff --git a/net/rfkill/core.c b/net/rfkill/core.c index 5bba7c36ac74..97101c55763d 100644 --- a/net/rfkill/core.c +++ b/net/rfkill/core.c @@ -1317,7 +1317,7 @@ static const struct file_operations rfkill_fops = { .release = rfkill_fop_release, #ifdef CONFIG_RFKILL_INPUT .unlocked_ioctl = rfkill_fop_ioctl, - .compat_ioctl = rfkill_fop_ioctl, + .compat_ioctl = compat_ptr_ioctl, #endif .llseek = no_llseek, }; From e3b7b58b0a87cdf8ab04fdde27722451604c03ba Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 21 Oct 2025 21:03:19 -0400 Subject: [PATCH 145/148] BACKPORT: fuse: fix livelock in synchronous file put from fuseblk workers [ Upstream commit 26e5c67deb2e1f42a951f022fdf5b9f7eb747b01 ] I observed a hang when running generic/323 against a fuseblk server. This test opens a file, initiates a lot of AIO writes to that file descriptor, and closes the file descriptor before the writes complete. Unsurprisingly, the AIO exerciser threads are mostly stuck waiting for responses from the fuseblk server: [<0>] request_wait_answer+0x1fe/0x2a0 [fuse] [<0>] __fuse_simple_request+0xd3/0x2b0 [fuse] [<0>] fuse_do_getattr+0xfc/0x1f0 [fuse] [<0>] fuse_file_read_iter+0xbe/0x1c0 [fuse] [<0>] aio_read+0x130/0x1e0 [<0>] io_submit_one+0x542/0x860 [<0>] __x64_sys_io_submit+0x98/0x1a0 [<0>] do_syscall_64+0x37/0xf0 [<0>] entry_SYSCALL_64_after_hwframe+0x4b/0x53 But the /weird/ part is that the fuseblk server threads are waiting for responses from itself: [<0>] request_wait_answer+0x1fe/0x2a0 [fuse] [<0>] __fuse_simple_request+0xd3/0x2b0 [fuse] [<0>] fuse_file_put+0x9a/0xd0 [fuse] [<0>] fuse_release+0x36/0x50 [fuse] [<0>] __fput+0xec/0x2b0 [<0>] task_work_run+0x55/0x90 [<0>] syscall_exit_to_user_mode+0xe9/0x100 [<0>] do_syscall_64+0x43/0xf0 [<0>] entry_SYSCALL_64_after_hwframe+0x4b/0x53 The fuseblk server is fuse2fs so there's nothing all that exciting in the server itself. So why is the fuse server calling fuse_file_put? The commit message for the fstest sheds some light on that: "By closing the file descriptor before calling io_destroy, you pretty much guarantee that the last put on the ioctx will be done in interrupt context (during I/O completion). Aha. AIO fgets a new struct file from the fd when it queues the ioctx. The completion of the FUSE_WRITE command from userspace causes the fuse server to call the AIO completion function. The completion puts the struct file, queuing a delayed fput to the fuse server task. When the fuse server task returns to userspace, it has to run the delayed fput, which in the case of a fuseblk server, it does synchronously. Sending the FUSE_RELEASE command sychronously from fuse server threads is a bad idea because a client program can initiate enough simultaneous AIOs such that all the fuse server threads end up in delayed_fput, and now there aren't any threads left to handle the queued fuse commands. Fix this by only using asynchronous fputs when closing files, and leave a comment explaining why. Change-Id: I78be67f615f1318447e2fdad9ebfb34e60857b95 Cc: stable@vger.kernel.org # v2.6.38 Fixes: 5a18ec176c934c ("fuse: fix hang of single threaded fuseblk filesystem") Signed-off-by: Darrick J. Wong Signed-off-by: Miklos Szeredi [ added isdir parameter to fuse_file_put() call ] Signed-off-by: Sasha Levin Signed-off-by: Greg Kroah-Hartman --- fs/fuse/file.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index d96daeaa807b..cb34aa80c4f8 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -359,8 +359,14 @@ void fuse_release_common(struct file *file, bool isdir) * Make the release synchronous if this is a fuseblk mount, * synchronous RELEASE is allowed (and desirable) in this case * because the server can be trusted not to screw up. + * + * Always use the asynchronous file put because the current thread + * might be the fuse server. This can happen if a process starts some + * aio and closes the fd before the aio completes. Since aio takes its + * own ref to the file, the IO completion has to drop the ref, which is + * how the fuse server can end up closing its clients' files. */ - fuse_file_put(ra->inode, ff, ff->fm->fc->destroy, isdir); + fuse_file_put(ra->inode, ff, false, isdir); } static int fuse_open(struct inode *inode, struct file *file) From 4884b8ae4c1103933d2310c7b1dccf100145f5f8 Mon Sep 17 00:00:00 2001 From: Sandeep Dhavale Date: Tue, 14 Apr 2026 15:07:58 -0700 Subject: [PATCH 146/148] ANDROID: fuse: Open-code vma_set_file() logic in fuse_backing_mmap() Manual manipulation of vma->vm_file and file reference counts in fuse_backing_mmap() is error-prone. Open-code the logic of vma_set_file() using swap() to safely transfer the file reference and ensure proper reference counting during VMA setup, as vma_set_file() is not available in this kernel version. Bug: 498749316 Signed-off-by: Sandeep Dhavale Cherrypick-From: https://android-review.googlesource.com/q/commit:5c3d7b8761264e2a2deb121020bb933273b240ba Merged-In: I22ba0618f9fde6b698aeeeff2005830c4392969c Change-Id: I22ba0618f9fde6b698aeeeff2005830c4392969c [dhavale: as mentioned in updated commit message, this patch is adjusted to account for lack of vma_set_file() helper ] --- fs/fuse/backing.c | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index e9533afd5961..013a03259f0c 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -1064,14 +1064,12 @@ ssize_t fuse_backing_mmap(struct file *file, struct vm_area_struct *vma) if (WARN_ON(file != vma->vm_file)) return -EIO; - vma->vm_file = get_file(backing_file); - + get_file(backing_file); + swap(vma->vm_file, backing_file); + fput(backing_file); ret = call_mmap(vma->vm_file, vma); - if (ret) - fput(backing_file); - else - fput(file); + return ret; if (file->f_flags & O_NOATIME) return ret; From 63de6916f039a64d1aaec23ec9238d7ff4c5fb7c Mon Sep 17 00:00:00 2001 From: Alexander Martinz Date: Tue, 12 May 2026 12:10:40 +0200 Subject: [PATCH 147/148] Reapply "UPSTREAM: seccomp: Remove bogus __user annotations" This reverts commit cb8dc8a108d69ce584fd1f44c5a738c4e11dcb49. As pointed out on gerrit[1]: > This revert is wrong. `android12-5.4` doesn't have > "sysctl: pass kernel pointers to ->proc_handler" but this kernel does. [1] - https://review.lineageos.org/c/LineageOS/android_kernel_qcom_sm8350/+/482457 Change-Id: I72cd996f6739a67cf78141843e2d933f19e2e92b Signed-off-by: Alexander Martinz --- kernel/seccomp.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/kernel/seccomp.c b/kernel/seccomp.c index 455330c8b9ba..1ec9bdc1e208 100644 --- a/kernel/seccomp.c +++ b/kernel/seccomp.c @@ -2035,7 +2035,7 @@ static bool seccomp_actions_logged_from_names(u32 *actions_logged, char *names) return true; } -static int read_actions_logged(struct ctl_table *ro_table, void __user *buffer, +static int read_actions_logged(struct ctl_table *ro_table, void *buffer, size_t *lenp, loff_t *ppos) { char names[sizeof(seccomp_actions_avail)]; @@ -2053,7 +2053,7 @@ static int read_actions_logged(struct ctl_table *ro_table, void __user *buffer, return proc_dostring(&table, 0, buffer, lenp, ppos); } -static int write_actions_logged(struct ctl_table *ro_table, void __user *buffer, +static int write_actions_logged(struct ctl_table *ro_table, void *buffer, size_t *lenp, loff_t *ppos, u32 *actions_logged) { char names[sizeof(seccomp_actions_avail)]; From 8e41e2a7244c2052d1a70251787dbbb37d3abe6a Mon Sep 17 00:00:00 2001 From: Daniel Rosenberg Date: Mon, 11 May 2026 22:24:01 +0800 Subject: [PATCH 148/148] ANDROID: Add FUSE_BPF to gki_defconfig Bug: 202785178 Test: test_fuse passes on linux, feature works on cuttlefish Signed-off-by: Paul Lawrence Signed-off-by: Daniel Rosenberg Change-Id: If5d56aa5ff5bf5ee660073ac8f1ef1573a74cfd1 --- arch/arm64/configs/gki_defconfig | 1 + arch/x86/configs/gki_defconfig | 1 + 2 files changed, 2 insertions(+) diff --git a/arch/arm64/configs/gki_defconfig b/arch/arm64/configs/gki_defconfig index 7b1714aaf30d..64beb2922cf9 100644 --- a/arch/arm64/configs/gki_defconfig +++ b/arch/arm64/configs/gki_defconfig @@ -513,6 +513,7 @@ CONFIG_FS_VERITY_BUILTIN_SIGNATURES=y CONFIG_QUOTA=y CONFIG_QFMT_V2=y CONFIG_FUSE_FS=y +CONFIG_FUSE_BPF=y CONFIG_OVERLAY_FS=y CONFIG_MSDOS_FS=y CONFIG_VFAT_FS=y diff --git a/arch/x86/configs/gki_defconfig b/arch/x86/configs/gki_defconfig index 5f88798bb496..ebab1f1901d8 100644 --- a/arch/x86/configs/gki_defconfig +++ b/arch/x86/configs/gki_defconfig @@ -442,6 +442,7 @@ CONFIG_FS_VERITY_BUILTIN_SIGNATURES=y CONFIG_QUOTA=y CONFIG_QFMT_V2=y CONFIG_FUSE_FS=y +CONFIG_FUSE_BPF=y CONFIG_OVERLAY_FS=y CONFIG_MSDOS_FS=y CONFIG_VFAT_FS=y