2005-09-09 13:10:27 -07:00
|
|
|
/*
|
|
|
|
FUSE: Filesystem in Userspace
|
2008-11-26 12:03:54 +01:00
|
|
|
Copyright (C) 2001-2008 Miklos Szeredi <miklos@szeredi.hu>
|
2005-09-09 13:10:27 -07:00
|
|
|
|
|
|
|
This program can be distributed under the terms of the GNU GPL.
|
|
|
|
See the file COPYING.
|
|
|
|
*/
|
|
|
|
|
|
|
|
#include "fuse_i.h"
|
|
|
|
|
|
|
|
#include <linux/init.h>
|
|
|
|
#include <linux/module.h>
|
|
|
|
#include <linux/poll.h>
|
|
|
|
#include <linux/uio.h>
|
|
|
|
#include <linux/miscdevice.h>
|
|
|
|
#include <linux/pagemap.h>
|
|
|
|
#include <linux/file.h>
|
|
|
|
#include <linux/slab.h>
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
#include <linux/pipe_fs_i.h>
|
2010-05-25 15:06:07 +02:00
|
|
|
#include <linux/swap.h>
|
|
|
|
#include <linux/splice.h>
|
2013-05-07 16:19:08 -07:00
|
|
|
#include <linux/aio.h>
|
2005-09-09 13:10:27 -07:00
|
|
|
|
|
|
|
MODULE_ALIAS_MISCDEV(FUSE_MINOR);
|
driver core: add devname module aliases to allow module on-demand auto-loading
This adds:
alias: devname:<name>
to some common kernel modules, which will allow the on-demand loading
of the kernel module when the device node is accessed.
Ideally all these modules would be compiled-in, but distros seems too
much in love with their modularization that we need to cover the common
cases with this new facility. It will allow us to remove a bunch of pretty
useless init scripts and modprobes from init scripts.
The static device node aliases will be carried in the module itself. The
program depmod will extract this information to a file in the module directory:
$ cat /lib/modules/2.6.34-00650-g537b60d-dirty/modules.devname
# Device nodes to trigger on-demand module loading.
microcode cpu/microcode c10:184
fuse fuse c10:229
ppp_generic ppp c108:0
tun net/tun c10:200
dm_mod mapper/control c10:235
Udev will pick up the depmod created file on startup and create all the
static device nodes which the kernel modules specify, so that these modules
get automatically loaded when the device node is accessed:
$ /sbin/udevd --debug
...
static_dev_create_from_modules: mknod '/dev/cpu/microcode' c10:184
static_dev_create_from_modules: mknod '/dev/fuse' c10:229
static_dev_create_from_modules: mknod '/dev/ppp' c108:0
static_dev_create_from_modules: mknod '/dev/net/tun' c10:200
static_dev_create_from_modules: mknod '/dev/mapper/control' c10:235
udev_rules_apply_static_dev_perms: chmod '/dev/net/tun' 0666
udev_rules_apply_static_dev_perms: chmod '/dev/fuse' 0666
A few device nodes are switched to statically allocated numbers, to allow
the static nodes to work. This might also useful for systems which still run
a plain static /dev, which is completely unsafe to use with any dynamic minor
numbers.
Note:
The devname aliases must be limited to the *common* and *single*instance*
device nodes, like the misc devices, and never be used for conceptually limited
systems like the loop devices, which should rather get fixed properly and get a
control node for losetup to talk to, instead of creating a random number of
device nodes in advance, regardless if they are ever used.
This facility is to hide the mess distros are creating with too modualized
kernels, and just to hide that these modules are not compiled-in, and not to
paper-over broken concepts. Thanks! :)
Cc: Greg Kroah-Hartman <gregkh@suse.de>
Cc: David S. Miller <davem@davemloft.net>
Cc: Miklos Szeredi <miklos@szeredi.hu>
Cc: Chris Mason <chris.mason@oracle.com>
Cc: Alasdair G Kergon <agk@redhat.com>
Cc: Tigran Aivazian <tigran@aivazian.fsnet.co.uk>
Cc: Ian Kent <raven@themaw.net>
Signed-Off-By: Kay Sievers <kay.sievers@vrfy.org>
Signed-off-by: Greg Kroah-Hartman <gregkh@suse.de>
2010-05-20 18:07:20 +02:00
|
|
|
MODULE_ALIAS("devname:fuse");
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2006-12-06 20:33:20 -08:00
|
|
|
static struct kmem_cache *fuse_req_cachep;
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2006-01-16 22:14:28 -08:00
|
|
|
static struct fuse_conn *fuse_get_conn(struct file *file)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
2006-04-10 22:54:55 -07:00
|
|
|
/*
|
|
|
|
* Lockless access is OK, because file->private data is set
|
|
|
|
* once during mount and is valid until the file is released.
|
|
|
|
*/
|
|
|
|
return file->private_data;
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
|
2012-10-26 19:48:07 +04:00
|
|
|
static void fuse_request_init(struct fuse_req *req, struct page **pages,
|
2012-10-26 19:49:24 +04:00
|
|
|
struct fuse_page_desc *page_descs,
|
2012-10-26 19:48:07 +04:00
|
|
|
unsigned npages)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
memset(req, 0, sizeof(*req));
|
2012-10-26 19:48:07 +04:00
|
|
|
memset(pages, 0, sizeof(*pages) * npages);
|
2012-10-26 19:49:24 +04:00
|
|
|
memset(page_descs, 0, sizeof(*page_descs) * npages);
|
2005-09-09 13:10:27 -07:00
|
|
|
INIT_LIST_HEAD(&req->list);
|
2006-06-25 05:48:54 -07:00
|
|
|
INIT_LIST_HEAD(&req->intr_entry);
|
2005-09-09 13:10:27 -07:00
|
|
|
init_waitqueue_head(&req->waitq);
|
|
|
|
atomic_set(&req->count, 1);
|
2012-10-26 19:48:07 +04:00
|
|
|
req->pages = pages;
|
2012-10-26 19:49:24 +04:00
|
|
|
req->page_descs = page_descs;
|
2012-10-26 19:48:07 +04:00
|
|
|
req->max_pages = npages;
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
|
2012-10-26 19:48:07 +04:00
|
|
|
static struct fuse_req *__fuse_request_alloc(unsigned npages, gfp_t flags)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
2012-10-26 19:48:07 +04:00
|
|
|
struct fuse_req *req = kmem_cache_alloc(fuse_req_cachep, flags);
|
|
|
|
if (req) {
|
|
|
|
struct page **pages;
|
2012-10-26 19:49:24 +04:00
|
|
|
struct fuse_page_desc *page_descs;
|
2012-10-26 19:48:07 +04:00
|
|
|
|
2012-10-26 19:49:24 +04:00
|
|
|
if (npages <= FUSE_REQ_INLINE_PAGES) {
|
2012-10-26 19:48:07 +04:00
|
|
|
pages = req->inline_pages;
|
2012-10-26 19:49:24 +04:00
|
|
|
page_descs = req->inline_page_descs;
|
|
|
|
} else {
|
2012-10-26 19:48:07 +04:00
|
|
|
pages = kmalloc(sizeof(struct page *) * npages, flags);
|
2012-10-26 19:49:24 +04:00
|
|
|
page_descs = kmalloc(sizeof(struct fuse_page_desc) *
|
|
|
|
npages, flags);
|
|
|
|
}
|
2012-10-26 19:48:07 +04:00
|
|
|
|
2012-10-26 19:49:24 +04:00
|
|
|
if (!pages || !page_descs) {
|
|
|
|
kfree(pages);
|
|
|
|
kfree(page_descs);
|
2012-10-26 19:48:07 +04:00
|
|
|
kmem_cache_free(fuse_req_cachep, req);
|
|
|
|
return NULL;
|
|
|
|
}
|
|
|
|
|
2012-10-26 19:49:24 +04:00
|
|
|
fuse_request_init(req, pages, page_descs, npages);
|
2012-10-26 19:48:07 +04:00
|
|
|
}
|
2005-09-09 13:10:27 -07:00
|
|
|
return req;
|
|
|
|
}
|
2012-10-26 19:48:07 +04:00
|
|
|
|
|
|
|
struct fuse_req *fuse_request_alloc(unsigned npages)
|
|
|
|
{
|
|
|
|
return __fuse_request_alloc(npages, GFP_KERNEL);
|
|
|
|
}
|
2009-04-14 10:54:53 +09:00
|
|
|
EXPORT_SYMBOL_GPL(fuse_request_alloc);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2012-10-26 19:48:07 +04:00
|
|
|
struct fuse_req *fuse_request_alloc_nofs(unsigned npages)
|
fuse: support writable mmap
Quoting Linus (3 years ago, FUSE inclusion discussions):
"User-space filesystems are hard to get right. I'd claim that they
are almost impossible, unless you limit them somehow (shared
writable mappings are the nastiest part - if you don't have those,
you can reasonably limit your problems by limiting the number of
dirty pages you accept through normal "write()" calls)."
Instead of attempting the impossible, I've just waited for the dirty page
accounting infrastructure to materialize (thanks to Peter Zijlstra and
others). This nicely solved the biggest problem: limiting the number of pages
used for write caching.
Some small details remained, however, which this largish patch attempts to
address. It provides a page writeback implementation for fuse, which is
completely safe against VM related deadlocks. Performance may not be very
good for certain usage patterns, but generally it should be acceptable.
It has been tested extensively with fsx-linux and bash-shared-mapping.
Fuse page writeback design
--------------------------
fuse_writepage() allocates a new temporary page with GFP_NOFS|__GFP_HIGHMEM.
It copies the contents of the original page, and queues a WRITE request to the
userspace filesystem using this temp page.
The writeback is finished instantly from the MM's point of view: the page is
removed from the radix trees, and the PageDirty and PageWriteback flags are
cleared.
For the duration of the actual write, the NR_WRITEBACK_TEMP counter is
incremented. The per-bdi writeback count is not decremented until the actual
write completes.
On dirtying the page, fuse waits for a previous write to finish before
proceeding. This makes sure, there can only be one temporary page used at a
time for one cached page.
This approach is wasteful in both memory and CPU bandwidth, so why is this
complication needed?
The basic problem is that there can be no guarantee about the time in which
the userspace filesystem will complete a write. It may be buggy or even
malicious, and fail to complete WRITE requests. We don't want unrelated parts
of the system to grind to a halt in such cases.
Also a filesystem may need additional resources (particularly memory) to
complete a WRITE request. There's a great danger of a deadlock if that
allocation may wait for the writepage to finish.
Currently there are several cases where the kernel can block on page
writeback:
- allocation order is larger than PAGE_ALLOC_COSTLY_ORDER
- page migration
- throttle_vm_writeout (through NR_WRITEBACK)
- sync(2)
Of course in some cases (fsync, msync) we explicitly want to allow blocking.
So for these cases new code has to be added to fuse, since the VM is not
tracking writeback pages for us any more.
As an extra safetly measure, the maximum dirty ratio allocated to a single
fuse filesystem is set to 1% by default. This way one (or several) buggy or
malicious fuse filesystems cannot slow down the rest of the system by hogging
dirty memory.
With appropriate privileges, this limit can be raised through
'/sys/class/bdi/<bdi>/max_ratio'.
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
Cc: Peter Zijlstra <a.p.zijlstra@chello.nl>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
2008-04-30 00:54:41 -07:00
|
|
|
{
|
2012-10-26 19:48:07 +04:00
|
|
|
return __fuse_request_alloc(npages, GFP_NOFS);
|
fuse: support writable mmap
Quoting Linus (3 years ago, FUSE inclusion discussions):
"User-space filesystems are hard to get right. I'd claim that they
are almost impossible, unless you limit them somehow (shared
writable mappings are the nastiest part - if you don't have those,
you can reasonably limit your problems by limiting the number of
dirty pages you accept through normal "write()" calls)."
Instead of attempting the impossible, I've just waited for the dirty page
accounting infrastructure to materialize (thanks to Peter Zijlstra and
others). This nicely solved the biggest problem: limiting the number of pages
used for write caching.
Some small details remained, however, which this largish patch attempts to
address. It provides a page writeback implementation for fuse, which is
completely safe against VM related deadlocks. Performance may not be very
good for certain usage patterns, but generally it should be acceptable.
It has been tested extensively with fsx-linux and bash-shared-mapping.
Fuse page writeback design
--------------------------
fuse_writepage() allocates a new temporary page with GFP_NOFS|__GFP_HIGHMEM.
It copies the contents of the original page, and queues a WRITE request to the
userspace filesystem using this temp page.
The writeback is finished instantly from the MM's point of view: the page is
removed from the radix trees, and the PageDirty and PageWriteback flags are
cleared.
For the duration of the actual write, the NR_WRITEBACK_TEMP counter is
incremented. The per-bdi writeback count is not decremented until the actual
write completes.
On dirtying the page, fuse waits for a previous write to finish before
proceeding. This makes sure, there can only be one temporary page used at a
time for one cached page.
This approach is wasteful in both memory and CPU bandwidth, so why is this
complication needed?
The basic problem is that there can be no guarantee about the time in which
the userspace filesystem will complete a write. It may be buggy or even
malicious, and fail to complete WRITE requests. We don't want unrelated parts
of the system to grind to a halt in such cases.
Also a filesystem may need additional resources (particularly memory) to
complete a WRITE request. There's a great danger of a deadlock if that
allocation may wait for the writepage to finish.
Currently there are several cases where the kernel can block on page
writeback:
- allocation order is larger than PAGE_ALLOC_COSTLY_ORDER
- page migration
- throttle_vm_writeout (through NR_WRITEBACK)
- sync(2)
Of course in some cases (fsync, msync) we explicitly want to allow blocking.
So for these cases new code has to be added to fuse, since the VM is not
tracking writeback pages for us any more.
As an extra safetly measure, the maximum dirty ratio allocated to a single
fuse filesystem is set to 1% by default. This way one (or several) buggy or
malicious fuse filesystems cannot slow down the rest of the system by hogging
dirty memory.
With appropriate privileges, this limit can be raised through
'/sys/class/bdi/<bdi>/max_ratio'.
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
Cc: Peter Zijlstra <a.p.zijlstra@chello.nl>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
2008-04-30 00:54:41 -07:00
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
void fuse_request_free(struct fuse_req *req)
|
|
|
|
{
|
2012-10-26 19:49:24 +04:00
|
|
|
if (req->pages != req->inline_pages) {
|
2012-10-26 19:48:07 +04:00
|
|
|
kfree(req->pages);
|
2012-10-26 19:49:24 +04:00
|
|
|
kfree(req->page_descs);
|
|
|
|
}
|
2005-09-09 13:10:27 -07:00
|
|
|
kmem_cache_free(fuse_req_cachep, req);
|
|
|
|
}
|
|
|
|
|
2006-01-16 22:14:28 -08:00
|
|
|
static void block_sigs(sigset_t *oldset)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
sigset_t mask;
|
|
|
|
|
|
|
|
siginitsetinv(&mask, sigmask(SIGKILL));
|
|
|
|
sigprocmask(SIG_BLOCK, &mask, oldset);
|
|
|
|
}
|
|
|
|
|
2006-01-16 22:14:28 -08:00
|
|
|
static void restore_sigs(sigset_t *oldset)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
sigprocmask(SIG_SETMASK, oldset, NULL);
|
|
|
|
}
|
|
|
|
|
2012-12-14 19:20:51 +04:00
|
|
|
void __fuse_get_request(struct fuse_req *req)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
atomic_inc(&req->count);
|
|
|
|
}
|
|
|
|
|
|
|
|
/* Must be called with > 1 refcount */
|
|
|
|
static void __fuse_put_request(struct fuse_req *req)
|
|
|
|
{
|
|
|
|
BUG_ON(atomic_read(&req->count) < 2);
|
|
|
|
atomic_dec(&req->count);
|
|
|
|
}
|
|
|
|
|
2006-06-25 05:48:52 -07:00
|
|
|
static void fuse_req_init_context(struct fuse_req *req)
|
|
|
|
{
|
2012-02-07 16:26:03 -08:00
|
|
|
req->in.h.uid = from_kuid_munged(&init_user_ns, current_fsuid());
|
|
|
|
req->in.h.gid = from_kgid_munged(&init_user_ns, current_fsgid());
|
2006-06-25 05:48:52 -07:00
|
|
|
req->in.h.pid = current->pid;
|
|
|
|
}
|
|
|
|
|
2013-03-21 18:02:28 +04:00
|
|
|
static bool fuse_block_alloc(struct fuse_conn *fc, bool for_background)
|
|
|
|
{
|
|
|
|
return !fc->initialized || (for_background && fc->blocked);
|
|
|
|
}
|
|
|
|
|
2013-03-21 18:02:04 +04:00
|
|
|
static struct fuse_req *__fuse_get_req(struct fuse_conn *fc, unsigned npages,
|
|
|
|
bool for_background)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
2006-04-10 22:54:59 -07:00
|
|
|
struct fuse_req *req;
|
|
|
|
int err;
|
2006-04-11 21:16:09 +02:00
|
|
|
atomic_inc(&fc->num_waiting);
|
2013-03-21 18:02:28 +04:00
|
|
|
|
|
|
|
if (fuse_block_alloc(fc, for_background)) {
|
|
|
|
sigset_t oldset;
|
|
|
|
int intr;
|
|
|
|
|
|
|
|
block_sigs(&oldset);
|
2013-03-21 18:02:36 +04:00
|
|
|
intr = wait_event_interruptible_exclusive(fc->blocked_waitq,
|
2013-03-21 18:02:28 +04:00
|
|
|
!fuse_block_alloc(fc, for_background));
|
|
|
|
restore_sigs(&oldset);
|
|
|
|
err = -EINTR;
|
|
|
|
if (intr)
|
|
|
|
goto out;
|
|
|
|
}
|
2006-04-10 22:54:59 -07:00
|
|
|
|
2006-06-25 05:48:50 -07:00
|
|
|
err = -ENOTCONN;
|
|
|
|
if (!fc->connected)
|
|
|
|
goto out;
|
|
|
|
|
2012-10-26 19:48:30 +04:00
|
|
|
req = fuse_request_alloc(npages);
|
2006-04-11 21:16:09 +02:00
|
|
|
err = -ENOMEM;
|
2013-03-21 18:02:36 +04:00
|
|
|
if (!req) {
|
|
|
|
if (for_background)
|
|
|
|
wake_up(&fc->blocked_waitq);
|
2006-04-11 21:16:09 +02:00
|
|
|
goto out;
|
2013-03-21 18:02:36 +04:00
|
|
|
}
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2006-06-25 05:48:52 -07:00
|
|
|
fuse_req_init_context(req);
|
2006-04-11 21:16:09 +02:00
|
|
|
req->waiting = 1;
|
2013-03-21 18:02:04 +04:00
|
|
|
req->background = for_background;
|
2005-09-09 13:10:27 -07:00
|
|
|
return req;
|
2006-04-11 21:16:09 +02:00
|
|
|
|
|
|
|
out:
|
|
|
|
atomic_dec(&fc->num_waiting);
|
|
|
|
return ERR_PTR(err);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
2013-03-21 18:02:04 +04:00
|
|
|
|
|
|
|
struct fuse_req *fuse_get_req(struct fuse_conn *fc, unsigned npages)
|
|
|
|
{
|
|
|
|
return __fuse_get_req(fc, npages, false);
|
|
|
|
}
|
2009-04-14 10:54:53 +09:00
|
|
|
EXPORT_SYMBOL_GPL(fuse_get_req);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2013-03-21 18:02:04 +04:00
|
|
|
struct fuse_req *fuse_get_req_for_background(struct fuse_conn *fc,
|
|
|
|
unsigned npages)
|
|
|
|
{
|
|
|
|
return __fuse_get_req(fc, npages, true);
|
|
|
|
}
|
|
|
|
EXPORT_SYMBOL_GPL(fuse_get_req_for_background);
|
|
|
|
|
2006-06-25 05:48:52 -07:00
|
|
|
/*
|
|
|
|
* Return request in fuse_file->reserved_req. However that may
|
|
|
|
* currently be in use. If that is the case, wait for it to become
|
|
|
|
* available.
|
|
|
|
*/
|
|
|
|
static struct fuse_req *get_reserved_req(struct fuse_conn *fc,
|
|
|
|
struct file *file)
|
|
|
|
{
|
|
|
|
struct fuse_req *req = NULL;
|
|
|
|
struct fuse_file *ff = file->private_data;
|
|
|
|
|
|
|
|
do {
|
2007-10-16 23:31:00 -07:00
|
|
|
wait_event(fc->reserved_req_waitq, ff->reserved_req);
|
2006-06-25 05:48:52 -07:00
|
|
|
spin_lock(&fc->lock);
|
|
|
|
if (ff->reserved_req) {
|
|
|
|
req = ff->reserved_req;
|
|
|
|
ff->reserved_req = NULL;
|
2012-08-27 14:48:26 -04:00
|
|
|
req->stolen_file = get_file(file);
|
2006-06-25 05:48:52 -07:00
|
|
|
}
|
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
} while (!req);
|
|
|
|
|
|
|
|
return req;
|
|
|
|
}
|
|
|
|
|
|
|
|
/*
|
|
|
|
* Put stolen request back into fuse_file->reserved_req
|
|
|
|
*/
|
|
|
|
static void put_reserved_req(struct fuse_conn *fc, struct fuse_req *req)
|
|
|
|
{
|
|
|
|
struct file *file = req->stolen_file;
|
|
|
|
struct fuse_file *ff = file->private_data;
|
|
|
|
|
|
|
|
spin_lock(&fc->lock);
|
2012-10-26 19:49:24 +04:00
|
|
|
fuse_request_init(req, req->pages, req->page_descs, req->max_pages);
|
2006-06-25 05:48:52 -07:00
|
|
|
BUG_ON(ff->reserved_req);
|
|
|
|
ff->reserved_req = req;
|
2007-10-16 23:31:00 -07:00
|
|
|
wake_up_all(&fc->reserved_req_waitq);
|
2006-06-25 05:48:52 -07:00
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
fput(file);
|
|
|
|
}
|
|
|
|
|
|
|
|
/*
|
|
|
|
* Gets a requests for a file operation, always succeeds
|
|
|
|
*
|
|
|
|
* This is used for sending the FLUSH request, which must get to
|
|
|
|
* userspace, due to POSIX locks which may need to be unlocked.
|
|
|
|
*
|
|
|
|
* If allocation fails due to OOM, use the reserved request in
|
|
|
|
* fuse_file.
|
|
|
|
*
|
|
|
|
* This is very unlikely to deadlock accidentally, since the
|
|
|
|
* filesystem should not have it's own file open. If deadlock is
|
|
|
|
* intentional, it can still be broken by "aborting" the filesystem.
|
|
|
|
*/
|
2012-10-26 19:48:30 +04:00
|
|
|
struct fuse_req *fuse_get_req_nofail_nopages(struct fuse_conn *fc,
|
|
|
|
struct file *file)
|
2006-06-25 05:48:52 -07:00
|
|
|
{
|
|
|
|
struct fuse_req *req;
|
|
|
|
|
|
|
|
atomic_inc(&fc->num_waiting);
|
2013-03-21 18:02:28 +04:00
|
|
|
wait_event(fc->blocked_waitq, fc->initialized);
|
2012-10-26 19:48:30 +04:00
|
|
|
req = fuse_request_alloc(0);
|
2006-06-25 05:48:52 -07:00
|
|
|
if (!req)
|
|
|
|
req = get_reserved_req(fc, file);
|
|
|
|
|
|
|
|
fuse_req_init_context(req);
|
|
|
|
req->waiting = 1;
|
2013-03-21 18:02:04 +04:00
|
|
|
req->background = 0;
|
2006-06-25 05:48:52 -07:00
|
|
|
return req;
|
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
void fuse_put_request(struct fuse_conn *fc, struct fuse_req *req)
|
2006-02-04 23:27:40 -08:00
|
|
|
{
|
|
|
|
if (atomic_dec_and_test(&req->count)) {
|
2013-03-21 18:02:36 +04:00
|
|
|
if (unlikely(req->background)) {
|
|
|
|
/*
|
|
|
|
* We get here in the unlikely case that a background
|
|
|
|
* request was allocated but not sent
|
|
|
|
*/
|
|
|
|
spin_lock(&fc->lock);
|
|
|
|
if (!fc->blocked)
|
|
|
|
wake_up(&fc->blocked_waitq);
|
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
}
|
|
|
|
|
2006-04-11 21:16:09 +02:00
|
|
|
if (req->waiting)
|
|
|
|
atomic_dec(&fc->num_waiting);
|
2006-06-25 05:48:52 -07:00
|
|
|
|
|
|
|
if (req->stolen_file)
|
|
|
|
put_reserved_req(fc, req);
|
|
|
|
else
|
|
|
|
fuse_request_free(req);
|
2006-02-04 23:27:40 -08:00
|
|
|
}
|
|
|
|
}
|
2009-04-14 10:54:53 +09:00
|
|
|
EXPORT_SYMBOL_GPL(fuse_put_request);
|
2006-02-04 23:27:40 -08:00
|
|
|
|
2008-02-06 01:38:39 -08:00
|
|
|
static unsigned len_args(unsigned numargs, struct fuse_arg *args)
|
|
|
|
{
|
|
|
|
unsigned nbytes = 0;
|
|
|
|
unsigned i;
|
|
|
|
|
|
|
|
for (i = 0; i < numargs; i++)
|
|
|
|
nbytes += args[i].size;
|
|
|
|
|
|
|
|
return nbytes;
|
|
|
|
}
|
|
|
|
|
|
|
|
static u64 fuse_get_unique(struct fuse_conn *fc)
|
|
|
|
{
|
|
|
|
fc->reqctr++;
|
|
|
|
/* zero is special */
|
|
|
|
if (fc->reqctr == 0)
|
|
|
|
fc->reqctr = 1;
|
|
|
|
|
|
|
|
return fc->reqctr;
|
|
|
|
}
|
|
|
|
|
|
|
|
static void queue_request(struct fuse_conn *fc, struct fuse_req *req)
|
|
|
|
{
|
|
|
|
req->in.h.len = sizeof(struct fuse_in_header) +
|
|
|
|
len_args(req->in.numargs, (struct fuse_arg *) req->in.args);
|
|
|
|
list_add_tail(&req->list, &fc->pending);
|
|
|
|
req->state = FUSE_REQ_PENDING;
|
|
|
|
if (!req->waiting) {
|
|
|
|
req->waiting = 1;
|
|
|
|
atomic_inc(&fc->num_waiting);
|
|
|
|
}
|
|
|
|
wake_up(&fc->waitq);
|
|
|
|
kill_fasync(&fc->fasync, SIGIO, POLL_IN);
|
|
|
|
}
|
|
|
|
|
2010-12-07 20:16:56 +01:00
|
|
|
void fuse_queue_forget(struct fuse_conn *fc, struct fuse_forget_link *forget,
|
|
|
|
u64 nodeid, u64 nlookup)
|
|
|
|
{
|
2010-12-07 20:16:56 +01:00
|
|
|
forget->forget_one.nodeid = nodeid;
|
|
|
|
forget->forget_one.nlookup = nlookup;
|
2010-12-07 20:16:56 +01:00
|
|
|
|
|
|
|
spin_lock(&fc->lock);
|
2011-09-12 09:38:03 +02:00
|
|
|
if (fc->connected) {
|
|
|
|
fc->forget_list_tail->next = forget;
|
|
|
|
fc->forget_list_tail = forget;
|
|
|
|
wake_up(&fc->waitq);
|
|
|
|
kill_fasync(&fc->fasync, SIGIO, POLL_IN);
|
|
|
|
} else {
|
|
|
|
kfree(forget);
|
|
|
|
}
|
2010-12-07 20:16:56 +01:00
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
}
|
|
|
|
|
2008-02-06 01:38:39 -08:00
|
|
|
static void flush_bg_queue(struct fuse_conn *fc)
|
|
|
|
{
|
2009-07-01 17:28:41 -07:00
|
|
|
while (fc->active_background < fc->max_background &&
|
2008-02-06 01:38:39 -08:00
|
|
|
!list_empty(&fc->bg_queue)) {
|
|
|
|
struct fuse_req *req;
|
|
|
|
|
|
|
|
req = list_entry(fc->bg_queue.next, struct fuse_req, list);
|
|
|
|
list_del(&req->list);
|
|
|
|
fc->active_background++;
|
2010-07-12 14:41:40 +02:00
|
|
|
req->in.h.unique = fuse_get_unique(fc);
|
2008-02-06 01:38:39 -08:00
|
|
|
queue_request(fc, req);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
/*
|
|
|
|
* This function is called when a request is finished. Either a reply
|
2006-06-25 05:48:53 -07:00
|
|
|
* has arrived or it was aborted (and not yet sent) or some error
|
2006-01-16 22:14:26 -08:00
|
|
|
* occurred during communication with userspace, or the device file
|
2006-06-25 05:48:50 -07:00
|
|
|
* was closed. The requester thread is woken up (if still waiting),
|
|
|
|
* the 'end' callback is called if given, else the reference to the
|
|
|
|
* request is released
|
2006-02-04 23:27:40 -08:00
|
|
|
*
|
2006-04-10 22:54:55 -07:00
|
|
|
* Called with fc->lock, unlocks it
|
2005-09-09 13:10:27 -07:00
|
|
|
*/
|
|
|
|
static void request_end(struct fuse_conn *fc, struct fuse_req *req)
|
2010-09-07 13:42:41 +02:00
|
|
|
__releases(fc->lock)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
2006-06-25 05:48:50 -07:00
|
|
|
void (*end) (struct fuse_conn *, struct fuse_req *) = req->end;
|
|
|
|
req->end = NULL;
|
2006-01-16 22:14:31 -08:00
|
|
|
list_del(&req->list);
|
2006-06-25 05:48:54 -07:00
|
|
|
list_del(&req->intr_entry);
|
2006-01-16 22:14:31 -08:00
|
|
|
req->state = FUSE_REQ_FINISHED;
|
2006-06-25 05:48:50 -07:00
|
|
|
if (req->background) {
|
2013-03-21 18:02:36 +04:00
|
|
|
req->background = 0;
|
|
|
|
|
|
|
|
if (fc->num_background == fc->max_background)
|
2006-06-25 05:48:50 -07:00
|
|
|
fc->blocked = 0;
|
2013-03-21 18:02:36 +04:00
|
|
|
|
|
|
|
/* Wake up next waiter, if any */
|
2013-04-17 21:50:58 +02:00
|
|
|
if (!fc->blocked && waitqueue_active(&fc->blocked_waitq))
|
2013-03-21 18:02:36 +04:00
|
|
|
wake_up(&fc->blocked_waitq);
|
|
|
|
|
2009-07-01 17:28:41 -07:00
|
|
|
if (fc->num_background == fc->congestion_threshold &&
|
2009-04-14 10:54:52 +09:00
|
|
|
fc->connected && fc->bdi_initialized) {
|
2009-07-09 14:52:32 +02:00
|
|
|
clear_bdi_congested(&fc->bdi, BLK_RW_SYNC);
|
|
|
|
clear_bdi_congested(&fc->bdi, BLK_RW_ASYNC);
|
2007-10-16 23:30:59 -07:00
|
|
|
}
|
2006-06-25 05:48:50 -07:00
|
|
|
fc->num_background--;
|
2008-02-06 01:38:39 -08:00
|
|
|
fc->active_background--;
|
|
|
|
flush_bg_queue(fc);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
2006-06-25 05:48:50 -07:00
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
wake_up(&req->waitq);
|
|
|
|
if (end)
|
|
|
|
end(fc, req);
|
2008-11-26 12:03:54 +01:00
|
|
|
fuse_put_request(fc, req);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
|
2006-06-25 05:48:54 -07:00
|
|
|
static void wait_answer_interruptible(struct fuse_conn *fc,
|
|
|
|
struct fuse_req *req)
|
2010-09-07 13:42:41 +02:00
|
|
|
__releases(fc->lock)
|
|
|
|
__acquires(fc->lock)
|
2006-06-25 05:48:54 -07:00
|
|
|
{
|
|
|
|
if (signal_pending(current))
|
|
|
|
return;
|
|
|
|
|
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
wait_event_interruptible(req->waitq, req->state == FUSE_REQ_FINISHED);
|
|
|
|
spin_lock(&fc->lock);
|
|
|
|
}
|
|
|
|
|
|
|
|
static void queue_interrupt(struct fuse_conn *fc, struct fuse_req *req)
|
|
|
|
{
|
|
|
|
list_add_tail(&req->intr_entry, &fc->interrupts);
|
|
|
|
wake_up(&fc->waitq);
|
|
|
|
kill_fasync(&fc->fasync, SIGIO, POLL_IN);
|
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:39 -07:00
|
|
|
static void request_wait_answer(struct fuse_conn *fc, struct fuse_req *req)
|
2010-09-07 13:42:41 +02:00
|
|
|
__releases(fc->lock)
|
|
|
|
__acquires(fc->lock)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
2006-06-25 05:48:54 -07:00
|
|
|
if (!fc->no_interrupt) {
|
|
|
|
/* Any signal may interrupt this */
|
|
|
|
wait_answer_interruptible(fc, req);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2006-06-25 05:48:54 -07:00
|
|
|
if (req->aborted)
|
|
|
|
goto aborted;
|
|
|
|
if (req->state == FUSE_REQ_FINISHED)
|
|
|
|
return;
|
|
|
|
|
|
|
|
req->interrupted = 1;
|
|
|
|
if (req->state == FUSE_REQ_SENT)
|
|
|
|
queue_interrupt(fc, req);
|
|
|
|
}
|
|
|
|
|
2007-10-16 23:31:04 -07:00
|
|
|
if (!req->force) {
|
2006-06-25 05:48:54 -07:00
|
|
|
sigset_t oldset;
|
|
|
|
|
|
|
|
/* Only fatal signals may interrupt this */
|
2006-06-25 05:48:50 -07:00
|
|
|
block_sigs(&oldset);
|
2006-06-25 05:48:54 -07:00
|
|
|
wait_answer_interruptible(fc, req);
|
2006-06-25 05:48:50 -07:00
|
|
|
restore_sigs(&oldset);
|
2007-10-16 23:31:04 -07:00
|
|
|
|
|
|
|
if (req->aborted)
|
|
|
|
goto aborted;
|
|
|
|
if (req->state == FUSE_REQ_FINISHED)
|
|
|
|
return;
|
|
|
|
|
|
|
|
/* Request is not yet in userspace, bail out */
|
|
|
|
if (req->state == FUSE_REQ_PENDING) {
|
|
|
|
list_del(&req->list);
|
|
|
|
__fuse_put_request(req);
|
|
|
|
req->out.h.error = -EINTR;
|
|
|
|
return;
|
|
|
|
}
|
2006-06-25 05:48:50 -07:00
|
|
|
}
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2007-10-16 23:31:04 -07:00
|
|
|
/*
|
|
|
|
* Either request is already in userspace, or it was forced.
|
|
|
|
* Wait it out.
|
|
|
|
*/
|
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
wait_event(req->waitq, req->state == FUSE_REQ_FINISHED);
|
|
|
|
spin_lock(&fc->lock);
|
2006-06-25 05:48:54 -07:00
|
|
|
|
2007-10-16 23:31:04 -07:00
|
|
|
if (!req->aborted)
|
|
|
|
return;
|
2006-06-25 05:48:54 -07:00
|
|
|
|
|
|
|
aborted:
|
2007-10-16 23:31:04 -07:00
|
|
|
BUG_ON(req->state != FUSE_REQ_FINISHED);
|
2005-09-09 13:10:27 -07:00
|
|
|
if (req->locked) {
|
|
|
|
/* This is uninterruptible sleep, because data is
|
|
|
|
being copied to/from the buffers of req. During
|
|
|
|
locked state, there mustn't be any filesystem
|
|
|
|
operation (e.g. page fault), since that could lead
|
|
|
|
to deadlock */
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
wait_event(req->waitq, !req->locked);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2013-02-04 13:04:44 +00:00
|
|
|
static void __fuse_request_send(struct fuse_conn *fc, struct fuse_req *req)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
2013-03-21 18:02:04 +04:00
|
|
|
BUG_ON(req->background);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2005-09-09 13:10:31 -07:00
|
|
|
if (!fc->connected)
|
2005-09-09 13:10:27 -07:00
|
|
|
req->out.h.error = -ENOTCONN;
|
|
|
|
else if (fc->conn_error)
|
|
|
|
req->out.h.error = -ECONNREFUSED;
|
|
|
|
else {
|
2010-07-12 14:41:40 +02:00
|
|
|
req->in.h.unique = fuse_get_unique(fc);
|
2005-09-09 13:10:27 -07:00
|
|
|
queue_request(fc, req);
|
|
|
|
/* acquire extra reference, since request is still needed
|
|
|
|
after request_end() */
|
|
|
|
__fuse_get_request(req);
|
|
|
|
|
2005-09-09 13:10:39 -07:00
|
|
|
request_wait_answer(fc, req);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
2013-02-04 13:04:44 +00:00
|
|
|
|
|
|
|
void fuse_request_send(struct fuse_conn *fc, struct fuse_req *req)
|
|
|
|
{
|
|
|
|
req->isreply = 1;
|
|
|
|
__fuse_request_send(fc, req);
|
|
|
|
}
|
2009-04-14 10:54:53 +09:00
|
|
|
EXPORT_SYMBOL_GPL(fuse_request_send);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2008-11-26 12:03:55 +01:00
|
|
|
static void fuse_request_send_nowait_locked(struct fuse_conn *fc,
|
|
|
|
struct fuse_req *req)
|
2008-02-06 01:38:39 -08:00
|
|
|
{
|
2013-03-21 18:02:04 +04:00
|
|
|
BUG_ON(!req->background);
|
2008-02-06 01:38:39 -08:00
|
|
|
fc->num_background++;
|
2009-07-01 17:28:41 -07:00
|
|
|
if (fc->num_background == fc->max_background)
|
2008-02-06 01:38:39 -08:00
|
|
|
fc->blocked = 1;
|
2009-07-01 17:28:41 -07:00
|
|
|
if (fc->num_background == fc->congestion_threshold &&
|
2009-04-14 10:54:52 +09:00
|
|
|
fc->bdi_initialized) {
|
2009-07-09 14:52:32 +02:00
|
|
|
set_bdi_congested(&fc->bdi, BLK_RW_SYNC);
|
|
|
|
set_bdi_congested(&fc->bdi, BLK_RW_ASYNC);
|
2008-02-06 01:38:39 -08:00
|
|
|
}
|
|
|
|
list_add_tail(&req->list, &fc->bg_queue);
|
|
|
|
flush_bg_queue(fc);
|
|
|
|
}
|
|
|
|
|
2008-11-26 12:03:55 +01:00
|
|
|
static void fuse_request_send_nowait(struct fuse_conn *fc, struct fuse_req *req)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2005-09-09 13:10:31 -07:00
|
|
|
if (fc->connected) {
|
2008-11-26 12:03:55 +01:00
|
|
|
fuse_request_send_nowait_locked(fc, req);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
} else {
|
|
|
|
req->out.h.error = -ENOTCONN;
|
|
|
|
request_end(fc, req);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2008-11-26 12:03:55 +01:00
|
|
|
void fuse_request_send_background(struct fuse_conn *fc, struct fuse_req *req)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
req->isreply = 1;
|
2008-11-26 12:03:55 +01:00
|
|
|
fuse_request_send_nowait(fc, req);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
2009-04-14 10:54:53 +09:00
|
|
|
EXPORT_SYMBOL_GPL(fuse_request_send_background);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2010-07-12 14:41:40 +02:00
|
|
|
static int fuse_request_send_notify_reply(struct fuse_conn *fc,
|
|
|
|
struct fuse_req *req, u64 unique)
|
|
|
|
{
|
|
|
|
int err = -ENODEV;
|
|
|
|
|
|
|
|
req->isreply = 0;
|
|
|
|
req->in.h.unique = unique;
|
|
|
|
spin_lock(&fc->lock);
|
|
|
|
if (fc->connected) {
|
|
|
|
queue_request(fc, req);
|
|
|
|
err = 0;
|
|
|
|
}
|
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
fuse: support writable mmap
Quoting Linus (3 years ago, FUSE inclusion discussions):
"User-space filesystems are hard to get right. I'd claim that they
are almost impossible, unless you limit them somehow (shared
writable mappings are the nastiest part - if you don't have those,
you can reasonably limit your problems by limiting the number of
dirty pages you accept through normal "write()" calls)."
Instead of attempting the impossible, I've just waited for the dirty page
accounting infrastructure to materialize (thanks to Peter Zijlstra and
others). This nicely solved the biggest problem: limiting the number of pages
used for write caching.
Some small details remained, however, which this largish patch attempts to
address. It provides a page writeback implementation for fuse, which is
completely safe against VM related deadlocks. Performance may not be very
good for certain usage patterns, but generally it should be acceptable.
It has been tested extensively with fsx-linux and bash-shared-mapping.
Fuse page writeback design
--------------------------
fuse_writepage() allocates a new temporary page with GFP_NOFS|__GFP_HIGHMEM.
It copies the contents of the original page, and queues a WRITE request to the
userspace filesystem using this temp page.
The writeback is finished instantly from the MM's point of view: the page is
removed from the radix trees, and the PageDirty and PageWriteback flags are
cleared.
For the duration of the actual write, the NR_WRITEBACK_TEMP counter is
incremented. The per-bdi writeback count is not decremented until the actual
write completes.
On dirtying the page, fuse waits for a previous write to finish before
proceeding. This makes sure, there can only be one temporary page used at a
time for one cached page.
This approach is wasteful in both memory and CPU bandwidth, so why is this
complication needed?
The basic problem is that there can be no guarantee about the time in which
the userspace filesystem will complete a write. It may be buggy or even
malicious, and fail to complete WRITE requests. We don't want unrelated parts
of the system to grind to a halt in such cases.
Also a filesystem may need additional resources (particularly memory) to
complete a WRITE request. There's a great danger of a deadlock if that
allocation may wait for the writepage to finish.
Currently there are several cases where the kernel can block on page
writeback:
- allocation order is larger than PAGE_ALLOC_COSTLY_ORDER
- page migration
- throttle_vm_writeout (through NR_WRITEBACK)
- sync(2)
Of course in some cases (fsync, msync) we explicitly want to allow blocking.
So for these cases new code has to be added to fuse, since the VM is not
tracking writeback pages for us any more.
As an extra safetly measure, the maximum dirty ratio allocated to a single
fuse filesystem is set to 1% by default. This way one (or several) buggy or
malicious fuse filesystems cannot slow down the rest of the system by hogging
dirty memory.
With appropriate privileges, this limit can be raised through
'/sys/class/bdi/<bdi>/max_ratio'.
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
Cc: Peter Zijlstra <a.p.zijlstra@chello.nl>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
2008-04-30 00:54:41 -07:00
|
|
|
/*
|
|
|
|
* Called under fc->lock
|
|
|
|
*
|
|
|
|
* fc->connected must have been checked previously
|
|
|
|
*/
|
2008-11-26 12:03:55 +01:00
|
|
|
void fuse_request_send_background_locked(struct fuse_conn *fc,
|
|
|
|
struct fuse_req *req)
|
fuse: support writable mmap
Quoting Linus (3 years ago, FUSE inclusion discussions):
"User-space filesystems are hard to get right. I'd claim that they
are almost impossible, unless you limit them somehow (shared
writable mappings are the nastiest part - if you don't have those,
you can reasonably limit your problems by limiting the number of
dirty pages you accept through normal "write()" calls)."
Instead of attempting the impossible, I've just waited for the dirty page
accounting infrastructure to materialize (thanks to Peter Zijlstra and
others). This nicely solved the biggest problem: limiting the number of pages
used for write caching.
Some small details remained, however, which this largish patch attempts to
address. It provides a page writeback implementation for fuse, which is
completely safe against VM related deadlocks. Performance may not be very
good for certain usage patterns, but generally it should be acceptable.
It has been tested extensively with fsx-linux and bash-shared-mapping.
Fuse page writeback design
--------------------------
fuse_writepage() allocates a new temporary page with GFP_NOFS|__GFP_HIGHMEM.
It copies the contents of the original page, and queues a WRITE request to the
userspace filesystem using this temp page.
The writeback is finished instantly from the MM's point of view: the page is
removed from the radix trees, and the PageDirty and PageWriteback flags are
cleared.
For the duration of the actual write, the NR_WRITEBACK_TEMP counter is
incremented. The per-bdi writeback count is not decremented until the actual
write completes.
On dirtying the page, fuse waits for a previous write to finish before
proceeding. This makes sure, there can only be one temporary page used at a
time for one cached page.
This approach is wasteful in both memory and CPU bandwidth, so why is this
complication needed?
The basic problem is that there can be no guarantee about the time in which
the userspace filesystem will complete a write. It may be buggy or even
malicious, and fail to complete WRITE requests. We don't want unrelated parts
of the system to grind to a halt in such cases.
Also a filesystem may need additional resources (particularly memory) to
complete a WRITE request. There's a great danger of a deadlock if that
allocation may wait for the writepage to finish.
Currently there are several cases where the kernel can block on page
writeback:
- allocation order is larger than PAGE_ALLOC_COSTLY_ORDER
- page migration
- throttle_vm_writeout (through NR_WRITEBACK)
- sync(2)
Of course in some cases (fsync, msync) we explicitly want to allow blocking.
So for these cases new code has to be added to fuse, since the VM is not
tracking writeback pages for us any more.
As an extra safetly measure, the maximum dirty ratio allocated to a single
fuse filesystem is set to 1% by default. This way one (or several) buggy or
malicious fuse filesystems cannot slow down the rest of the system by hogging
dirty memory.
With appropriate privileges, this limit can be raised through
'/sys/class/bdi/<bdi>/max_ratio'.
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
Cc: Peter Zijlstra <a.p.zijlstra@chello.nl>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
2008-04-30 00:54:41 -07:00
|
|
|
{
|
|
|
|
req->isreply = 1;
|
2008-11-26 12:03:55 +01:00
|
|
|
fuse_request_send_nowait_locked(fc, req);
|
fuse: support writable mmap
Quoting Linus (3 years ago, FUSE inclusion discussions):
"User-space filesystems are hard to get right. I'd claim that they
are almost impossible, unless you limit them somehow (shared
writable mappings are the nastiest part - if you don't have those,
you can reasonably limit your problems by limiting the number of
dirty pages you accept through normal "write()" calls)."
Instead of attempting the impossible, I've just waited for the dirty page
accounting infrastructure to materialize (thanks to Peter Zijlstra and
others). This nicely solved the biggest problem: limiting the number of pages
used for write caching.
Some small details remained, however, which this largish patch attempts to
address. It provides a page writeback implementation for fuse, which is
completely safe against VM related deadlocks. Performance may not be very
good for certain usage patterns, but generally it should be acceptable.
It has been tested extensively with fsx-linux and bash-shared-mapping.
Fuse page writeback design
--------------------------
fuse_writepage() allocates a new temporary page with GFP_NOFS|__GFP_HIGHMEM.
It copies the contents of the original page, and queues a WRITE request to the
userspace filesystem using this temp page.
The writeback is finished instantly from the MM's point of view: the page is
removed from the radix trees, and the PageDirty and PageWriteback flags are
cleared.
For the duration of the actual write, the NR_WRITEBACK_TEMP counter is
incremented. The per-bdi writeback count is not decremented until the actual
write completes.
On dirtying the page, fuse waits for a previous write to finish before
proceeding. This makes sure, there can only be one temporary page used at a
time for one cached page.
This approach is wasteful in both memory and CPU bandwidth, so why is this
complication needed?
The basic problem is that there can be no guarantee about the time in which
the userspace filesystem will complete a write. It may be buggy or even
malicious, and fail to complete WRITE requests. We don't want unrelated parts
of the system to grind to a halt in such cases.
Also a filesystem may need additional resources (particularly memory) to
complete a WRITE request. There's a great danger of a deadlock if that
allocation may wait for the writepage to finish.
Currently there are several cases where the kernel can block on page
writeback:
- allocation order is larger than PAGE_ALLOC_COSTLY_ORDER
- page migration
- throttle_vm_writeout (through NR_WRITEBACK)
- sync(2)
Of course in some cases (fsync, msync) we explicitly want to allow blocking.
So for these cases new code has to be added to fuse, since the VM is not
tracking writeback pages for us any more.
As an extra safetly measure, the maximum dirty ratio allocated to a single
fuse filesystem is set to 1% by default. This way one (or several) buggy or
malicious fuse filesystems cannot slow down the rest of the system by hogging
dirty memory.
With appropriate privileges, this limit can be raised through
'/sys/class/bdi/<bdi>/max_ratio'.
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
Cc: Peter Zijlstra <a.p.zijlstra@chello.nl>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
2008-04-30 00:54:41 -07:00
|
|
|
}
|
|
|
|
|
2012-08-19 08:53:23 -04:00
|
|
|
void fuse_force_forget(struct file *file, u64 nodeid)
|
|
|
|
{
|
2013-02-27 16:59:05 -05:00
|
|
|
struct inode *inode = file_inode(file);
|
2012-08-19 08:53:23 -04:00
|
|
|
struct fuse_conn *fc = get_fuse_conn(inode);
|
|
|
|
struct fuse_req *req;
|
|
|
|
struct fuse_forget_in inarg;
|
|
|
|
|
|
|
|
memset(&inarg, 0, sizeof(inarg));
|
|
|
|
inarg.nlookup = 1;
|
2012-10-26 19:48:30 +04:00
|
|
|
req = fuse_get_req_nofail_nopages(fc, file);
|
2012-08-19 08:53:23 -04:00
|
|
|
req->in.h.opcode = FUSE_FORGET;
|
|
|
|
req->in.h.nodeid = nodeid;
|
|
|
|
req->in.numargs = 1;
|
|
|
|
req->in.args[0].size = sizeof(inarg);
|
|
|
|
req->in.args[0].value = &inarg;
|
|
|
|
req->isreply = 0;
|
2013-02-04 13:04:44 +00:00
|
|
|
__fuse_request_send(fc, req);
|
|
|
|
/* ignore errors */
|
|
|
|
fuse_put_request(fc, req);
|
2012-08-19 08:53:23 -04:00
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
/*
|
|
|
|
* Lock the request. Up to the next unlock_request() there mustn't be
|
|
|
|
* anything that could cause a page-fault. If the request was already
|
2006-06-25 05:48:53 -07:00
|
|
|
* aborted bail out.
|
2005-09-09 13:10:27 -07:00
|
|
|
*/
|
2006-04-10 22:54:55 -07:00
|
|
|
static int lock_request(struct fuse_conn *fc, struct fuse_req *req)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
int err = 0;
|
|
|
|
if (req) {
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2006-06-25 05:48:53 -07:00
|
|
|
if (req->aborted)
|
2005-09-09 13:10:27 -07:00
|
|
|
err = -ENOENT;
|
|
|
|
else
|
|
|
|
req->locked = 1;
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
|
|
|
/*
|
2006-06-25 05:48:53 -07:00
|
|
|
* Unlock request. If it was aborted during being locked, the
|
2005-09-09 13:10:27 -07:00
|
|
|
* requester thread is currently waiting for it to be unlocked, so
|
|
|
|
* wake it up.
|
|
|
|
*/
|
2006-04-10 22:54:55 -07:00
|
|
|
static void unlock_request(struct fuse_conn *fc, struct fuse_req *req)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
if (req) {
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
req->locked = 0;
|
2006-06-25 05:48:53 -07:00
|
|
|
if (req->aborted)
|
2005-09-09 13:10:27 -07:00
|
|
|
wake_up(&req->waitq);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
struct fuse_copy_state {
|
2006-04-10 22:54:55 -07:00
|
|
|
struct fuse_conn *fc;
|
2005-09-09 13:10:27 -07:00
|
|
|
int write;
|
|
|
|
struct fuse_req *req;
|
|
|
|
const struct iovec *iov;
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
struct pipe_buffer *pipebufs;
|
|
|
|
struct pipe_buffer *currbuf;
|
|
|
|
struct pipe_inode_info *pipe;
|
2005-09-09 13:10:27 -07:00
|
|
|
unsigned long nr_segs;
|
|
|
|
unsigned long seglen;
|
|
|
|
unsigned long addr;
|
|
|
|
struct page *pg;
|
|
|
|
unsigned len;
|
2014-07-07 15:28:51 +02:00
|
|
|
unsigned offset;
|
2010-05-25 15:06:07 +02:00
|
|
|
unsigned move_pages:1;
|
2005-09-09 13:10:27 -07:00
|
|
|
};
|
|
|
|
|
2006-04-10 22:54:55 -07:00
|
|
|
static void fuse_copy_init(struct fuse_copy_state *cs, struct fuse_conn *fc,
|
2010-05-25 15:06:07 +02:00
|
|
|
int write,
|
2006-04-10 22:54:55 -07:00
|
|
|
const struct iovec *iov, unsigned long nr_segs)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
memset(cs, 0, sizeof(*cs));
|
2006-04-10 22:54:55 -07:00
|
|
|
cs->fc = fc;
|
2005-09-09 13:10:27 -07:00
|
|
|
cs->write = write;
|
|
|
|
cs->iov = iov;
|
|
|
|
cs->nr_segs = nr_segs;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* Unmap and put previous page of userspace buffer */
|
2006-01-16 22:14:28 -08:00
|
|
|
static void fuse_copy_finish(struct fuse_copy_state *cs)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
if (cs->currbuf) {
|
|
|
|
struct pipe_buffer *buf = cs->currbuf;
|
|
|
|
|
2014-07-07 15:28:51 +02:00
|
|
|
if (cs->write)
|
2010-05-25 15:06:07 +02:00
|
|
|
buf->len = PAGE_SIZE - cs->len;
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
cs->currbuf = NULL;
|
2014-07-07 15:28:51 +02:00
|
|
|
} else if (cs->pg) {
|
2005-09-09 13:10:27 -07:00
|
|
|
if (cs->write) {
|
|
|
|
flush_dcache_page(cs->pg);
|
|
|
|
set_page_dirty_lock(cs->pg);
|
|
|
|
}
|
|
|
|
put_page(cs->pg);
|
|
|
|
}
|
2014-07-07 15:28:51 +02:00
|
|
|
cs->pg = NULL;
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
|
|
|
|
/*
|
|
|
|
* Get another pagefull of userspace buffer, and map it to kernel
|
|
|
|
* address space, and lock request
|
|
|
|
*/
|
|
|
|
static int fuse_copy_fill(struct fuse_copy_state *cs)
|
|
|
|
{
|
2014-07-07 15:28:51 +02:00
|
|
|
struct page *page;
|
2005-09-09 13:10:27 -07:00
|
|
|
int err;
|
|
|
|
|
2006-04-10 22:54:55 -07:00
|
|
|
unlock_request(cs->fc, cs->req);
|
2005-09-09 13:10:27 -07:00
|
|
|
fuse_copy_finish(cs);
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
if (cs->pipebufs) {
|
|
|
|
struct pipe_buffer *buf = cs->pipebufs;
|
|
|
|
|
2010-05-25 15:06:07 +02:00
|
|
|
if (!cs->write) {
|
|
|
|
err = buf->ops->confirm(cs->pipe, buf);
|
|
|
|
if (err)
|
|
|
|
return err;
|
|
|
|
|
|
|
|
BUG_ON(!cs->nr_segs);
|
|
|
|
cs->currbuf = buf;
|
2014-07-07 15:28:51 +02:00
|
|
|
cs->pg = buf->page;
|
|
|
|
cs->offset = buf->offset;
|
2010-05-25 15:06:07 +02:00
|
|
|
cs->len = buf->len;
|
|
|
|
cs->pipebufs++;
|
|
|
|
cs->nr_segs--;
|
|
|
|
} else {
|
|
|
|
if (cs->nr_segs == cs->pipe->buffers)
|
|
|
|
return -EIO;
|
|
|
|
|
|
|
|
page = alloc_page(GFP_HIGHUSER);
|
|
|
|
if (!page)
|
|
|
|
return -ENOMEM;
|
|
|
|
|
|
|
|
buf->page = page;
|
|
|
|
buf->offset = 0;
|
|
|
|
buf->len = 0;
|
|
|
|
|
|
|
|
cs->currbuf = buf;
|
2014-07-07 15:28:51 +02:00
|
|
|
cs->pg = page;
|
|
|
|
cs->offset = 0;
|
2010-05-25 15:06:07 +02:00
|
|
|
cs->len = PAGE_SIZE;
|
|
|
|
cs->pipebufs++;
|
|
|
|
cs->nr_segs++;
|
|
|
|
}
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
} else {
|
|
|
|
if (!cs->seglen) {
|
|
|
|
BUG_ON(!cs->nr_segs);
|
|
|
|
cs->seglen = cs->iov[0].iov_len;
|
|
|
|
cs->addr = (unsigned long) cs->iov[0].iov_base;
|
|
|
|
cs->iov++;
|
|
|
|
cs->nr_segs--;
|
|
|
|
}
|
2014-07-07 15:28:51 +02:00
|
|
|
err = get_user_pages_fast(cs->addr, 1, cs->write, &page);
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
if (err < 0)
|
|
|
|
return err;
|
|
|
|
BUG_ON(err != 1);
|
2014-07-07 15:28:51 +02:00
|
|
|
cs->pg = page;
|
|
|
|
cs->offset = cs->addr % PAGE_SIZE;
|
|
|
|
cs->len = min(PAGE_SIZE - cs->offset, cs->seglen);
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
cs->seglen -= cs->len;
|
|
|
|
cs->addr += cs->len;
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
|
2006-04-10 22:54:55 -07:00
|
|
|
return lock_request(cs->fc, cs->req);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
|
|
|
|
/* Do as much copy to/from userspace buffer as we can */
|
2006-01-16 22:14:28 -08:00
|
|
|
static int fuse_copy_do(struct fuse_copy_state *cs, void **val, unsigned *size)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
unsigned ncpy = min(*size, cs->len);
|
|
|
|
if (val) {
|
2014-07-07 15:28:51 +02:00
|
|
|
void *pgaddr = kmap_atomic(cs->pg);
|
|
|
|
void *buf = pgaddr + cs->offset;
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
if (cs->write)
|
2014-07-07 15:28:51 +02:00
|
|
|
memcpy(buf, *val, ncpy);
|
2005-09-09 13:10:27 -07:00
|
|
|
else
|
2014-07-07 15:28:51 +02:00
|
|
|
memcpy(*val, buf, ncpy);
|
|
|
|
|
|
|
|
kunmap_atomic(pgaddr);
|
2005-09-09 13:10:27 -07:00
|
|
|
*val += ncpy;
|
|
|
|
}
|
|
|
|
*size -= ncpy;
|
|
|
|
cs->len -= ncpy;
|
2014-07-07 15:28:51 +02:00
|
|
|
cs->offset += ncpy;
|
2005-09-09 13:10:27 -07:00
|
|
|
return ncpy;
|
|
|
|
}
|
|
|
|
|
2010-05-25 15:06:07 +02:00
|
|
|
static int fuse_check_page(struct page *page)
|
|
|
|
{
|
|
|
|
if (page_mapcount(page) ||
|
|
|
|
page->mapping != NULL ||
|
|
|
|
page_count(page) != 1 ||
|
|
|
|
(page->flags & PAGE_FLAGS_CHECK_AT_PREP &
|
|
|
|
~(1 << PG_locked |
|
|
|
|
1 << PG_referenced |
|
|
|
|
1 << PG_uptodate |
|
|
|
|
1 << PG_lru |
|
|
|
|
1 << PG_active |
|
|
|
|
1 << PG_reclaim))) {
|
|
|
|
printk(KERN_WARNING "fuse: trying to steal weird page\n");
|
|
|
|
printk(KERN_WARNING " page=%p index=%li flags=%08lx, count=%i, mapcount=%i, mapping=%p\n", page, page->index, page->flags, page_count(page), page_mapcount(page), page->mapping);
|
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
return 0;
|
|
|
|
}
|
|
|
|
|
|
|
|
static int fuse_try_move_page(struct fuse_copy_state *cs, struct page **pagep)
|
|
|
|
{
|
|
|
|
int err;
|
|
|
|
struct page *oldpage = *pagep;
|
|
|
|
struct page *newpage;
|
|
|
|
struct pipe_buffer *buf = cs->pipebufs;
|
|
|
|
|
|
|
|
unlock_request(cs->fc, cs->req);
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
|
|
|
|
err = buf->ops->confirm(cs->pipe, buf);
|
|
|
|
if (err)
|
|
|
|
return err;
|
|
|
|
|
|
|
|
BUG_ON(!cs->nr_segs);
|
|
|
|
cs->currbuf = buf;
|
|
|
|
cs->len = buf->len;
|
|
|
|
cs->pipebufs++;
|
|
|
|
cs->nr_segs--;
|
|
|
|
|
|
|
|
if (cs->len != PAGE_SIZE)
|
|
|
|
goto out_fallback;
|
|
|
|
|
|
|
|
if (buf->ops->steal(cs->pipe, buf) != 0)
|
|
|
|
goto out_fallback;
|
|
|
|
|
|
|
|
newpage = buf->page;
|
|
|
|
|
|
|
|
if (WARN_ON(!PageUptodate(newpage)))
|
|
|
|
return -EIO;
|
|
|
|
|
|
|
|
ClearPageMappedToDisk(newpage);
|
|
|
|
|
|
|
|
if (fuse_check_page(newpage) != 0)
|
|
|
|
goto out_fallback_unlock;
|
|
|
|
|
|
|
|
/*
|
|
|
|
* This is a new and locked page, it shouldn't be mapped or
|
|
|
|
* have any special flags on it
|
|
|
|
*/
|
|
|
|
if (WARN_ON(page_mapped(oldpage)))
|
|
|
|
goto out_fallback_unlock;
|
|
|
|
if (WARN_ON(page_has_private(oldpage)))
|
|
|
|
goto out_fallback_unlock;
|
|
|
|
if (WARN_ON(PageDirty(oldpage) || PageWriteback(oldpage)))
|
|
|
|
goto out_fallback_unlock;
|
|
|
|
if (WARN_ON(PageMlocked(oldpage)))
|
|
|
|
goto out_fallback_unlock;
|
|
|
|
|
2011-03-22 16:30:52 -07:00
|
|
|
err = replace_page_cache_page(oldpage, newpage, GFP_KERNEL);
|
2010-05-25 15:06:07 +02:00
|
|
|
if (err) {
|
2011-03-22 16:30:52 -07:00
|
|
|
unlock_page(newpage);
|
|
|
|
return err;
|
2010-05-25 15:06:07 +02:00
|
|
|
}
|
2011-03-22 16:30:52 -07:00
|
|
|
|
2010-05-25 15:06:07 +02:00
|
|
|
page_cache_get(newpage);
|
|
|
|
|
|
|
|
if (!(buf->flags & PIPE_BUF_FLAG_LRU))
|
|
|
|
lru_cache_add_file(newpage);
|
|
|
|
|
|
|
|
err = 0;
|
|
|
|
spin_lock(&cs->fc->lock);
|
|
|
|
if (cs->req->aborted)
|
|
|
|
err = -ENOENT;
|
|
|
|
else
|
|
|
|
*pagep = newpage;
|
|
|
|
spin_unlock(&cs->fc->lock);
|
|
|
|
|
|
|
|
if (err) {
|
|
|
|
unlock_page(newpage);
|
|
|
|
page_cache_release(newpage);
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
|
|
|
unlock_page(oldpage);
|
|
|
|
page_cache_release(oldpage);
|
|
|
|
cs->len = 0;
|
|
|
|
|
|
|
|
return 0;
|
|
|
|
|
|
|
|
out_fallback_unlock:
|
|
|
|
unlock_page(newpage);
|
|
|
|
out_fallback:
|
2014-07-07 15:28:51 +02:00
|
|
|
cs->pg = buf->page;
|
|
|
|
cs->offset = buf->offset;
|
2010-05-25 15:06:07 +02:00
|
|
|
|
|
|
|
err = lock_request(cs->fc, cs->req);
|
|
|
|
if (err)
|
|
|
|
return err;
|
|
|
|
|
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
|
2010-05-25 15:06:07 +02:00
|
|
|
static int fuse_ref_page(struct fuse_copy_state *cs, struct page *page,
|
|
|
|
unsigned offset, unsigned count)
|
|
|
|
{
|
|
|
|
struct pipe_buffer *buf;
|
|
|
|
|
|
|
|
if (cs->nr_segs == cs->pipe->buffers)
|
|
|
|
return -EIO;
|
|
|
|
|
|
|
|
unlock_request(cs->fc, cs->req);
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
|
|
|
|
buf = cs->pipebufs;
|
|
|
|
page_cache_get(page);
|
|
|
|
buf->page = page;
|
|
|
|
buf->offset = offset;
|
|
|
|
buf->len = count;
|
|
|
|
|
|
|
|
cs->pipebufs++;
|
|
|
|
cs->nr_segs++;
|
|
|
|
cs->len = 0;
|
|
|
|
|
|
|
|
return 0;
|
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
/*
|
|
|
|
* Copy a page in the request to/from the userspace buffer. Must be
|
|
|
|
* done atomically
|
|
|
|
*/
|
2010-05-25 15:06:07 +02:00
|
|
|
static int fuse_copy_page(struct fuse_copy_state *cs, struct page **pagep,
|
2006-01-16 22:14:28 -08:00
|
|
|
unsigned offset, unsigned count, int zeroing)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
2010-05-25 15:06:07 +02:00
|
|
|
int err;
|
|
|
|
struct page *page = *pagep;
|
|
|
|
|
2010-10-26 14:22:27 -07:00
|
|
|
if (page && zeroing && count < PAGE_SIZE)
|
|
|
|
clear_highpage(page);
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
while (count) {
|
2010-05-25 15:06:07 +02:00
|
|
|
if (cs->write && cs->pipebufs && page) {
|
|
|
|
return fuse_ref_page(cs, page, offset, count);
|
|
|
|
} else if (!cs->len) {
|
2010-05-25 15:06:07 +02:00
|
|
|
if (cs->move_pages && page &&
|
|
|
|
offset == 0 && count == PAGE_SIZE) {
|
|
|
|
err = fuse_try_move_page(cs, pagep);
|
|
|
|
if (err <= 0)
|
|
|
|
return err;
|
|
|
|
} else {
|
|
|
|
err = fuse_copy_fill(cs);
|
|
|
|
if (err)
|
|
|
|
return err;
|
|
|
|
}
|
2008-11-26 12:03:54 +01:00
|
|
|
}
|
2005-09-09 13:10:27 -07:00
|
|
|
if (page) {
|
2011-11-25 23:14:30 +08:00
|
|
|
void *mapaddr = kmap_atomic(page);
|
2005-09-09 13:10:27 -07:00
|
|
|
void *buf = mapaddr + offset;
|
|
|
|
offset += fuse_copy_do(cs, &buf, &count);
|
2011-11-25 23:14:30 +08:00
|
|
|
kunmap_atomic(mapaddr);
|
2005-09-09 13:10:27 -07:00
|
|
|
} else
|
|
|
|
offset += fuse_copy_do(cs, NULL, &count);
|
|
|
|
}
|
|
|
|
if (page && !cs->write)
|
|
|
|
flush_dcache_page(page);
|
|
|
|
return 0;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* Copy pages in the request to/from userspace buffer */
|
|
|
|
static int fuse_copy_pages(struct fuse_copy_state *cs, unsigned nbytes,
|
|
|
|
int zeroing)
|
|
|
|
{
|
|
|
|
unsigned i;
|
|
|
|
struct fuse_req *req = cs->req;
|
|
|
|
|
|
|
|
for (i = 0; i < req->num_pages && (nbytes || zeroing); i++) {
|
2010-05-25 15:06:07 +02:00
|
|
|
int err;
|
2012-10-26 19:49:33 +04:00
|
|
|
unsigned offset = req->page_descs[i].offset;
|
|
|
|
unsigned count = min(nbytes, req->page_descs[i].length);
|
2010-05-25 15:06:07 +02:00
|
|
|
|
|
|
|
err = fuse_copy_page(cs, &req->pages[i], offset, count,
|
|
|
|
zeroing);
|
2005-09-09 13:10:27 -07:00
|
|
|
if (err)
|
|
|
|
return err;
|
|
|
|
|
|
|
|
nbytes -= count;
|
|
|
|
}
|
|
|
|
return 0;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* Copy a single argument in the request to/from userspace buffer */
|
|
|
|
static int fuse_copy_one(struct fuse_copy_state *cs, void *val, unsigned size)
|
|
|
|
{
|
|
|
|
while (size) {
|
2008-11-26 12:03:54 +01:00
|
|
|
if (!cs->len) {
|
|
|
|
int err = fuse_copy_fill(cs);
|
|
|
|
if (err)
|
|
|
|
return err;
|
|
|
|
}
|
2005-09-09 13:10:27 -07:00
|
|
|
fuse_copy_do(cs, &val, &size);
|
|
|
|
}
|
|
|
|
return 0;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* Copy request arguments to/from userspace buffer */
|
|
|
|
static int fuse_copy_args(struct fuse_copy_state *cs, unsigned numargs,
|
|
|
|
unsigned argpages, struct fuse_arg *args,
|
|
|
|
int zeroing)
|
|
|
|
{
|
|
|
|
int err = 0;
|
|
|
|
unsigned i;
|
|
|
|
|
|
|
|
for (i = 0; !err && i < numargs; i++) {
|
|
|
|
struct fuse_arg *arg = &args[i];
|
|
|
|
if (i == numargs - 1 && argpages)
|
|
|
|
err = fuse_copy_pages(cs, arg->size, zeroing);
|
|
|
|
else
|
|
|
|
err = fuse_copy_one(cs, arg->value, arg->size);
|
|
|
|
}
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
2010-12-07 20:16:56 +01:00
|
|
|
static int forget_pending(struct fuse_conn *fc)
|
|
|
|
{
|
|
|
|
return fc->forget_list_head.next != NULL;
|
|
|
|
}
|
|
|
|
|
2006-06-25 05:48:54 -07:00
|
|
|
static int request_pending(struct fuse_conn *fc)
|
|
|
|
{
|
2010-12-07 20:16:56 +01:00
|
|
|
return !list_empty(&fc->pending) || !list_empty(&fc->interrupts) ||
|
|
|
|
forget_pending(fc);
|
2006-06-25 05:48:54 -07:00
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
/* Wait until a request is available on the pending list */
|
|
|
|
static void request_wait(struct fuse_conn *fc)
|
2010-09-07 13:42:41 +02:00
|
|
|
__releases(fc->lock)
|
|
|
|
__acquires(fc->lock)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
DECLARE_WAITQUEUE(wait, current);
|
|
|
|
|
|
|
|
add_wait_queue_exclusive(&fc->waitq, &wait);
|
2006-06-25 05:48:54 -07:00
|
|
|
while (fc->connected && !request_pending(fc)) {
|
2005-09-09 13:10:27 -07:00
|
|
|
set_current_state(TASK_INTERRUPTIBLE);
|
|
|
|
if (signal_pending(current))
|
|
|
|
break;
|
|
|
|
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
schedule();
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
set_current_state(TASK_RUNNING);
|
|
|
|
remove_wait_queue(&fc->waitq, &wait);
|
|
|
|
}
|
|
|
|
|
2006-06-25 05:48:54 -07:00
|
|
|
/*
|
|
|
|
* Transfer an interrupt request to userspace
|
|
|
|
*
|
|
|
|
* Unlike other requests this is assembled on demand, without a need
|
|
|
|
* to allocate a separate fuse_req structure.
|
|
|
|
*
|
|
|
|
* Called with fc->lock held, releases it
|
|
|
|
*/
|
2010-05-25 15:06:07 +02:00
|
|
|
static int fuse_read_interrupt(struct fuse_conn *fc, struct fuse_copy_state *cs,
|
|
|
|
size_t nbytes, struct fuse_req *req)
|
2010-09-07 13:42:41 +02:00
|
|
|
__releases(fc->lock)
|
2006-06-25 05:48:54 -07:00
|
|
|
{
|
|
|
|
struct fuse_in_header ih;
|
|
|
|
struct fuse_interrupt_in arg;
|
|
|
|
unsigned reqsize = sizeof(ih) + sizeof(arg);
|
|
|
|
int err;
|
|
|
|
|
|
|
|
list_del_init(&req->intr_entry);
|
|
|
|
req->intr_unique = fuse_get_unique(fc);
|
|
|
|
memset(&ih, 0, sizeof(ih));
|
|
|
|
memset(&arg, 0, sizeof(arg));
|
|
|
|
ih.len = reqsize;
|
|
|
|
ih.opcode = FUSE_INTERRUPT;
|
|
|
|
ih.unique = req->intr_unique;
|
|
|
|
arg.unique = req->in.h.unique;
|
|
|
|
|
|
|
|
spin_unlock(&fc->lock);
|
2010-05-25 15:06:07 +02:00
|
|
|
if (nbytes < reqsize)
|
2006-06-25 05:48:54 -07:00
|
|
|
return -EINVAL;
|
|
|
|
|
2010-05-25 15:06:07 +02:00
|
|
|
err = fuse_copy_one(cs, &ih, sizeof(ih));
|
2006-06-25 05:48:54 -07:00
|
|
|
if (!err)
|
2010-05-25 15:06:07 +02:00
|
|
|
err = fuse_copy_one(cs, &arg, sizeof(arg));
|
|
|
|
fuse_copy_finish(cs);
|
2006-06-25 05:48:54 -07:00
|
|
|
|
|
|
|
return err ? err : reqsize;
|
|
|
|
}
|
|
|
|
|
2010-12-07 20:16:56 +01:00
|
|
|
static struct fuse_forget_link *dequeue_forget(struct fuse_conn *fc,
|
|
|
|
unsigned max,
|
|
|
|
unsigned *countp)
|
2010-12-07 20:16:56 +01:00
|
|
|
{
|
2010-12-07 20:16:56 +01:00
|
|
|
struct fuse_forget_link *head = fc->forget_list_head.next;
|
|
|
|
struct fuse_forget_link **newhead = &head;
|
|
|
|
unsigned count;
|
2010-12-07 20:16:56 +01:00
|
|
|
|
2010-12-07 20:16:56 +01:00
|
|
|
for (count = 0; *newhead != NULL && count < max; count++)
|
|
|
|
newhead = &(*newhead)->next;
|
|
|
|
|
|
|
|
fc->forget_list_head.next = *newhead;
|
|
|
|
*newhead = NULL;
|
2010-12-07 20:16:56 +01:00
|
|
|
if (fc->forget_list_head.next == NULL)
|
|
|
|
fc->forget_list_tail = &fc->forget_list_head;
|
|
|
|
|
2010-12-07 20:16:56 +01:00
|
|
|
if (countp != NULL)
|
|
|
|
*countp = count;
|
|
|
|
|
|
|
|
return head;
|
2010-12-07 20:16:56 +01:00
|
|
|
}
|
|
|
|
|
|
|
|
static int fuse_read_single_forget(struct fuse_conn *fc,
|
|
|
|
struct fuse_copy_state *cs,
|
|
|
|
size_t nbytes)
|
|
|
|
__releases(fc->lock)
|
|
|
|
{
|
|
|
|
int err;
|
2010-12-07 20:16:56 +01:00
|
|
|
struct fuse_forget_link *forget = dequeue_forget(fc, 1, NULL);
|
2010-12-07 20:16:56 +01:00
|
|
|
struct fuse_forget_in arg = {
|
2010-12-07 20:16:56 +01:00
|
|
|
.nlookup = forget->forget_one.nlookup,
|
2010-12-07 20:16:56 +01:00
|
|
|
};
|
|
|
|
struct fuse_in_header ih = {
|
|
|
|
.opcode = FUSE_FORGET,
|
2010-12-07 20:16:56 +01:00
|
|
|
.nodeid = forget->forget_one.nodeid,
|
2010-12-07 20:16:56 +01:00
|
|
|
.unique = fuse_get_unique(fc),
|
|
|
|
.len = sizeof(ih) + sizeof(arg),
|
|
|
|
};
|
|
|
|
|
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
kfree(forget);
|
|
|
|
if (nbytes < ih.len)
|
|
|
|
return -EINVAL;
|
|
|
|
|
|
|
|
err = fuse_copy_one(cs, &ih, sizeof(ih));
|
|
|
|
if (!err)
|
|
|
|
err = fuse_copy_one(cs, &arg, sizeof(arg));
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
|
|
|
|
if (err)
|
|
|
|
return err;
|
|
|
|
|
|
|
|
return ih.len;
|
|
|
|
}
|
|
|
|
|
2010-12-07 20:16:56 +01:00
|
|
|
static int fuse_read_batch_forget(struct fuse_conn *fc,
|
|
|
|
struct fuse_copy_state *cs, size_t nbytes)
|
|
|
|
__releases(fc->lock)
|
|
|
|
{
|
|
|
|
int err;
|
|
|
|
unsigned max_forgets;
|
|
|
|
unsigned count;
|
|
|
|
struct fuse_forget_link *head;
|
|
|
|
struct fuse_batch_forget_in arg = { .count = 0 };
|
|
|
|
struct fuse_in_header ih = {
|
|
|
|
.opcode = FUSE_BATCH_FORGET,
|
|
|
|
.unique = fuse_get_unique(fc),
|
|
|
|
.len = sizeof(ih) + sizeof(arg),
|
|
|
|
};
|
|
|
|
|
|
|
|
if (nbytes < ih.len) {
|
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
return -EINVAL;
|
|
|
|
}
|
|
|
|
|
|
|
|
max_forgets = (nbytes - ih.len) / sizeof(struct fuse_forget_one);
|
|
|
|
head = dequeue_forget(fc, max_forgets, &count);
|
|
|
|
spin_unlock(&fc->lock);
|
|
|
|
|
|
|
|
arg.count = count;
|
|
|
|
ih.len += count * sizeof(struct fuse_forget_one);
|
|
|
|
err = fuse_copy_one(cs, &ih, sizeof(ih));
|
|
|
|
if (!err)
|
|
|
|
err = fuse_copy_one(cs, &arg, sizeof(arg));
|
|
|
|
|
|
|
|
while (head) {
|
|
|
|
struct fuse_forget_link *forget = head;
|
|
|
|
|
|
|
|
if (!err) {
|
|
|
|
err = fuse_copy_one(cs, &forget->forget_one,
|
|
|
|
sizeof(forget->forget_one));
|
|
|
|
}
|
|
|
|
head = forget->next;
|
|
|
|
kfree(forget);
|
|
|
|
}
|
|
|
|
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
|
|
|
|
if (err)
|
|
|
|
return err;
|
|
|
|
|
|
|
|
return ih.len;
|
|
|
|
}
|
|
|
|
|
|
|
|
static int fuse_read_forget(struct fuse_conn *fc, struct fuse_copy_state *cs,
|
|
|
|
size_t nbytes)
|
|
|
|
__releases(fc->lock)
|
|
|
|
{
|
|
|
|
if (fc->minor < 16 || fc->forget_list_head.next->next == NULL)
|
|
|
|
return fuse_read_single_forget(fc, cs, nbytes);
|
|
|
|
else
|
|
|
|
return fuse_read_batch_forget(fc, cs, nbytes);
|
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
/*
|
|
|
|
* Read a single request into the userspace filesystem's buffer. This
|
|
|
|
* function waits until a request is available, then removes it from
|
|
|
|
* the pending list and copies request data to userspace buffer. If
|
2006-06-25 05:48:53 -07:00
|
|
|
* no reply is needed (FORGET) or request has been aborted or there
|
|
|
|
* was an error during the copying then it's finished by calling
|
2005-09-09 13:10:27 -07:00
|
|
|
* request_end(). Otherwise add it to the processing list, and set
|
|
|
|
* the 'sent' flag.
|
|
|
|
*/
|
2010-05-25 15:06:07 +02:00
|
|
|
static ssize_t fuse_dev_do_read(struct fuse_conn *fc, struct file *file,
|
|
|
|
struct fuse_copy_state *cs, size_t nbytes)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
int err;
|
|
|
|
struct fuse_req *req;
|
|
|
|
struct fuse_in *in;
|
|
|
|
unsigned reqsize;
|
|
|
|
|
2006-01-06 00:19:40 -08:00
|
|
|
restart:
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2006-04-10 22:54:53 -07:00
|
|
|
err = -EAGAIN;
|
|
|
|
if ((file->f_flags & O_NONBLOCK) && fc->connected &&
|
2006-06-25 05:48:54 -07:00
|
|
|
!request_pending(fc))
|
2006-04-10 22:54:53 -07:00
|
|
|
goto err_unlock;
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
request_wait(fc);
|
|
|
|
err = -ENODEV;
|
2006-01-16 22:14:34 -08:00
|
|
|
if (!fc->connected)
|
2005-09-09 13:10:27 -07:00
|
|
|
goto err_unlock;
|
|
|
|
err = -ERESTARTSYS;
|
2006-06-25 05:48:54 -07:00
|
|
|
if (!request_pending(fc))
|
2005-09-09 13:10:27 -07:00
|
|
|
goto err_unlock;
|
|
|
|
|
2006-06-25 05:48:54 -07:00
|
|
|
if (!list_empty(&fc->interrupts)) {
|
|
|
|
req = list_entry(fc->interrupts.next, struct fuse_req,
|
|
|
|
intr_entry);
|
2010-05-25 15:06:07 +02:00
|
|
|
return fuse_read_interrupt(fc, cs, nbytes, req);
|
2006-06-25 05:48:54 -07:00
|
|
|
}
|
|
|
|
|
2010-12-07 20:16:56 +01:00
|
|
|
if (forget_pending(fc)) {
|
|
|
|
if (list_empty(&fc->pending) || fc->forget_batch-- > 0)
|
2010-12-07 20:16:56 +01:00
|
|
|
return fuse_read_forget(fc, cs, nbytes);
|
2010-12-07 20:16:56 +01:00
|
|
|
|
|
|
|
if (fc->forget_batch <= -8)
|
|
|
|
fc->forget_batch = 16;
|
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
req = list_entry(fc->pending.next, struct fuse_req, list);
|
2006-01-16 22:14:31 -08:00
|
|
|
req->state = FUSE_REQ_READING;
|
2006-01-16 22:14:31 -08:00
|
|
|
list_move(&req->list, &fc->io);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
|
|
|
in = &req->in;
|
2006-01-06 00:19:40 -08:00
|
|
|
reqsize = in->h.len;
|
|
|
|
/* If request is too large, reply with an error and restart the read */
|
2010-05-25 15:06:07 +02:00
|
|
|
if (nbytes < reqsize) {
|
2006-01-06 00:19:40 -08:00
|
|
|
req->out.h.error = -EIO;
|
|
|
|
/* SETXATTR is special, since it may contain too large data */
|
|
|
|
if (in->h.opcode == FUSE_SETXATTR)
|
|
|
|
req->out.h.error = -E2BIG;
|
|
|
|
request_end(fc, req);
|
|
|
|
goto restart;
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2010-05-25 15:06:07 +02:00
|
|
|
cs->req = req;
|
|
|
|
err = fuse_copy_one(cs, &in->h, sizeof(in->h));
|
2006-01-06 00:19:40 -08:00
|
|
|
if (!err)
|
2010-05-25 15:06:07 +02:00
|
|
|
err = fuse_copy_args(cs, in->numargs, in->argpages,
|
2006-01-06 00:19:40 -08:00
|
|
|
(struct fuse_arg *) in->args, 0);
|
2010-05-25 15:06:07 +02:00
|
|
|
fuse_copy_finish(cs);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
req->locked = 0;
|
2007-10-16 23:31:05 -07:00
|
|
|
if (req->aborted) {
|
|
|
|
request_end(fc, req);
|
|
|
|
return -ENODEV;
|
|
|
|
}
|
2005-09-09 13:10:27 -07:00
|
|
|
if (err) {
|
2007-10-16 23:31:05 -07:00
|
|
|
req->out.h.error = -EIO;
|
2005-09-09 13:10:27 -07:00
|
|
|
request_end(fc, req);
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
if (!req->isreply)
|
|
|
|
request_end(fc, req);
|
|
|
|
else {
|
2006-01-16 22:14:31 -08:00
|
|
|
req->state = FUSE_REQ_SENT;
|
2006-01-16 22:14:31 -08:00
|
|
|
list_move_tail(&req->list, &fc->processing);
|
2006-06-25 05:48:54 -07:00
|
|
|
if (req->interrupted)
|
|
|
|
queue_interrupt(fc, req);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
return reqsize;
|
|
|
|
|
|
|
|
err_unlock:
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
2010-05-25 15:06:07 +02:00
|
|
|
static ssize_t fuse_dev_read(struct kiocb *iocb, const struct iovec *iov,
|
|
|
|
unsigned long nr_segs, loff_t pos)
|
|
|
|
{
|
|
|
|
struct fuse_copy_state cs;
|
|
|
|
struct file *file = iocb->ki_filp;
|
|
|
|
struct fuse_conn *fc = fuse_get_conn(file);
|
|
|
|
if (!fc)
|
|
|
|
return -EPERM;
|
|
|
|
|
|
|
|
fuse_copy_init(&cs, fc, 1, iov, nr_segs);
|
|
|
|
|
|
|
|
return fuse_dev_do_read(fc, file, &cs, iov_length(iov, nr_segs));
|
|
|
|
}
|
|
|
|
|
|
|
|
static ssize_t fuse_dev_splice_read(struct file *in, loff_t *ppos,
|
|
|
|
struct pipe_inode_info *pipe,
|
|
|
|
size_t len, unsigned int flags)
|
|
|
|
{
|
|
|
|
int ret;
|
|
|
|
int page_nr = 0;
|
|
|
|
int do_wakeup = 0;
|
|
|
|
struct pipe_buffer *bufs;
|
|
|
|
struct fuse_copy_state cs;
|
|
|
|
struct fuse_conn *fc = fuse_get_conn(in);
|
|
|
|
if (!fc)
|
|
|
|
return -EPERM;
|
|
|
|
|
2010-12-07 20:16:56 +01:00
|
|
|
bufs = kmalloc(pipe->buffers * sizeof(struct pipe_buffer), GFP_KERNEL);
|
2010-05-25 15:06:07 +02:00
|
|
|
if (!bufs)
|
|
|
|
return -ENOMEM;
|
|
|
|
|
|
|
|
fuse_copy_init(&cs, fc, 1, NULL, 0);
|
|
|
|
cs.pipebufs = bufs;
|
|
|
|
cs.pipe = pipe;
|
|
|
|
ret = fuse_dev_do_read(fc, in, &cs, len);
|
|
|
|
if (ret < 0)
|
|
|
|
goto out;
|
|
|
|
|
|
|
|
ret = 0;
|
|
|
|
pipe_lock(pipe);
|
|
|
|
|
|
|
|
if (!pipe->readers) {
|
|
|
|
send_sig(SIGPIPE, current, 0);
|
|
|
|
if (!ret)
|
|
|
|
ret = -EPIPE;
|
|
|
|
goto out_unlock;
|
|
|
|
}
|
|
|
|
|
|
|
|
if (pipe->nrbufs + cs.nr_segs > pipe->buffers) {
|
|
|
|
ret = -EIO;
|
|
|
|
goto out_unlock;
|
|
|
|
}
|
|
|
|
|
|
|
|
while (page_nr < cs.nr_segs) {
|
|
|
|
int newbuf = (pipe->curbuf + pipe->nrbufs) & (pipe->buffers - 1);
|
|
|
|
struct pipe_buffer *buf = pipe->bufs + newbuf;
|
|
|
|
|
|
|
|
buf->page = bufs[page_nr].page;
|
|
|
|
buf->offset = bufs[page_nr].offset;
|
|
|
|
buf->len = bufs[page_nr].len;
|
2014-01-22 19:36:57 +01:00
|
|
|
/*
|
|
|
|
* Need to be careful about this. Having buf->ops in module
|
|
|
|
* code can Oops if the buffer persists after module unload.
|
|
|
|
*/
|
|
|
|
buf->ops = &nosteal_pipe_buf_ops;
|
2010-05-25 15:06:07 +02:00
|
|
|
|
|
|
|
pipe->nrbufs++;
|
|
|
|
page_nr++;
|
|
|
|
ret += buf->len;
|
|
|
|
|
2013-03-21 11:01:38 -04:00
|
|
|
if (pipe->files)
|
2010-05-25 15:06:07 +02:00
|
|
|
do_wakeup = 1;
|
|
|
|
}
|
|
|
|
|
|
|
|
out_unlock:
|
|
|
|
pipe_unlock(pipe);
|
|
|
|
|
|
|
|
if (do_wakeup) {
|
|
|
|
smp_mb();
|
|
|
|
if (waitqueue_active(&pipe->wait))
|
|
|
|
wake_up_interruptible(&pipe->wait);
|
|
|
|
kill_fasync(&pipe->fasync_readers, SIGIO, POLL_IN);
|
|
|
|
}
|
|
|
|
|
|
|
|
out:
|
|
|
|
for (; page_nr < cs.nr_segs; page_nr++)
|
|
|
|
page_cache_release(bufs[page_nr].page);
|
|
|
|
|
|
|
|
kfree(bufs);
|
|
|
|
return ret;
|
|
|
|
}
|
|
|
|
|
2008-11-26 12:03:55 +01:00
|
|
|
static int fuse_notify_poll(struct fuse_conn *fc, unsigned int size,
|
|
|
|
struct fuse_copy_state *cs)
|
|
|
|
{
|
|
|
|
struct fuse_notify_poll_wakeup_out outarg;
|
2009-01-26 15:00:59 +01:00
|
|
|
int err = -EINVAL;
|
2008-11-26 12:03:55 +01:00
|
|
|
|
|
|
|
if (size != sizeof(outarg))
|
2009-01-26 15:00:59 +01:00
|
|
|
goto err;
|
2008-11-26 12:03:55 +01:00
|
|
|
|
|
|
|
err = fuse_copy_one(cs, &outarg, sizeof(outarg));
|
|
|
|
if (err)
|
2009-01-26 15:00:59 +01:00
|
|
|
goto err;
|
2008-11-26 12:03:55 +01:00
|
|
|
|
2009-01-26 15:00:59 +01:00
|
|
|
fuse_copy_finish(cs);
|
2008-11-26 12:03:55 +01:00
|
|
|
return fuse_notify_poll_wakeup(fc, &outarg);
|
2009-01-26 15:00:59 +01:00
|
|
|
|
|
|
|
err:
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
return err;
|
2008-11-26 12:03:55 +01:00
|
|
|
}
|
|
|
|
|
2009-05-31 11:13:57 -04:00
|
|
|
static int fuse_notify_inval_inode(struct fuse_conn *fc, unsigned int size,
|
|
|
|
struct fuse_copy_state *cs)
|
|
|
|
{
|
|
|
|
struct fuse_notify_inval_inode_out outarg;
|
|
|
|
int err = -EINVAL;
|
|
|
|
|
|
|
|
if (size != sizeof(outarg))
|
|
|
|
goto err;
|
|
|
|
|
|
|
|
err = fuse_copy_one(cs, &outarg, sizeof(outarg));
|
|
|
|
if (err)
|
|
|
|
goto err;
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
|
|
|
|
down_read(&fc->killsb);
|
|
|
|
err = -ENOENT;
|
2010-02-05 12:08:31 +01:00
|
|
|
if (fc->sb) {
|
|
|
|
err = fuse_reverse_inval_inode(fc->sb, outarg.ino,
|
|
|
|
outarg.off, outarg.len);
|
|
|
|
}
|
2009-05-31 11:13:57 -04:00
|
|
|
up_read(&fc->killsb);
|
|
|
|
return err;
|
|
|
|
|
|
|
|
err:
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
|
|
|
static int fuse_notify_inval_entry(struct fuse_conn *fc, unsigned int size,
|
|
|
|
struct fuse_copy_state *cs)
|
|
|
|
{
|
|
|
|
struct fuse_notify_inval_entry_out outarg;
|
2009-12-30 18:37:13 +08:00
|
|
|
int err = -ENOMEM;
|
|
|
|
char *buf;
|
2009-05-31 11:13:57 -04:00
|
|
|
struct qstr name;
|
|
|
|
|
2009-12-30 18:37:13 +08:00
|
|
|
buf = kzalloc(FUSE_NAME_MAX + 1, GFP_KERNEL);
|
|
|
|
if (!buf)
|
|
|
|
goto err;
|
|
|
|
|
|
|
|
err = -EINVAL;
|
2009-05-31 11:13:57 -04:00
|
|
|
if (size < sizeof(outarg))
|
|
|
|
goto err;
|
|
|
|
|
|
|
|
err = fuse_copy_one(cs, &outarg, sizeof(outarg));
|
|
|
|
if (err)
|
|
|
|
goto err;
|
|
|
|
|
|
|
|
err = -ENAMETOOLONG;
|
|
|
|
if (outarg.namelen > FUSE_NAME_MAX)
|
|
|
|
goto err;
|
|
|
|
|
2011-08-24 10:20:17 +02:00
|
|
|
err = -EINVAL;
|
|
|
|
if (size != sizeof(outarg) + outarg.namelen + 1)
|
|
|
|
goto err;
|
|
|
|
|
2009-05-31 11:13:57 -04:00
|
|
|
name.name = buf;
|
|
|
|
name.len = outarg.namelen;
|
|
|
|
err = fuse_copy_one(cs, buf, outarg.namelen + 1);
|
|
|
|
if (err)
|
|
|
|
goto err;
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
buf[outarg.namelen] = 0;
|
|
|
|
name.hash = full_name_hash(name.name, name.len);
|
|
|
|
|
|
|
|
down_read(&fc->killsb);
|
|
|
|
err = -ENOENT;
|
2010-02-05 12:08:31 +01:00
|
|
|
if (fc->sb)
|
2011-12-06 21:50:06 +01:00
|
|
|
err = fuse_reverse_inval_entry(fc->sb, outarg.parent, 0, &name);
|
|
|
|
up_read(&fc->killsb);
|
|
|
|
kfree(buf);
|
|
|
|
return err;
|
|
|
|
|
|
|
|
err:
|
|
|
|
kfree(buf);
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
|
|
|
static int fuse_notify_delete(struct fuse_conn *fc, unsigned int size,
|
|
|
|
struct fuse_copy_state *cs)
|
|
|
|
{
|
|
|
|
struct fuse_notify_delete_out outarg;
|
|
|
|
int err = -ENOMEM;
|
|
|
|
char *buf;
|
|
|
|
struct qstr name;
|
|
|
|
|
|
|
|
buf = kzalloc(FUSE_NAME_MAX + 1, GFP_KERNEL);
|
|
|
|
if (!buf)
|
|
|
|
goto err;
|
|
|
|
|
|
|
|
err = -EINVAL;
|
|
|
|
if (size < sizeof(outarg))
|
|
|
|
goto err;
|
|
|
|
|
|
|
|
err = fuse_copy_one(cs, &outarg, sizeof(outarg));
|
|
|
|
if (err)
|
|
|
|
goto err;
|
|
|
|
|
|
|
|
err = -ENAMETOOLONG;
|
|
|
|
if (outarg.namelen > FUSE_NAME_MAX)
|
|
|
|
goto err;
|
|
|
|
|
|
|
|
err = -EINVAL;
|
|
|
|
if (size != sizeof(outarg) + outarg.namelen + 1)
|
|
|
|
goto err;
|
|
|
|
|
|
|
|
name.name = buf;
|
|
|
|
name.len = outarg.namelen;
|
|
|
|
err = fuse_copy_one(cs, buf, outarg.namelen + 1);
|
|
|
|
if (err)
|
|
|
|
goto err;
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
buf[outarg.namelen] = 0;
|
|
|
|
name.hash = full_name_hash(name.name, name.len);
|
|
|
|
|
|
|
|
down_read(&fc->killsb);
|
|
|
|
err = -ENOENT;
|
|
|
|
if (fc->sb)
|
|
|
|
err = fuse_reverse_inval_entry(fc->sb, outarg.parent,
|
|
|
|
outarg.child, &name);
|
2009-05-31 11:13:57 -04:00
|
|
|
up_read(&fc->killsb);
|
2009-12-30 18:37:13 +08:00
|
|
|
kfree(buf);
|
2009-05-31 11:13:57 -04:00
|
|
|
return err;
|
|
|
|
|
|
|
|
err:
|
2009-12-30 18:37:13 +08:00
|
|
|
kfree(buf);
|
2009-05-31 11:13:57 -04:00
|
|
|
fuse_copy_finish(cs);
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
2010-07-12 14:41:40 +02:00
|
|
|
static int fuse_notify_store(struct fuse_conn *fc, unsigned int size,
|
|
|
|
struct fuse_copy_state *cs)
|
|
|
|
{
|
|
|
|
struct fuse_notify_store_out outarg;
|
|
|
|
struct inode *inode;
|
|
|
|
struct address_space *mapping;
|
|
|
|
u64 nodeid;
|
|
|
|
int err;
|
|
|
|
pgoff_t index;
|
|
|
|
unsigned int offset;
|
|
|
|
unsigned int num;
|
|
|
|
loff_t file_size;
|
|
|
|
loff_t end;
|
|
|
|
|
|
|
|
err = -EINVAL;
|
|
|
|
if (size < sizeof(outarg))
|
|
|
|
goto out_finish;
|
|
|
|
|
|
|
|
err = fuse_copy_one(cs, &outarg, sizeof(outarg));
|
|
|
|
if (err)
|
|
|
|
goto out_finish;
|
|
|
|
|
|
|
|
err = -EINVAL;
|
|
|
|
if (size - sizeof(outarg) != outarg.size)
|
|
|
|
goto out_finish;
|
|
|
|
|
|
|
|
nodeid = outarg.nodeid;
|
|
|
|
|
|
|
|
down_read(&fc->killsb);
|
|
|
|
|
|
|
|
err = -ENOENT;
|
|
|
|
if (!fc->sb)
|
|
|
|
goto out_up_killsb;
|
|
|
|
|
|
|
|
inode = ilookup5(fc->sb, nodeid, fuse_inode_eq, &nodeid);
|
|
|
|
if (!inode)
|
|
|
|
goto out_up_killsb;
|
|
|
|
|
|
|
|
mapping = inode->i_mapping;
|
|
|
|
index = outarg.offset >> PAGE_CACHE_SHIFT;
|
|
|
|
offset = outarg.offset & ~PAGE_CACHE_MASK;
|
|
|
|
file_size = i_size_read(inode);
|
|
|
|
end = outarg.offset + outarg.size;
|
|
|
|
if (end > file_size) {
|
|
|
|
file_size = end;
|
|
|
|
fuse_write_update_size(inode, file_size);
|
|
|
|
}
|
|
|
|
|
|
|
|
num = outarg.size;
|
|
|
|
while (num) {
|
|
|
|
struct page *page;
|
|
|
|
unsigned int this_num;
|
|
|
|
|
|
|
|
err = -ENOMEM;
|
|
|
|
page = find_or_create_page(mapping, index,
|
|
|
|
mapping_gfp_mask(mapping));
|
|
|
|
if (!page)
|
|
|
|
goto out_iput;
|
|
|
|
|
|
|
|
this_num = min_t(unsigned, num, PAGE_CACHE_SIZE - offset);
|
|
|
|
err = fuse_copy_page(cs, &page, offset, this_num, 0);
|
2014-01-22 19:36:58 +01:00
|
|
|
if (!err && offset == 0 &&
|
|
|
|
(this_num == PAGE_CACHE_SIZE || file_size == end))
|
2010-07-12 14:41:40 +02:00
|
|
|
SetPageUptodate(page);
|
|
|
|
unlock_page(page);
|
|
|
|
page_cache_release(page);
|
|
|
|
|
|
|
|
if (err)
|
|
|
|
goto out_iput;
|
|
|
|
|
|
|
|
num -= this_num;
|
|
|
|
offset = 0;
|
|
|
|
index++;
|
|
|
|
}
|
|
|
|
|
|
|
|
err = 0;
|
|
|
|
|
|
|
|
out_iput:
|
|
|
|
iput(inode);
|
|
|
|
out_up_killsb:
|
|
|
|
up_read(&fc->killsb);
|
|
|
|
out_finish:
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
2010-07-12 14:41:40 +02:00
|
|
|
static void fuse_retrieve_end(struct fuse_conn *fc, struct fuse_req *req)
|
|
|
|
{
|
2014-06-04 16:10:22 -07:00
|
|
|
release_pages(req->pages, req->num_pages, false);
|
2010-07-12 14:41:40 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
static int fuse_retrieve(struct fuse_conn *fc, struct inode *inode,
|
|
|
|
struct fuse_notify_retrieve_out *outarg)
|
|
|
|
{
|
|
|
|
int err;
|
|
|
|
struct address_space *mapping = inode->i_mapping;
|
|
|
|
struct fuse_req *req;
|
|
|
|
pgoff_t index;
|
|
|
|
loff_t file_size;
|
|
|
|
unsigned int num;
|
|
|
|
unsigned int offset;
|
2010-09-30 22:06:21 +02:00
|
|
|
size_t total_len = 0;
|
2012-10-26 19:48:42 +04:00
|
|
|
int num_pages;
|
2010-07-12 14:41:40 +02:00
|
|
|
|
2012-10-26 19:48:42 +04:00
|
|
|
offset = outarg->offset & ~PAGE_CACHE_MASK;
|
|
|
|
file_size = i_size_read(inode);
|
|
|
|
|
|
|
|
num = outarg->size;
|
|
|
|
if (outarg->offset > file_size)
|
|
|
|
num = 0;
|
|
|
|
else if (outarg->offset + num > file_size)
|
|
|
|
num = file_size - outarg->offset;
|
|
|
|
|
|
|
|
num_pages = (num + offset + PAGE_SIZE - 1) >> PAGE_SHIFT;
|
|
|
|
num_pages = min(num_pages, FUSE_MAX_PAGES_PER_REQ);
|
|
|
|
|
|
|
|
req = fuse_get_req(fc, num_pages);
|
2010-07-12 14:41:40 +02:00
|
|
|
if (IS_ERR(req))
|
|
|
|
return PTR_ERR(req);
|
|
|
|
|
|
|
|
req->in.h.opcode = FUSE_NOTIFY_REPLY;
|
|
|
|
req->in.h.nodeid = outarg->nodeid;
|
|
|
|
req->in.numargs = 2;
|
|
|
|
req->in.argpages = 1;
|
2012-10-26 19:49:24 +04:00
|
|
|
req->page_descs[0].offset = offset;
|
2010-07-12 14:41:40 +02:00
|
|
|
req->end = fuse_retrieve_end;
|
|
|
|
|
|
|
|
index = outarg->offset >> PAGE_CACHE_SHIFT;
|
|
|
|
|
2012-10-26 19:48:42 +04:00
|
|
|
while (num && req->num_pages < num_pages) {
|
2010-07-12 14:41:40 +02:00
|
|
|
struct page *page;
|
|
|
|
unsigned int this_num;
|
|
|
|
|
|
|
|
page = find_get_page(mapping, index);
|
|
|
|
if (!page)
|
|
|
|
break;
|
|
|
|
|
|
|
|
this_num = min_t(unsigned, num, PAGE_CACHE_SIZE - offset);
|
|
|
|
req->pages[req->num_pages] = page;
|
2012-10-26 19:49:33 +04:00
|
|
|
req->page_descs[req->num_pages].length = this_num;
|
2010-07-12 14:41:40 +02:00
|
|
|
req->num_pages++;
|
|
|
|
|
2012-09-04 18:45:54 +02:00
|
|
|
offset = 0;
|
2010-07-12 14:41:40 +02:00
|
|
|
num -= this_num;
|
|
|
|
total_len += this_num;
|
2011-12-13 10:36:59 +01:00
|
|
|
index++;
|
2010-07-12 14:41:40 +02:00
|
|
|
}
|
|
|
|
req->misc.retrieve_in.offset = outarg->offset;
|
|
|
|
req->misc.retrieve_in.size = total_len;
|
|
|
|
req->in.args[0].size = sizeof(req->misc.retrieve_in);
|
|
|
|
req->in.args[0].value = &req->misc.retrieve_in;
|
|
|
|
req->in.args[1].size = total_len;
|
|
|
|
|
|
|
|
err = fuse_request_send_notify_reply(fc, req, outarg->notify_unique);
|
|
|
|
if (err)
|
|
|
|
fuse_retrieve_end(fc, req);
|
|
|
|
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
|
|
|
static int fuse_notify_retrieve(struct fuse_conn *fc, unsigned int size,
|
|
|
|
struct fuse_copy_state *cs)
|
|
|
|
{
|
|
|
|
struct fuse_notify_retrieve_out outarg;
|
|
|
|
struct inode *inode;
|
|
|
|
int err;
|
|
|
|
|
|
|
|
err = -EINVAL;
|
|
|
|
if (size != sizeof(outarg))
|
|
|
|
goto copy_finish;
|
|
|
|
|
|
|
|
err = fuse_copy_one(cs, &outarg, sizeof(outarg));
|
|
|
|
if (err)
|
|
|
|
goto copy_finish;
|
|
|
|
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
|
|
|
|
down_read(&fc->killsb);
|
|
|
|
err = -ENOENT;
|
|
|
|
if (fc->sb) {
|
|
|
|
u64 nodeid = outarg.nodeid;
|
|
|
|
|
|
|
|
inode = ilookup5(fc->sb, nodeid, fuse_inode_eq, &nodeid);
|
|
|
|
if (inode) {
|
|
|
|
err = fuse_retrieve(fc, inode, &outarg);
|
|
|
|
iput(inode);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
up_read(&fc->killsb);
|
|
|
|
|
|
|
|
return err;
|
|
|
|
|
|
|
|
copy_finish:
|
|
|
|
fuse_copy_finish(cs);
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
2008-11-26 12:03:55 +01:00
|
|
|
static int fuse_notify(struct fuse_conn *fc, enum fuse_notify_code code,
|
|
|
|
unsigned int size, struct fuse_copy_state *cs)
|
|
|
|
{
|
|
|
|
switch (code) {
|
2008-11-26 12:03:55 +01:00
|
|
|
case FUSE_NOTIFY_POLL:
|
|
|
|
return fuse_notify_poll(fc, size, cs);
|
|
|
|
|
2009-05-31 11:13:57 -04:00
|
|
|
case FUSE_NOTIFY_INVAL_INODE:
|
|
|
|
return fuse_notify_inval_inode(fc, size, cs);
|
|
|
|
|
|
|
|
case FUSE_NOTIFY_INVAL_ENTRY:
|
|
|
|
return fuse_notify_inval_entry(fc, size, cs);
|
|
|
|
|
2010-07-12 14:41:40 +02:00
|
|
|
case FUSE_NOTIFY_STORE:
|
|
|
|
return fuse_notify_store(fc, size, cs);
|
|
|
|
|
2010-07-12 14:41:40 +02:00
|
|
|
case FUSE_NOTIFY_RETRIEVE:
|
|
|
|
return fuse_notify_retrieve(fc, size, cs);
|
|
|
|
|
2011-12-06 21:50:06 +01:00
|
|
|
case FUSE_NOTIFY_DELETE:
|
|
|
|
return fuse_notify_delete(fc, size, cs);
|
|
|
|
|
2008-11-26 12:03:55 +01:00
|
|
|
default:
|
2009-01-26 15:00:59 +01:00
|
|
|
fuse_copy_finish(cs);
|
2008-11-26 12:03:55 +01:00
|
|
|
return -EINVAL;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
/* Look up request on processing list by unique ID */
|
|
|
|
static struct fuse_req *request_find(struct fuse_conn *fc, u64 unique)
|
|
|
|
{
|
2013-07-30 22:50:01 -04:00
|
|
|
struct fuse_req *req;
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2013-07-30 22:50:01 -04:00
|
|
|
list_for_each_entry(req, &fc->processing, list) {
|
2006-06-25 05:48:54 -07:00
|
|
|
if (req->in.h.unique == unique || req->intr_unique == unique)
|
2005-09-09 13:10:27 -07:00
|
|
|
return req;
|
|
|
|
}
|
|
|
|
return NULL;
|
|
|
|
}
|
|
|
|
|
|
|
|
static int copy_out_args(struct fuse_copy_state *cs, struct fuse_out *out,
|
|
|
|
unsigned nbytes)
|
|
|
|
{
|
|
|
|
unsigned reqsize = sizeof(struct fuse_out_header);
|
|
|
|
|
|
|
|
if (out->h.error)
|
|
|
|
return nbytes != reqsize ? -EINVAL : 0;
|
|
|
|
|
|
|
|
reqsize += len_args(out->numargs, out->args);
|
|
|
|
|
|
|
|
if (reqsize < nbytes || (reqsize > nbytes && !out->argvar))
|
|
|
|
return -EINVAL;
|
|
|
|
else if (reqsize > nbytes) {
|
|
|
|
struct fuse_arg *lastarg = &out->args[out->numargs-1];
|
|
|
|
unsigned diffsize = reqsize - nbytes;
|
|
|
|
if (diffsize > lastarg->size)
|
|
|
|
return -EINVAL;
|
|
|
|
lastarg->size -= diffsize;
|
|
|
|
}
|
|
|
|
return fuse_copy_args(cs, out->numargs, out->argpages, out->args,
|
|
|
|
out->page_zeroing);
|
|
|
|
}
|
|
|
|
|
|
|
|
/*
|
|
|
|
* Write a single reply to a request. First the header is copied from
|
|
|
|
* the write buffer. The request is then searched on the processing
|
|
|
|
* list by the unique ID found in the header. If found, then remove
|
|
|
|
* it from the list and copy the rest of the buffer to the request.
|
|
|
|
* The request is finished by calling request_end()
|
|
|
|
*/
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
static ssize_t fuse_dev_do_write(struct fuse_conn *fc,
|
|
|
|
struct fuse_copy_state *cs, size_t nbytes)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
int err;
|
|
|
|
struct fuse_req *req;
|
|
|
|
struct fuse_out_header oh;
|
|
|
|
|
|
|
|
if (nbytes < sizeof(struct fuse_out_header))
|
|
|
|
return -EINVAL;
|
|
|
|
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
err = fuse_copy_one(cs, &oh, sizeof(oh));
|
2005-09-09 13:10:27 -07:00
|
|
|
if (err)
|
|
|
|
goto err_finish;
|
2008-11-26 12:03:55 +01:00
|
|
|
|
|
|
|
err = -EINVAL;
|
|
|
|
if (oh.len != nbytes)
|
|
|
|
goto err_finish;
|
|
|
|
|
|
|
|
/*
|
|
|
|
* Zero oh.unique indicates unsolicited notification message
|
|
|
|
* and error contains notification code.
|
|
|
|
*/
|
|
|
|
if (!oh.unique) {
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
err = fuse_notify(fc, oh.error, nbytes - sizeof(oh), cs);
|
2008-11-26 12:03:55 +01:00
|
|
|
return err ? err : nbytes;
|
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
err = -EINVAL;
|
2008-11-26 12:03:55 +01:00
|
|
|
if (oh.error <= -1000 || oh.error > 0)
|
2005-09-09 13:10:27 -07:00
|
|
|
goto err_finish;
|
|
|
|
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2006-01-16 22:14:41 -08:00
|
|
|
err = -ENOENT;
|
|
|
|
if (!fc->connected)
|
|
|
|
goto err_unlock;
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
req = request_find(fc, oh.unique);
|
|
|
|
if (!req)
|
|
|
|
goto err_unlock;
|
|
|
|
|
2006-06-25 05:48:53 -07:00
|
|
|
if (req->aborted) {
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
fuse_copy_finish(cs);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2006-01-16 22:14:25 -08:00
|
|
|
request_end(fc, req);
|
2005-09-09 13:10:27 -07:00
|
|
|
return -ENOENT;
|
|
|
|
}
|
2006-06-25 05:48:54 -07:00
|
|
|
/* Is it an interrupt reply? */
|
|
|
|
if (req->intr_unique == oh.unique) {
|
|
|
|
err = -EINVAL;
|
|
|
|
if (nbytes != sizeof(struct fuse_out_header))
|
|
|
|
goto err_unlock;
|
|
|
|
|
|
|
|
if (oh.error == -ENOSYS)
|
|
|
|
fc->no_interrupt = 1;
|
|
|
|
else if (oh.error == -EAGAIN)
|
|
|
|
queue_interrupt(fc, req);
|
|
|
|
|
|
|
|
spin_unlock(&fc->lock);
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
fuse_copy_finish(cs);
|
2006-06-25 05:48:54 -07:00
|
|
|
return nbytes;
|
|
|
|
}
|
|
|
|
|
|
|
|
req->state = FUSE_REQ_WRITING;
|
2006-01-16 22:14:31 -08:00
|
|
|
list_move(&req->list, &fc->io);
|
2005-09-09 13:10:27 -07:00
|
|
|
req->out.h = oh;
|
|
|
|
req->locked = 1;
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
cs->req = req;
|
2010-05-25 15:06:07 +02:00
|
|
|
if (!req->out.page_replace)
|
|
|
|
cs->move_pages = 0;
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
err = copy_out_args(cs, &req->out, nbytes);
|
|
|
|
fuse_copy_finish(cs);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
req->locked = 0;
|
|
|
|
if (!err) {
|
2006-06-25 05:48:53 -07:00
|
|
|
if (req->aborted)
|
2005-09-09 13:10:27 -07:00
|
|
|
err = -ENOENT;
|
2006-06-25 05:48:53 -07:00
|
|
|
} else if (!req->aborted)
|
2005-09-09 13:10:27 -07:00
|
|
|
req->out.h.error = -EIO;
|
|
|
|
request_end(fc, req);
|
|
|
|
|
|
|
|
return err ? err : nbytes;
|
|
|
|
|
|
|
|
err_unlock:
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
err_finish:
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
fuse_copy_finish(cs);
|
2005-09-09 13:10:27 -07:00
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
static ssize_t fuse_dev_write(struct kiocb *iocb, const struct iovec *iov,
|
|
|
|
unsigned long nr_segs, loff_t pos)
|
|
|
|
{
|
|
|
|
struct fuse_copy_state cs;
|
|
|
|
struct fuse_conn *fc = fuse_get_conn(iocb->ki_filp);
|
|
|
|
if (!fc)
|
|
|
|
return -EPERM;
|
|
|
|
|
2010-05-25 15:06:07 +02:00
|
|
|
fuse_copy_init(&cs, fc, 0, iov, nr_segs);
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
|
|
|
|
return fuse_dev_do_write(fc, &cs, iov_length(iov, nr_segs));
|
|
|
|
}
|
|
|
|
|
|
|
|
static ssize_t fuse_dev_splice_write(struct pipe_inode_info *pipe,
|
|
|
|
struct file *out, loff_t *ppos,
|
|
|
|
size_t len, unsigned int flags)
|
|
|
|
{
|
|
|
|
unsigned nbuf;
|
|
|
|
unsigned idx;
|
|
|
|
struct pipe_buffer *bufs;
|
|
|
|
struct fuse_copy_state cs;
|
|
|
|
struct fuse_conn *fc;
|
|
|
|
size_t rem;
|
|
|
|
ssize_t ret;
|
|
|
|
|
|
|
|
fc = fuse_get_conn(out);
|
|
|
|
if (!fc)
|
|
|
|
return -EPERM;
|
|
|
|
|
2010-12-07 20:16:56 +01:00
|
|
|
bufs = kmalloc(pipe->buffers * sizeof(struct pipe_buffer), GFP_KERNEL);
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
if (!bufs)
|
|
|
|
return -ENOMEM;
|
|
|
|
|
|
|
|
pipe_lock(pipe);
|
|
|
|
nbuf = 0;
|
|
|
|
rem = 0;
|
|
|
|
for (idx = 0; idx < pipe->nrbufs && rem < len; idx++)
|
|
|
|
rem += pipe->bufs[(pipe->curbuf + idx) & (pipe->buffers - 1)].len;
|
|
|
|
|
|
|
|
ret = -EINVAL;
|
|
|
|
if (rem < len) {
|
|
|
|
pipe_unlock(pipe);
|
|
|
|
goto out;
|
|
|
|
}
|
|
|
|
|
|
|
|
rem = len;
|
|
|
|
while (rem) {
|
|
|
|
struct pipe_buffer *ibuf;
|
|
|
|
struct pipe_buffer *obuf;
|
|
|
|
|
|
|
|
BUG_ON(nbuf >= pipe->buffers);
|
|
|
|
BUG_ON(!pipe->nrbufs);
|
|
|
|
ibuf = &pipe->bufs[pipe->curbuf];
|
|
|
|
obuf = &bufs[nbuf];
|
|
|
|
|
|
|
|
if (rem >= ibuf->len) {
|
|
|
|
*obuf = *ibuf;
|
|
|
|
ibuf->ops = NULL;
|
|
|
|
pipe->curbuf = (pipe->curbuf + 1) & (pipe->buffers - 1);
|
|
|
|
pipe->nrbufs--;
|
|
|
|
} else {
|
|
|
|
ibuf->ops->get(pipe, ibuf);
|
|
|
|
*obuf = *ibuf;
|
|
|
|
obuf->flags &= ~PIPE_BUF_FLAG_GIFT;
|
|
|
|
obuf->len = rem;
|
|
|
|
ibuf->offset += obuf->len;
|
|
|
|
ibuf->len -= obuf->len;
|
|
|
|
}
|
|
|
|
nbuf++;
|
|
|
|
rem -= obuf->len;
|
|
|
|
}
|
|
|
|
pipe_unlock(pipe);
|
|
|
|
|
2010-05-25 15:06:07 +02:00
|
|
|
fuse_copy_init(&cs, fc, 0, NULL, nbuf);
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
cs.pipebufs = bufs;
|
|
|
|
cs.pipe = pipe;
|
|
|
|
|
2010-05-25 15:06:07 +02:00
|
|
|
if (flags & SPLICE_F_MOVE)
|
|
|
|
cs.move_pages = 1;
|
|
|
|
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
ret = fuse_dev_do_write(fc, &cs, len);
|
|
|
|
|
|
|
|
for (idx = 0; idx < nbuf; idx++) {
|
|
|
|
struct pipe_buffer *buf = &bufs[idx];
|
|
|
|
buf->ops->release(pipe, buf);
|
|
|
|
}
|
|
|
|
out:
|
|
|
|
kfree(bufs);
|
|
|
|
return ret;
|
|
|
|
}
|
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
static unsigned fuse_dev_poll(struct file *file, poll_table *wait)
|
|
|
|
{
|
|
|
|
unsigned mask = POLLOUT | POLLWRNORM;
|
2006-04-10 22:54:50 -07:00
|
|
|
struct fuse_conn *fc = fuse_get_conn(file);
|
2005-09-09 13:10:27 -07:00
|
|
|
if (!fc)
|
2006-04-10 22:54:50 -07:00
|
|
|
return POLLERR;
|
2005-09-09 13:10:27 -07:00
|
|
|
|
|
|
|
poll_wait(file, &fc->waitq, wait);
|
|
|
|
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2006-04-10 22:54:50 -07:00
|
|
|
if (!fc->connected)
|
|
|
|
mask = POLLERR;
|
2006-06-25 05:48:54 -07:00
|
|
|
else if (request_pending(fc))
|
2006-04-10 22:54:50 -07:00
|
|
|
mask |= POLLIN | POLLRDNORM;
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
|
|
|
return mask;
|
|
|
|
}
|
|
|
|
|
2006-01-16 22:14:41 -08:00
|
|
|
/*
|
|
|
|
* Abort all requests on the given list (pending or processing)
|
|
|
|
*
|
2006-04-10 22:54:55 -07:00
|
|
|
* This function releases and reacquires fc->lock
|
2006-01-16 22:14:41 -08:00
|
|
|
*/
|
2005-09-09 13:10:27 -07:00
|
|
|
static void end_requests(struct fuse_conn *fc, struct list_head *head)
|
2010-09-07 13:42:41 +02:00
|
|
|
__releases(fc->lock)
|
|
|
|
__acquires(fc->lock)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
|
|
|
while (!list_empty(head)) {
|
|
|
|
struct fuse_req *req;
|
|
|
|
req = list_entry(head->next, struct fuse_req, list);
|
|
|
|
req->out.h.error = -ECONNABORTED;
|
|
|
|
request_end(fc, req);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2005-09-09 13:10:27 -07:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2006-01-16 22:14:41 -08:00
|
|
|
/*
|
|
|
|
* Abort requests under I/O
|
|
|
|
*
|
2006-06-25 05:48:53 -07:00
|
|
|
* The requests are set to aborted and finished, and the request
|
2006-01-16 22:14:41 -08:00
|
|
|
* waiter is woken up. This will make request_wait_answer() wait
|
|
|
|
* until the request is unlocked and then return.
|
2006-01-16 22:14:42 -08:00
|
|
|
*
|
|
|
|
* If the request is asynchronous, then the end function needs to be
|
|
|
|
* called after waiting for the request to be unlocked (if it was
|
|
|
|
* locked).
|
2006-01-16 22:14:41 -08:00
|
|
|
*/
|
|
|
|
static void end_io_requests(struct fuse_conn *fc)
|
2010-09-07 13:42:41 +02:00
|
|
|
__releases(fc->lock)
|
|
|
|
__acquires(fc->lock)
|
2006-01-16 22:14:41 -08:00
|
|
|
{
|
|
|
|
while (!list_empty(&fc->io)) {
|
2006-01-16 22:14:42 -08:00
|
|
|
struct fuse_req *req =
|
|
|
|
list_entry(fc->io.next, struct fuse_req, list);
|
|
|
|
void (*end) (struct fuse_conn *, struct fuse_req *) = req->end;
|
|
|
|
|
2006-06-25 05:48:53 -07:00
|
|
|
req->aborted = 1;
|
2006-01-16 22:14:41 -08:00
|
|
|
req->out.h.error = -ECONNABORTED;
|
|
|
|
req->state = FUSE_REQ_FINISHED;
|
|
|
|
list_del_init(&req->list);
|
|
|
|
wake_up(&req->waitq);
|
2006-01-16 22:14:42 -08:00
|
|
|
if (end) {
|
|
|
|
req->end = NULL;
|
|
|
|
__fuse_get_request(req);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2006-01-16 22:14:42 -08:00
|
|
|
wait_event(req->waitq, !req->locked);
|
|
|
|
end(fc, req);
|
2008-11-26 12:03:54 +01:00
|
|
|
fuse_put_request(fc, req);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2006-01-16 22:14:42 -08:00
|
|
|
}
|
2006-01-16 22:14:41 -08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2010-09-07 13:42:41 +02:00
|
|
|
static void end_queued_requests(struct fuse_conn *fc)
|
2010-09-07 13:42:41 +02:00
|
|
|
__releases(fc->lock)
|
|
|
|
__acquires(fc->lock)
|
2010-09-07 13:42:41 +02:00
|
|
|
{
|
|
|
|
fc->max_background = UINT_MAX;
|
|
|
|
flush_bg_queue(fc);
|
|
|
|
end_requests(fc, &fc->pending);
|
|
|
|
end_requests(fc, &fc->processing);
|
2010-12-07 20:16:56 +01:00
|
|
|
while (forget_pending(fc))
|
2010-12-07 20:16:56 +01:00
|
|
|
kfree(dequeue_forget(fc, 1, NULL));
|
2010-09-07 13:42:41 +02:00
|
|
|
}
|
|
|
|
|
2011-03-01 16:43:52 -08:00
|
|
|
static void end_polls(struct fuse_conn *fc)
|
|
|
|
{
|
|
|
|
struct rb_node *p;
|
|
|
|
|
|
|
|
p = rb_first(&fc->polled_files);
|
|
|
|
|
|
|
|
while (p) {
|
|
|
|
struct fuse_file *ff;
|
|
|
|
ff = rb_entry(p, struct fuse_file, polled_node);
|
|
|
|
wake_up_interruptible_all(&ff->poll_wait);
|
|
|
|
|
|
|
|
p = rb_next(p);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2006-01-16 22:14:41 -08:00
|
|
|
/*
|
|
|
|
* Abort all requests.
|
|
|
|
*
|
|
|
|
* Emergency exit in case of a malicious or accidental deadlock, or
|
|
|
|
* just a hung filesystem.
|
|
|
|
*
|
|
|
|
* The same effect is usually achievable through killing the
|
|
|
|
* filesystem daemon and all users of the filesystem. The exception
|
|
|
|
* is the combination of an asynchronous request and the tricky
|
|
|
|
* deadlock (see Documentation/filesystems/fuse.txt).
|
|
|
|
*
|
|
|
|
* During the aborting, progression of requests from the pending and
|
|
|
|
* processing lists onto the io list, and progression of new requests
|
|
|
|
* onto the pending list is prevented by req->connected being false.
|
|
|
|
*
|
|
|
|
* Progression of requests under I/O to the processing list is
|
2006-06-25 05:48:53 -07:00
|
|
|
* prevented by the req->aborted flag being true for these requests.
|
|
|
|
* For this reason requests on the io list must be aborted first.
|
2006-01-16 22:14:41 -08:00
|
|
|
*/
|
|
|
|
void fuse_abort_conn(struct fuse_conn *fc)
|
|
|
|
{
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2006-01-16 22:14:41 -08:00
|
|
|
if (fc->connected) {
|
|
|
|
fc->connected = 0;
|
2006-06-25 05:48:50 -07:00
|
|
|
fc->blocked = 0;
|
2013-03-21 18:02:15 +04:00
|
|
|
fc->initialized = 1;
|
2006-01-16 22:14:41 -08:00
|
|
|
end_io_requests(fc);
|
2010-09-07 13:42:41 +02:00
|
|
|
end_queued_requests(fc);
|
2011-03-01 16:43:52 -08:00
|
|
|
end_polls(fc);
|
2006-01-16 22:14:41 -08:00
|
|
|
wake_up_all(&fc->waitq);
|
2006-06-25 05:48:50 -07:00
|
|
|
wake_up_all(&fc->blocked_waitq);
|
2006-04-10 22:54:52 -07:00
|
|
|
kill_fasync(&fc->fasync, SIGIO, POLL_IN);
|
2006-01-16 22:14:41 -08:00
|
|
|
}
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2006-01-16 22:14:41 -08:00
|
|
|
}
|
2009-04-14 10:54:53 +09:00
|
|
|
EXPORT_SYMBOL_GPL(fuse_abort_conn);
|
2006-01-16 22:14:41 -08:00
|
|
|
|
2009-04-14 10:54:53 +09:00
|
|
|
int fuse_dev_release(struct inode *inode, struct file *file)
|
2005-09-09 13:10:27 -07:00
|
|
|
{
|
2006-04-10 22:54:55 -07:00
|
|
|
struct fuse_conn *fc = fuse_get_conn(file);
|
2005-09-09 13:10:27 -07:00
|
|
|
if (fc) {
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_lock(&fc->lock);
|
2005-09-09 13:10:31 -07:00
|
|
|
fc->connected = 0;
|
2010-09-07 13:42:41 +02:00
|
|
|
fc->blocked = 0;
|
2013-03-21 18:02:28 +04:00
|
|
|
fc->initialized = 1;
|
2010-09-07 13:42:41 +02:00
|
|
|
end_queued_requests(fc);
|
2011-03-01 16:43:52 -08:00
|
|
|
end_polls(fc);
|
2010-09-07 13:42:41 +02:00
|
|
|
wake_up_all(&fc->blocked_waitq);
|
2006-04-10 22:54:55 -07:00
|
|
|
spin_unlock(&fc->lock);
|
2006-06-25 05:48:51 -07:00
|
|
|
fuse_conn_put(fc);
|
2006-04-10 22:54:52 -07:00
|
|
|
}
|
2006-01-16 22:14:35 -08:00
|
|
|
|
2005-09-09 13:10:27 -07:00
|
|
|
return 0;
|
|
|
|
}
|
2009-04-14 10:54:53 +09:00
|
|
|
EXPORT_SYMBOL_GPL(fuse_dev_release);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
2006-04-10 22:54:52 -07:00
|
|
|
static int fuse_dev_fasync(int fd, struct file *file, int on)
|
|
|
|
{
|
|
|
|
struct fuse_conn *fc = fuse_get_conn(file);
|
|
|
|
if (!fc)
|
2006-04-10 22:54:56 -07:00
|
|
|
return -EPERM;
|
2006-04-10 22:54:52 -07:00
|
|
|
|
|
|
|
/* No locking - fasync_helper does its own locking */
|
|
|
|
return fasync_helper(fd, file, on, &fc->fasync);
|
|
|
|
}
|
|
|
|
|
2006-03-28 01:56:42 -08:00
|
|
|
const struct file_operations fuse_dev_operations = {
|
2005-09-09 13:10:27 -07:00
|
|
|
.owner = THIS_MODULE,
|
|
|
|
.llseek = no_llseek,
|
2006-09-30 23:28:47 -07:00
|
|
|
.read = do_sync_read,
|
|
|
|
.aio_read = fuse_dev_read,
|
2010-05-25 15:06:07 +02:00
|
|
|
.splice_read = fuse_dev_splice_read,
|
2006-09-30 23:28:47 -07:00
|
|
|
.write = do_sync_write,
|
|
|
|
.aio_write = fuse_dev_write,
|
fuse: support splice() writing to fuse device
Allow userspace filesystem implementation to use splice() to write to
the fuse device. The semantics of using splice() are:
1) buffer the message header and data in a temporary pipe
2) with a *single* splice() call move the message from the temporary pipe
to the fuse device
The READ reply message has the most interesting use for this, since
now the data from an arbitrary file descriptor (which could be a
regular file, a block device or a socket) can be tranferred into the
fuse device without having to go through a userspace buffer. It will
also allow zero copy moving of pages.
One caveat is that the protocol on the fuse device requires the length
of the whole message to be written into the header. But the length of
the data transferred into the temporary pipe may not be known in
advance. The current library implementation works around this by
using vmplice to write the header and modifying the header after
splicing the data into the pipe (error handling omitted):
struct fuse_out_header out;
iov.iov_base = &out;
iov.iov_len = sizeof(struct fuse_out_header);
vmsplice(pip[1], &iov, 1, 0);
len = splice(input_fd, input_offset, pip[1], NULL, len, 0);
/* retrospectively modify the header: */
out.len = len + sizeof(struct fuse_out_header);
splice(pip[0], NULL, fuse_chan_fd(req->ch), NULL, out.len, flags);
This works since vmsplice only saves a pointer to the data, it does
not copy the data itself.
Since pipes are currently limited to 16 pages and messages need to be
spliced atomically, the length of the data is limited to 15 pages (or
60kB for 4k pages).
Signed-off-by: Miklos Szeredi <mszeredi@suse.cz>
2010-05-25 15:06:06 +02:00
|
|
|
.splice_write = fuse_dev_splice_write,
|
2005-09-09 13:10:27 -07:00
|
|
|
.poll = fuse_dev_poll,
|
|
|
|
.release = fuse_dev_release,
|
2006-04-10 22:54:52 -07:00
|
|
|
.fasync = fuse_dev_fasync,
|
2005-09-09 13:10:27 -07:00
|
|
|
};
|
2009-04-14 10:54:53 +09:00
|
|
|
EXPORT_SYMBOL_GPL(fuse_dev_operations);
|
2005-09-09 13:10:27 -07:00
|
|
|
|
|
|
|
static struct miscdevice fuse_miscdevice = {
|
|
|
|
.minor = FUSE_MINOR,
|
|
|
|
.name = "fuse",
|
|
|
|
.fops = &fuse_dev_operations,
|
|
|
|
};
|
|
|
|
|
|
|
|
int __init fuse_dev_init(void)
|
|
|
|
{
|
|
|
|
int err = -ENOMEM;
|
|
|
|
fuse_req_cachep = kmem_cache_create("fuse_request",
|
|
|
|
sizeof(struct fuse_req),
|
2007-07-20 10:11:58 +09:00
|
|
|
0, 0, NULL);
|
2005-09-09 13:10:27 -07:00
|
|
|
if (!fuse_req_cachep)
|
|
|
|
goto out;
|
|
|
|
|
|
|
|
err = misc_register(&fuse_miscdevice);
|
|
|
|
if (err)
|
|
|
|
goto out_cache_clean;
|
|
|
|
|
|
|
|
return 0;
|
|
|
|
|
|
|
|
out_cache_clean:
|
|
|
|
kmem_cache_destroy(fuse_req_cachep);
|
|
|
|
out:
|
|
|
|
return err;
|
|
|
|
}
|
|
|
|
|
|
|
|
void fuse_dev_cleanup(void)
|
|
|
|
{
|
|
|
|
misc_deregister(&fuse_miscdevice);
|
|
|
|
kmem_cache_destroy(fuse_req_cachep);
|
|
|
|
}
|