Commit f6762b7a authored by Jens Axboe's avatar Jens Axboe

[PATCH] pipe: enable atomic copying of pipe data to/from user space

The pipe ->map() method uses kmap() to virtually map the pages, which
is both slow and has known scalability issues on SMP. This patch enables
atomic copying of pipe pages, by pre-faulting data and using kmap_atomic()
instead.

lmbench bw_pipe and lat_pipe measurements agree this is a Good Thing. Here
are results from that on a UP machine with highmem (1.5GiB of RAM), running
first a UP kernel, SMP kernel, and SMP kernel patched.

Vanilla-UP:
Pipe bandwidth: 1622.28 MB/sec
Pipe bandwidth: 1610.59 MB/sec
Pipe bandwidth: 1608.30 MB/sec
Pipe latency: 7.3275 microseconds
Pipe latency: 7.2995 microseconds
Pipe latency: 7.3097 microseconds

Vanilla-SMP:
Pipe bandwidth: 1382.19 MB/sec
Pipe bandwidth: 1317.27 MB/sec
Pipe bandwidth: 1355.61 MB/sec
Pipe latency: 9.6402 microseconds
Pipe latency: 9.6696 microseconds
Pipe latency: 9.6153 microseconds

Patched-SMP:
Pipe bandwidth: 1578.70 MB/sec
Pipe bandwidth: 1579.95 MB/sec
Pipe bandwidth: 1578.63 MB/sec
Pipe latency: 9.1654 microseconds
Pipe latency: 9.2266 microseconds
Pipe latency: 9.1527 microseconds
Signed-off-by: default avatarJens Axboe <axboe@suse.de>
parent e27dedd8
...@@ -55,7 +55,8 @@ void pipe_wait(struct pipe_inode_info *pipe) ...@@ -55,7 +55,8 @@ void pipe_wait(struct pipe_inode_info *pipe)
} }
static int static int
pipe_iov_copy_from_user(void *to, struct iovec *iov, unsigned long len) pipe_iov_copy_from_user(void *to, struct iovec *iov, unsigned long len,
int atomic)
{ {
unsigned long copy; unsigned long copy;
...@@ -64,8 +65,13 @@ pipe_iov_copy_from_user(void *to, struct iovec *iov, unsigned long len) ...@@ -64,8 +65,13 @@ pipe_iov_copy_from_user(void *to, struct iovec *iov, unsigned long len)
iov++; iov++;
copy = min_t(unsigned long, len, iov->iov_len); copy = min_t(unsigned long, len, iov->iov_len);
if (atomic) {
if (__copy_from_user_inatomic(to, iov->iov_base, copy))
return -EFAULT;
} else {
if (copy_from_user(to, iov->iov_base, copy)) if (copy_from_user(to, iov->iov_base, copy))
return -EFAULT; return -EFAULT;
}
to += copy; to += copy;
len -= copy; len -= copy;
iov->iov_base += copy; iov->iov_base += copy;
...@@ -75,7 +81,8 @@ pipe_iov_copy_from_user(void *to, struct iovec *iov, unsigned long len) ...@@ -75,7 +81,8 @@ pipe_iov_copy_from_user(void *to, struct iovec *iov, unsigned long len)
} }
static int static int
pipe_iov_copy_to_user(struct iovec *iov, const void *from, unsigned long len) pipe_iov_copy_to_user(struct iovec *iov, const void *from, unsigned long len,
int atomic)
{ {
unsigned long copy; unsigned long copy;
...@@ -84,8 +91,13 @@ pipe_iov_copy_to_user(struct iovec *iov, const void *from, unsigned long len) ...@@ -84,8 +91,13 @@ pipe_iov_copy_to_user(struct iovec *iov, const void *from, unsigned long len)
iov++; iov++;
copy = min_t(unsigned long, len, iov->iov_len); copy = min_t(unsigned long, len, iov->iov_len);
if (atomic) {
if (__copy_to_user_inatomic(iov->iov_base, from, copy))
return -EFAULT;
} else {
if (copy_to_user(iov->iov_base, from, copy)) if (copy_to_user(iov->iov_base, from, copy))
return -EFAULT; return -EFAULT;
}
from += copy; from += copy;
len -= copy; len -= copy;
iov->iov_base += copy; iov->iov_base += copy;
...@@ -94,6 +106,47 @@ pipe_iov_copy_to_user(struct iovec *iov, const void *from, unsigned long len) ...@@ -94,6 +106,47 @@ pipe_iov_copy_to_user(struct iovec *iov, const void *from, unsigned long len)
return 0; return 0;
} }
/*
* Attempt to pre-fault in the user memory, so we can use atomic copies.
* Returns the number of bytes not faulted in.
*/
static int iov_fault_in_pages_write(struct iovec *iov, unsigned long len)
{
while (!iov->iov_len)
iov++;
while (len > 0) {
unsigned long this_len;
this_len = min_t(unsigned long, len, iov->iov_len);
if (fault_in_pages_writeable(iov->iov_base, this_len))
break;
len -= this_len;
iov++;
}
return len;
}
/*
* Pre-fault in the user memory, so we can use atomic copies.
*/
static void iov_fault_in_pages_read(struct iovec *iov, unsigned long len)
{
while (!iov->iov_len)
iov++;
while (len > 0) {
unsigned long this_len;
this_len = min_t(unsigned long, len, iov->iov_len);
fault_in_pages_readable(iov->iov_base, this_len);
len -= this_len;
iov++;
}
}
static void anon_pipe_buf_release(struct pipe_inode_info *pipe, static void anon_pipe_buf_release(struct pipe_inode_info *pipe,
struct pipe_buffer *buf) struct pipe_buffer *buf)
{ {
...@@ -111,14 +164,23 @@ static void anon_pipe_buf_release(struct pipe_inode_info *pipe, ...@@ -111,14 +164,23 @@ static void anon_pipe_buf_release(struct pipe_inode_info *pipe,
} }
void *generic_pipe_buf_map(struct pipe_inode_info *pipe, void *generic_pipe_buf_map(struct pipe_inode_info *pipe,
struct pipe_buffer *buf) struct pipe_buffer *buf, int atomic)
{ {
if (atomic) {
buf->flags |= PIPE_BUF_FLAG_ATOMIC;
return kmap_atomic(buf->page, KM_USER0);
}
return kmap(buf->page); return kmap(buf->page);
} }
void generic_pipe_buf_unmap(struct pipe_inode_info *pipe, void generic_pipe_buf_unmap(struct pipe_inode_info *pipe,
struct pipe_buffer *buf) struct pipe_buffer *buf, void *map_data)
{ {
if (buf->flags & PIPE_BUF_FLAG_ATOMIC) {
buf->flags &= ~PIPE_BUF_FLAG_ATOMIC;
kunmap_atomic(map_data, KM_USER0);
} else
kunmap(buf->page); kunmap(buf->page);
} }
...@@ -183,7 +245,7 @@ pipe_readv(struct file *filp, const struct iovec *_iov, ...@@ -183,7 +245,7 @@ pipe_readv(struct file *filp, const struct iovec *_iov,
struct pipe_buf_operations *ops = buf->ops; struct pipe_buf_operations *ops = buf->ops;
void *addr; void *addr;
size_t chars = buf->len; size_t chars = buf->len;
int error; int error, atomic;
if (chars > total_len) if (chars > total_len)
chars = total_len; chars = total_len;
...@@ -195,12 +257,21 @@ pipe_readv(struct file *filp, const struct iovec *_iov, ...@@ -195,12 +257,21 @@ pipe_readv(struct file *filp, const struct iovec *_iov,
break; break;
} }
addr = ops->map(pipe, buf); atomic = !iov_fault_in_pages_write(iov, chars);
error = pipe_iov_copy_to_user(iov, addr + buf->offset, chars); redo:
ops->unmap(pipe, buf); addr = ops->map(pipe, buf, atomic);
error = pipe_iov_copy_to_user(iov, addr + buf->offset, chars, atomic);
ops->unmap(pipe, buf, addr);
if (unlikely(error)) { if (unlikely(error)) {
/*
* Just retry with the slow path if we failed.
*/
if (atomic) {
atomic = 0;
goto redo;
}
if (!ret) if (!ret)
ret = -EFAULT; ret = error;
break; break;
} }
ret += chars; ret += chars;
...@@ -304,21 +375,28 @@ pipe_writev(struct file *filp, const struct iovec *_iov, ...@@ -304,21 +375,28 @@ pipe_writev(struct file *filp, const struct iovec *_iov,
int offset = buf->offset + buf->len; int offset = buf->offset + buf->len;
if (ops->can_merge && offset + chars <= PAGE_SIZE) { if (ops->can_merge && offset + chars <= PAGE_SIZE) {
int error, atomic = 1;
void *addr; void *addr;
int error;
error = ops->pin(pipe, buf); error = ops->pin(pipe, buf);
if (error) if (error)
goto out; goto out;
addr = ops->map(pipe, buf); iov_fault_in_pages_read(iov, chars);
redo1:
addr = ops->map(pipe, buf, atomic);
error = pipe_iov_copy_from_user(offset + addr, iov, error = pipe_iov_copy_from_user(offset + addr, iov,
chars); chars, atomic);
ops->unmap(pipe, buf); ops->unmap(pipe, buf, addr);
ret = error; ret = error;
do_wakeup = 1; do_wakeup = 1;
if (error) if (error) {
if (atomic) {
atomic = 0;
goto redo1;
}
goto out; goto out;
}
buf->len += chars; buf->len += chars;
total_len -= chars; total_len -= chars;
ret = chars; ret = chars;
...@@ -341,7 +419,8 @@ pipe_writev(struct file *filp, const struct iovec *_iov, ...@@ -341,7 +419,8 @@ pipe_writev(struct file *filp, const struct iovec *_iov,
int newbuf = (pipe->curbuf + bufs) & (PIPE_BUFFERS-1); int newbuf = (pipe->curbuf + bufs) & (PIPE_BUFFERS-1);
struct pipe_buffer *buf = pipe->bufs + newbuf; struct pipe_buffer *buf = pipe->bufs + newbuf;
struct page *page = pipe->tmp_page; struct page *page = pipe->tmp_page;
int error; char *src;
int error, atomic = 1;
if (!page) { if (!page) {
page = alloc_page(GFP_HIGHUSER); page = alloc_page(GFP_HIGHUSER);
...@@ -361,11 +440,27 @@ pipe_writev(struct file *filp, const struct iovec *_iov, ...@@ -361,11 +440,27 @@ pipe_writev(struct file *filp, const struct iovec *_iov,
if (chars > total_len) if (chars > total_len)
chars = total_len; chars = total_len;
error = pipe_iov_copy_from_user(kmap(page), iov, chars); iov_fault_in_pages_read(iov, chars);
redo2:
if (atomic)
src = kmap_atomic(page, KM_USER0);
else
src = kmap(page);
error = pipe_iov_copy_from_user(src, iov, chars,
atomic);
if (atomic)
kunmap_atomic(src, KM_USER0);
else
kunmap(page); kunmap(page);
if (unlikely(error)) { if (unlikely(error)) {
if (atomic) {
atomic = 0;
goto redo2;
}
if (!ret) if (!ret)
ret = -EFAULT; ret = error;
break; break;
} }
ret += chars; ret += chars;
......
...@@ -640,13 +640,13 @@ static int pipe_to_file(struct pipe_inode_info *info, struct pipe_buffer *buf, ...@@ -640,13 +640,13 @@ static int pipe_to_file(struct pipe_inode_info *info, struct pipe_buffer *buf,
/* /*
* Careful, ->map() uses KM_USER0! * Careful, ->map() uses KM_USER0!
*/ */
char *src = buf->ops->map(info, buf); char *src = buf->ops->map(info, buf, 1);
char *dst = kmap_atomic(page, KM_USER1); char *dst = kmap_atomic(page, KM_USER1);
memcpy(dst + offset, src + buf->offset, this_len); memcpy(dst + offset, src + buf->offset, this_len);
flush_dcache_page(page); flush_dcache_page(page);
kunmap_atomic(dst, KM_USER1); kunmap_atomic(dst, KM_USER1);
buf->ops->unmap(info, buf); buf->ops->unmap(info, buf, src);
} }
ret = mapping->a_ops->commit_write(file, page, offset, offset+this_len); ret = mapping->a_ops->commit_write(file, page, offset, offset+this_len);
......
...@@ -5,7 +5,8 @@ ...@@ -5,7 +5,8 @@
#define PIPE_BUFFERS (16) #define PIPE_BUFFERS (16)
#define PIPE_BUF_FLAG_LRU 0x01 #define PIPE_BUF_FLAG_LRU 0x01 /* page is on the LRU */
#define PIPE_BUF_FLAG_ATOMIC 0x02 /* was atomically mapped */
struct pipe_buffer { struct pipe_buffer {
struct page *page; struct page *page;
...@@ -28,8 +29,8 @@ struct pipe_buffer { ...@@ -28,8 +29,8 @@ struct pipe_buffer {
*/ */
struct pipe_buf_operations { struct pipe_buf_operations {
int can_merge; int can_merge;
void * (*map)(struct pipe_inode_info *, struct pipe_buffer *); void * (*map)(struct pipe_inode_info *, struct pipe_buffer *, int);
void (*unmap)(struct pipe_inode_info *, struct pipe_buffer *); void (*unmap)(struct pipe_inode_info *, struct pipe_buffer *, void *);
int (*pin)(struct pipe_inode_info *, struct pipe_buffer *); int (*pin)(struct pipe_inode_info *, struct pipe_buffer *);
void (*release)(struct pipe_inode_info *, struct pipe_buffer *); void (*release)(struct pipe_inode_info *, struct pipe_buffer *);
int (*steal)(struct pipe_inode_info *, struct pipe_buffer *); int (*steal)(struct pipe_inode_info *, struct pipe_buffer *);
...@@ -64,8 +65,8 @@ void free_pipe_info(struct inode * inode); ...@@ -64,8 +65,8 @@ void free_pipe_info(struct inode * inode);
void __free_pipe_info(struct pipe_inode_info *); void __free_pipe_info(struct pipe_inode_info *);
/* Generic pipe buffer ops functions */ /* Generic pipe buffer ops functions */
void *generic_pipe_buf_map(struct pipe_inode_info *, struct pipe_buffer *); void *generic_pipe_buf_map(struct pipe_inode_info *, struct pipe_buffer *, int);
void generic_pipe_buf_unmap(struct pipe_inode_info *, struct pipe_buffer *); void generic_pipe_buf_unmap(struct pipe_inode_info *, struct pipe_buffer *, void *);
void generic_pipe_buf_get(struct pipe_inode_info *, struct pipe_buffer *); void generic_pipe_buf_get(struct pipe_inode_info *, struct pipe_buffer *);
int generic_pipe_buf_pin(struct pipe_inode_info *, struct pipe_buffer *); int generic_pipe_buf_pin(struct pipe_inode_info *, struct pipe_buffer *);
......
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment