forked from rubenslte/android_kernel_samsung_msm8226
Change-Id: Ib07ead1e23e816c96552254c049016825a164f2c
UPSTREAM: zram/zcomp: use GFP_NOIO to allocate streams
(cherry picked from commit 3d5fe03a3ea013060ebba2a811aeb0f23f56aefa)
We can end up allocating a new compression stream with GFP_KERNEL from
within the IO path, which may result is nested (recursive) IO
operations. That can introduce problems if the IO path in question is a
reclaimer, holding some locks that will deadlock nested IOs.
Allocate streams and working memory using GFP_NOIO flag, forbidding
recursive IO and FS operations.
An example:
inconsistent {IN-RECLAIM_FS-W} -> {RECLAIM_FS-ON-W} usage.
git/20158 [HC0[0]:SC0[0]:HE1:SE1] takes:
(jbd2_handle){+.+.?.}, at: start_this_handle+0x4ca/0x555
{IN-RECLAIM_FS-W} state was registered at:
__lock_acquire+0x8da/0x117b
lock_acquire+0x10c/0x1a7
start_this_handle+0x52d/0x555
jbd2__journal_start+0xb4/0x237
__ext4_journal_start_sb+0x108/0x17e
ext4_dirty_inode+0x32/0x61
__mark_inode_dirty+0x16b/0x60c
iput+0x11e/0x274
__dentry_kill+0x148/0x1b8
shrink_dentry_list+0x274/0x44a
prune_dcache_sb+0x4a/0x55
super_cache_scan+0xfc/0x176
shrink_slab.part.14.constprop.25+0x2a2/0x4d3
shrink_zone+0x74/0x140
kswapd+0x6b7/0x930
kthread+0x107/0x10f
ret_from_fork+0x3f/0x70
irq event stamp: 138297
hardirqs last enabled at (138297): debug_check_no_locks_freed+0x113/0x12f
hardirqs last disabled at (138296): debug_check_no_locks_freed+0x33/0x12f
softirqs last enabled at (137818): __do_softirq+0x2d3/0x3e9
softirqs last disabled at (137813): irq_exit+0x41/0x95
other info that might help us debug this:
Possible unsafe locking scenario:
CPU0
----
lock(jbd2_handle);
<Interrupt>
lock(jbd2_handle);
*** DEADLOCK ***
5 locks held by git/20158:
#0: (sb_writers#7){.+.+.+}, at: [<ffffffff81155411>] mnt_want_write+0x24/0x4b
#1: (&type->i_mutex_dir_key#2/1){+.+.+.}, at: [<ffffffff81145087>] lock_rename+0xd9/0xe3
#2: (&sb->s_type->i_mutex_key#11){+.+.+.}, at: [<ffffffff8114f8e2>] lock_two_nondirectories+0x3f/0x6b
#3: (&sb->s_type->i_mutex_key#11/4){+.+.+.}, at: [<ffffffff8114f909>] lock_two_nondirectories+0x66/0x6b
#4: (jbd2_handle){+.+.?.}, at: [<ffffffff811e31db>] start_this_handle+0x4ca/0x555
stack backtrace:
CPU: 2 PID: 20158 Comm: git Not tainted 4.1.0-rc7-next-20150615-dbg-00016-g8bdf555-dirty #211
Call Trace:
dump_stack+0x4c/0x6e
mark_lock+0x384/0x56d
mark_held_locks+0x5f/0x76
lockdep_trace_alloc+0xb2/0xb5
kmem_cache_alloc_trace+0x32/0x1e2
zcomp_strm_alloc+0x25/0x73 [zram]
zcomp_strm_multi_find+0xe7/0x173 [zram]
zcomp_strm_find+0xc/0xe [zram]
zram_bvec_rw+0x2ca/0x7e0 [zram]
zram_make_request+0x1fa/0x301 [zram]
generic_make_request+0x9c/0xdb
submit_bio+0xf7/0x120
ext4_io_submit+0x2e/0x43
ext4_bio_write_page+0x1b7/0x300
mpage_submit_page+0x60/0x77
mpage_map_and_submit_buffers+0x10f/0x21d
ext4_writepages+0xc8c/0xe1b
do_writepages+0x23/0x2c
__filemap_fdatawrite_range+0x84/0x8b
filemap_flush+0x1c/0x1e
ext4_alloc_da_blocks+0xb8/0x117
ext4_rename+0x132/0x6dc
? mark_held_locks+0x5f/0x76
ext4_rename2+0x29/0x2b
vfs_rename+0x540/0x636
SyS_renameat2+0x359/0x44d
SyS_rename+0x1e/0x20
entry_SYSCALL_64_fastpath+0x12/0x6f
[minchan@kernel.org: add stable mark]
Signed-off-by: Sergey Senozhatsky <sergey.senozhatsky@gmail.com>
Acked-by: Minchan Kim <minchan@kernel.org>
Cc: Kyeongdon Kim <kyeongdon.kim@lge.com>
Cc: <stable@vger.kernel.org>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
UPSTREAM: zram: try vmalloc() after kmalloc()
(cherry picked from commit d913897abace843bba20249f3190167f7895e9c3)
When we're using LZ4 multi compression streams for zram swap, we found
out page allocation failure message in system running test. That was
not only once, but a few(2 - 5 times per test). Also, some failure
cases were continually occurring to try allocation order 3.
In order to make parallel compression private data, we should call
kzalloc() with order 2/3 in runtime(lzo/lz4). But if there is no order
2/3 size memory to allocate in that time, page allocation fails. This
patch makes to use vmalloc() as fallback of kmalloc(), this prevents
page alloc failure warning.
After using this, we never found warning message in running test, also
It could reduce process startup latency about 60-120ms in each case.
For reference a call trace :
Binder_1: page allocation failure: order:3, mode:0x10c0d0
CPU: 0 PID: 424 Comm: Binder_1 Tainted: GW 3.10.49-perf-g991d02b-dirty #20
Call trace:
dump_backtrace+0x0/0x270
show_stack+0x10/0x1c
dump_stack+0x1c/0x28
warn_alloc_failed+0xfc/0x11c
__alloc_pages_nodemask+0x724/0x7f0
__get_free_pages+0x14/0x5c
kmalloc_order_trace+0x38/0xd8
zcomp_lz4_create+0x2c/0x38
zcomp_strm_alloc+0x34/0x78
zcomp_strm_multi_find+0x124/0x1ec
zcomp_strm_find+0xc/0x18
zram_bvec_rw+0x2fc/0x780
zram_make_request+0x25c/0x2d4
generic_make_request+0x80/0xbc
submit_bio+0xa4/0x15c
__swap_writepage+0x218/0x230
swap_writepage+0x3c/0x4c
shrink_page_list+0x51c/0x8d0
shrink_inactive_list+0x3f8/0x60c
shrink_lruvec+0x33c/0x4cc
shrink_zone+0x3c/0x100
try_to_free_pages+0x2b8/0x54c
__alloc_pages_nodemask+0x514/0x7f0
__get_free_pages+0x14/0x5c
proc_info_read+0x50/0xe4
vfs_read+0xa0/0x12c
SyS_read+0x44/0x74
DMA: 3397*4kB (MC) 26*8kB (RC) 0*16kB 0*32kB 0*64kB 0*128kB 0*256kB
0*512kB 0*1024kB 0*2048kB 0*4096kB = 13796kB
[minchan@kernel.org: change vmalloc gfp and adding comment about gfp]
[sergey.senozhatsky@gmail.com: tweak comments and styles]
Signed-off-by: Kyeongdon Kim <kyeongdon.kim@lge.com>
Signed-off-by: Minchan Kim <minchan@kernel.org>
Acked-by: Sergey Senozhatsky <sergey.senozhatsky@gmail.com>
Sergey Senozhatsky <sergey.senozhatsky.work@gmail.com>
Cc: <stable@vger.kernel.org>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
UPSTREAM: zram: pass gfp from zcomp frontend to backend
(cherry picked from commit 75d8947a36d0c9aedd69118d1f14bf424005c7c2)
Each zcomp backend uses own gfp flag but it's pointless because the
context they could be called is driven by upper layer(ie, zcomp
frontend). As well, zcomp frondend could call them in different
context. One context(ie, zram init part) is it should be better to make
sure successful allocation other context(ie, further stream allocation
part for accelarating I/O speed) is just optional so let's pass gfp down
from driver (ie, zcomp frontend) like normal MM convention.
[sergey.senozhatsky@gmail.com: add missing __vmalloc zero and highmem gfps]
Signed-off-by: Minchan Kim <minchan@kernel.org>
Signed-off-by: Sergey Senozhatsky <sergey.senozhatsky@gmail.com>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
UPSTREAM: zram/zcomp: do not zero out zcomp private pages
(cherry picked from commit e02d238c9852a91b30da9ea32ce36d1416cdc683)
Do not __GFP_ZERO allocated zcomp ->private pages. We keep allocated
streams around and use them for read/write requests, so we supply a
zeroed out ->private to compression algorithm as a scratch buffer only
once -- the first time we use that stream. For the rest of IO requests
served by this stream ->private usually contains some temporarily data
from the previous requests.
Signed-off-by: Sergey Senozhatsky <sergey.senozhatsky@gmail.com>
Acked-by: Minchan Kim <minchan@kernel.org>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
UPSTREAM: block: disable entropy contributions for nonrot devices
(cherry picked from commit b277da0a8a594308e17881f4926879bd5fca2a2d)
Clear QUEUE_FLAG_ADD_RANDOM in all block drivers that set
QUEUE_FLAG_NONROT.
Historically, all block devices have automatically made entropy
contributions. But as previously stated in commit e2e1a148 ("block: add
sysfs knob for turning off disk entropy contributions"):
- On SSD disks, the completion times aren't as random as they
are for rotational drives. So it's questionable whether they
should contribute to the random pool in the first place.
- Calling add_disk_randomness() has a lot of overhead.
There are more reliable sources for randomness than non-rotational block
devices. From a security perspective it is better to err on the side of
caution than to allow entropy contributions from unreliable "random"
sources.
Change-Id: I2a4f86bacee8786e2cb1a82d45156338f79d64e0
Signed-off-by: Mike Snitzer <snitzer@redhat.com>
Signed-off-by: Jens Axboe <axboe@fb.com>
Signed-off-by: Kevin F. Haggerty <haggertk@lineageos.org>
628 lines
14 KiB
C
628 lines
14 KiB
C
/*
|
|
* Interface to Linux block layer for MTD 'translation layers'.
|
|
*
|
|
* Copyright © 2003-2010 David Woodhouse <dwmw2@infradead.org>
|
|
*
|
|
* This program is free software; you can redistribute it and/or modify
|
|
* it under the terms of the GNU General Public License as published by
|
|
* the Free Software Foundation; either version 2 of the License, or
|
|
* (at your option) any later version.
|
|
*
|
|
* This program is distributed in the hope that it will be useful,
|
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
* GNU General Public License for more details.
|
|
*
|
|
* You should have received a copy of the GNU General Public License
|
|
* along with this program; if not, write to the Free Software
|
|
* Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA
|
|
*
|
|
*/
|
|
|
|
#include <linux/kernel.h>
|
|
#include <linux/slab.h>
|
|
#include <linux/module.h>
|
|
#include <linux/list.h>
|
|
#include <linux/fs.h>
|
|
#include <linux/mtd/blktrans.h>
|
|
#include <linux/mtd/mtd.h>
|
|
#include <linux/blkdev.h>
|
|
#include <linux/blkpg.h>
|
|
#include <linux/spinlock.h>
|
|
#include <linux/hdreg.h>
|
|
#include <linux/init.h>
|
|
#include <linux/mutex.h>
|
|
#include <linux/kthread.h>
|
|
#include <asm/uaccess.h>
|
|
|
|
#include "mtdcore.h"
|
|
|
|
static LIST_HEAD(blktrans_majors);
|
|
static DEFINE_MUTEX(blktrans_ref_mutex);
|
|
|
|
static void blktrans_dev_release(struct kref *kref)
|
|
{
|
|
struct mtd_blktrans_dev *dev =
|
|
container_of(kref, struct mtd_blktrans_dev, ref);
|
|
|
|
dev->disk->private_data = NULL;
|
|
blk_cleanup_queue(dev->rq);
|
|
put_disk(dev->disk);
|
|
list_del(&dev->list);
|
|
kfree(dev);
|
|
}
|
|
|
|
static struct mtd_blktrans_dev *blktrans_dev_get(struct gendisk *disk)
|
|
{
|
|
struct mtd_blktrans_dev *dev;
|
|
|
|
mutex_lock(&blktrans_ref_mutex);
|
|
dev = disk->private_data;
|
|
|
|
if (!dev)
|
|
goto unlock;
|
|
kref_get(&dev->ref);
|
|
unlock:
|
|
mutex_unlock(&blktrans_ref_mutex);
|
|
return dev;
|
|
}
|
|
|
|
static void blktrans_dev_put(struct mtd_blktrans_dev *dev)
|
|
{
|
|
mutex_lock(&blktrans_ref_mutex);
|
|
kref_put(&dev->ref, blktrans_dev_release);
|
|
mutex_unlock(&blktrans_ref_mutex);
|
|
}
|
|
|
|
|
|
static int do_blktrans_request(struct mtd_blktrans_ops *tr,
|
|
struct mtd_blktrans_dev *dev,
|
|
struct request *req)
|
|
{
|
|
unsigned long block, nsect;
|
|
char *buf;
|
|
|
|
block = blk_rq_pos(req) << 9 >> tr->blkshift;
|
|
nsect = blk_rq_cur_bytes(req) >> tr->blkshift;
|
|
|
|
buf = req->buffer;
|
|
|
|
if (req->cmd_type != REQ_TYPE_FS)
|
|
return -EIO;
|
|
|
|
if (blk_rq_pos(req) + blk_rq_cur_sectors(req) >
|
|
get_capacity(req->rq_disk))
|
|
return -EIO;
|
|
|
|
if (req->cmd_flags & REQ_DISCARD)
|
|
return tr->discard(dev, block, nsect);
|
|
|
|
switch(rq_data_dir(req)) {
|
|
case READ:
|
|
for (; nsect > 0; nsect--, block++, buf += tr->blksize)
|
|
if (tr->readsect(dev, block, buf))
|
|
return -EIO;
|
|
rq_flush_dcache_pages(req);
|
|
return 0;
|
|
case WRITE:
|
|
if (!tr->writesect)
|
|
return -EIO;
|
|
|
|
rq_flush_dcache_pages(req);
|
|
for (; nsect > 0; nsect--, block++, buf += tr->blksize)
|
|
if (tr->writesect(dev, block, buf))
|
|
return -EIO;
|
|
return 0;
|
|
default:
|
|
printk(KERN_NOTICE "Unknown request %u\n", rq_data_dir(req));
|
|
return -EIO;
|
|
}
|
|
}
|
|
|
|
int mtd_blktrans_cease_background(struct mtd_blktrans_dev *dev)
|
|
{
|
|
if (kthread_should_stop())
|
|
return 1;
|
|
|
|
return dev->bg_stop;
|
|
}
|
|
EXPORT_SYMBOL_GPL(mtd_blktrans_cease_background);
|
|
|
|
static int mtd_blktrans_thread(void *arg)
|
|
{
|
|
struct mtd_blktrans_dev *dev = arg;
|
|
struct mtd_blktrans_ops *tr = dev->tr;
|
|
struct request_queue *rq = dev->rq;
|
|
struct request *req = NULL;
|
|
int background_done = 0;
|
|
|
|
spin_lock_irq(rq->queue_lock);
|
|
|
|
while (!kthread_should_stop()) {
|
|
int res;
|
|
|
|
dev->bg_stop = false;
|
|
if (!req && !(req = blk_fetch_request(rq))) {
|
|
if (tr->background && !background_done) {
|
|
spin_unlock_irq(rq->queue_lock);
|
|
mutex_lock(&dev->lock);
|
|
tr->background(dev);
|
|
mutex_unlock(&dev->lock);
|
|
spin_lock_irq(rq->queue_lock);
|
|
/*
|
|
* Do background processing just once per idle
|
|
* period.
|
|
*/
|
|
background_done = !dev->bg_stop;
|
|
continue;
|
|
}
|
|
set_current_state(TASK_INTERRUPTIBLE);
|
|
|
|
if (kthread_should_stop())
|
|
set_current_state(TASK_RUNNING);
|
|
|
|
spin_unlock_irq(rq->queue_lock);
|
|
schedule();
|
|
spin_lock_irq(rq->queue_lock);
|
|
continue;
|
|
}
|
|
|
|
spin_unlock_irq(rq->queue_lock);
|
|
|
|
mutex_lock(&dev->lock);
|
|
res = do_blktrans_request(dev->tr, dev, req);
|
|
mutex_unlock(&dev->lock);
|
|
|
|
spin_lock_irq(rq->queue_lock);
|
|
|
|
if (!__blk_end_request_cur(req, res))
|
|
req = NULL;
|
|
|
|
background_done = 0;
|
|
}
|
|
|
|
if (req)
|
|
__blk_end_request_all(req, -EIO);
|
|
|
|
spin_unlock_irq(rq->queue_lock);
|
|
|
|
return 0;
|
|
}
|
|
|
|
static void mtd_blktrans_request(struct request_queue *rq)
|
|
{
|
|
struct mtd_blktrans_dev *dev;
|
|
struct request *req = NULL;
|
|
|
|
dev = rq->queuedata;
|
|
|
|
if (!dev)
|
|
while ((req = blk_fetch_request(rq)) != NULL)
|
|
__blk_end_request_all(req, -ENODEV);
|
|
else {
|
|
dev->bg_stop = true;
|
|
wake_up_process(dev->thread);
|
|
}
|
|
}
|
|
|
|
static int blktrans_open(struct block_device *bdev, fmode_t mode)
|
|
{
|
|
struct mtd_blktrans_dev *dev = blktrans_dev_get(bdev->bd_disk);
|
|
int ret = 0;
|
|
|
|
if (!dev)
|
|
return -ERESTARTSYS; /* FIXME: busy loop! -arnd*/
|
|
|
|
mutex_lock(&dev->lock);
|
|
mutex_lock(&mtd_table_mutex);
|
|
|
|
if (dev->open)
|
|
goto unlock;
|
|
|
|
kref_get(&dev->ref);
|
|
__module_get(dev->tr->owner);
|
|
|
|
if (!dev->mtd)
|
|
goto unlock;
|
|
|
|
if (dev->tr->open) {
|
|
ret = dev->tr->open(dev);
|
|
if (ret)
|
|
goto error_put;
|
|
}
|
|
|
|
ret = __get_mtd_device(dev->mtd);
|
|
if (ret)
|
|
goto error_release;
|
|
dev->file_mode = mode;
|
|
|
|
unlock:
|
|
dev->open++;
|
|
mutex_unlock(&mtd_table_mutex);
|
|
mutex_unlock(&dev->lock);
|
|
blktrans_dev_put(dev);
|
|
return ret;
|
|
|
|
error_release:
|
|
if (dev->tr->release)
|
|
dev->tr->release(dev);
|
|
error_put:
|
|
module_put(dev->tr->owner);
|
|
kref_put(&dev->ref, blktrans_dev_release);
|
|
mutex_unlock(&mtd_table_mutex);
|
|
mutex_unlock(&dev->lock);
|
|
blktrans_dev_put(dev);
|
|
return ret;
|
|
}
|
|
|
|
static int blktrans_release(struct gendisk *disk, fmode_t mode)
|
|
{
|
|
struct mtd_blktrans_dev *dev = blktrans_dev_get(disk);
|
|
int ret = 0;
|
|
|
|
if (!dev)
|
|
return ret;
|
|
|
|
mutex_lock(&dev->lock);
|
|
mutex_lock(&mtd_table_mutex);
|
|
|
|
if (--dev->open)
|
|
goto unlock;
|
|
|
|
kref_put(&dev->ref, blktrans_dev_release);
|
|
module_put(dev->tr->owner);
|
|
|
|
if (dev->mtd) {
|
|
ret = dev->tr->release ? dev->tr->release(dev) : 0;
|
|
__put_mtd_device(dev->mtd);
|
|
}
|
|
unlock:
|
|
mutex_unlock(&mtd_table_mutex);
|
|
mutex_unlock(&dev->lock);
|
|
blktrans_dev_put(dev);
|
|
return ret;
|
|
}
|
|
|
|
static int blktrans_getgeo(struct block_device *bdev, struct hd_geometry *geo)
|
|
{
|
|
struct mtd_blktrans_dev *dev = blktrans_dev_get(bdev->bd_disk);
|
|
int ret = -ENXIO;
|
|
|
|
if (!dev)
|
|
return ret;
|
|
|
|
mutex_lock(&dev->lock);
|
|
|
|
if (!dev->mtd)
|
|
goto unlock;
|
|
|
|
ret = dev->tr->getgeo ? dev->tr->getgeo(dev, geo) : 0;
|
|
unlock:
|
|
mutex_unlock(&dev->lock);
|
|
blktrans_dev_put(dev);
|
|
return ret;
|
|
}
|
|
|
|
static int blktrans_ioctl(struct block_device *bdev, fmode_t mode,
|
|
unsigned int cmd, unsigned long arg)
|
|
{
|
|
struct mtd_blktrans_dev *dev = blktrans_dev_get(bdev->bd_disk);
|
|
int ret = -ENXIO;
|
|
|
|
if (!dev)
|
|
return ret;
|
|
|
|
mutex_lock(&dev->lock);
|
|
|
|
if (!dev->mtd)
|
|
goto unlock;
|
|
|
|
switch (cmd) {
|
|
case BLKFLSBUF:
|
|
ret = dev->tr->flush ? dev->tr->flush(dev) : 0;
|
|
break;
|
|
default:
|
|
ret = -ENOTTY;
|
|
}
|
|
unlock:
|
|
mutex_unlock(&dev->lock);
|
|
blktrans_dev_put(dev);
|
|
return ret;
|
|
}
|
|
|
|
static const struct block_device_operations mtd_blktrans_ops = {
|
|
.owner = THIS_MODULE,
|
|
.open = blktrans_open,
|
|
.release = blktrans_release,
|
|
.ioctl = blktrans_ioctl,
|
|
.getgeo = blktrans_getgeo,
|
|
};
|
|
|
|
int add_mtd_blktrans_dev(struct mtd_blktrans_dev *new)
|
|
{
|
|
struct mtd_blktrans_ops *tr = new->tr;
|
|
struct mtd_blktrans_dev *d;
|
|
int last_devnum = -1;
|
|
struct gendisk *gd;
|
|
int ret;
|
|
|
|
if (mutex_trylock(&mtd_table_mutex)) {
|
|
mutex_unlock(&mtd_table_mutex);
|
|
BUG();
|
|
}
|
|
|
|
mutex_lock(&blktrans_ref_mutex);
|
|
list_for_each_entry(d, &tr->devs, list) {
|
|
if (new->devnum == -1) {
|
|
/* Use first free number */
|
|
if (d->devnum != last_devnum+1) {
|
|
/* Found a free devnum. Plug it in here */
|
|
new->devnum = last_devnum+1;
|
|
list_add_tail(&new->list, &d->list);
|
|
goto added;
|
|
}
|
|
} else if (d->devnum == new->devnum) {
|
|
/* Required number taken */
|
|
mutex_unlock(&blktrans_ref_mutex);
|
|
return -EBUSY;
|
|
} else if (d->devnum > new->devnum) {
|
|
/* Required number was free */
|
|
list_add_tail(&new->list, &d->list);
|
|
goto added;
|
|
}
|
|
last_devnum = d->devnum;
|
|
}
|
|
|
|
ret = -EBUSY;
|
|
if (new->devnum == -1)
|
|
new->devnum = last_devnum+1;
|
|
|
|
/* Check that the device and any partitions will get valid
|
|
* minor numbers and that the disk naming code below can cope
|
|
* with this number. */
|
|
if (new->devnum > (MINORMASK >> tr->part_bits) ||
|
|
(tr->part_bits && new->devnum >= 27 * 26)) {
|
|
mutex_unlock(&blktrans_ref_mutex);
|
|
goto error1;
|
|
}
|
|
|
|
list_add_tail(&new->list, &tr->devs);
|
|
added:
|
|
mutex_unlock(&blktrans_ref_mutex);
|
|
|
|
mutex_init(&new->lock);
|
|
kref_init(&new->ref);
|
|
if (!tr->writesect)
|
|
new->readonly = 1;
|
|
|
|
/* Create gendisk */
|
|
ret = -ENOMEM;
|
|
gd = alloc_disk(1 << tr->part_bits);
|
|
|
|
if (!gd)
|
|
goto error2;
|
|
|
|
new->disk = gd;
|
|
gd->private_data = new;
|
|
gd->major = tr->major;
|
|
gd->first_minor = (new->devnum) << tr->part_bits;
|
|
gd->fops = &mtd_blktrans_ops;
|
|
|
|
if (tr->part_bits)
|
|
if (new->devnum < 26)
|
|
snprintf(gd->disk_name, sizeof(gd->disk_name),
|
|
"%s%c", tr->name, 'a' + new->devnum);
|
|
else
|
|
snprintf(gd->disk_name, sizeof(gd->disk_name),
|
|
"%s%c%c", tr->name,
|
|
'a' - 1 + new->devnum / 26,
|
|
'a' + new->devnum % 26);
|
|
else
|
|
snprintf(gd->disk_name, sizeof(gd->disk_name),
|
|
"%s%d", tr->name, new->devnum);
|
|
|
|
set_capacity(gd, (new->size * tr->blksize) >> 9);
|
|
|
|
/* Create the request queue */
|
|
spin_lock_init(&new->queue_lock);
|
|
new->rq = blk_init_queue(mtd_blktrans_request, &new->queue_lock);
|
|
|
|
if (!new->rq)
|
|
goto error3;
|
|
|
|
new->rq->queuedata = new;
|
|
|
|
/*
|
|
* Empirical measurements revealed that read ahead values larger than
|
|
* 4 slowed down boot time, so start out with this small value.
|
|
*/
|
|
new->rq->backing_dev_info.ra_pages = (4 * 1024) / PAGE_CACHE_SIZE;
|
|
|
|
blk_queue_logical_block_size(new->rq, tr->blksize);
|
|
|
|
queue_flag_set_unlocked(QUEUE_FLAG_NONROT, new->rq);
|
|
queue_flag_clear_unlocked(QUEUE_FLAG_ADD_RANDOM, new->rq);
|
|
|
|
if (tr->discard) {
|
|
queue_flag_set_unlocked(QUEUE_FLAG_DISCARD, new->rq);
|
|
new->rq->limits.max_discard_sectors = UINT_MAX;
|
|
}
|
|
|
|
gd->queue = new->rq;
|
|
|
|
/* Create processing thread */
|
|
/* TODO: workqueue ? */
|
|
new->thread = kthread_run(mtd_blktrans_thread, new,
|
|
"%s%d", tr->name, new->mtd->index);
|
|
if (IS_ERR(new->thread)) {
|
|
ret = PTR_ERR(new->thread);
|
|
goto error4;
|
|
}
|
|
gd->driverfs_dev = &new->mtd->dev;
|
|
|
|
if (new->readonly)
|
|
set_disk_ro(gd, 1);
|
|
|
|
add_disk(gd);
|
|
|
|
if (new->disk_attributes) {
|
|
ret = sysfs_create_group(&disk_to_dev(gd)->kobj,
|
|
new->disk_attributes);
|
|
WARN_ON(ret);
|
|
}
|
|
return 0;
|
|
error4:
|
|
blk_cleanup_queue(new->rq);
|
|
error3:
|
|
put_disk(new->disk);
|
|
error2:
|
|
list_del(&new->list);
|
|
error1:
|
|
return ret;
|
|
}
|
|
|
|
int del_mtd_blktrans_dev(struct mtd_blktrans_dev *old)
|
|
{
|
|
unsigned long flags;
|
|
|
|
if (mutex_trylock(&mtd_table_mutex)) {
|
|
mutex_unlock(&mtd_table_mutex);
|
|
BUG();
|
|
}
|
|
|
|
if (old->disk_attributes)
|
|
sysfs_remove_group(&disk_to_dev(old->disk)->kobj,
|
|
old->disk_attributes);
|
|
|
|
/* Stop new requests to arrive */
|
|
del_gendisk(old->disk);
|
|
|
|
|
|
/* Stop the thread */
|
|
kthread_stop(old->thread);
|
|
|
|
/* Kill current requests */
|
|
spin_lock_irqsave(&old->queue_lock, flags);
|
|
old->rq->queuedata = NULL;
|
|
blk_start_queue(old->rq);
|
|
spin_unlock_irqrestore(&old->queue_lock, flags);
|
|
|
|
/* If the device is currently open, tell trans driver to close it,
|
|
then put mtd device, and don't touch it again */
|
|
mutex_lock(&old->lock);
|
|
if (old->open) {
|
|
if (old->tr->release)
|
|
old->tr->release(old);
|
|
__put_mtd_device(old->mtd);
|
|
}
|
|
|
|
old->mtd = NULL;
|
|
|
|
mutex_unlock(&old->lock);
|
|
blktrans_dev_put(old);
|
|
return 0;
|
|
}
|
|
|
|
static void blktrans_notify_remove(struct mtd_info *mtd)
|
|
{
|
|
struct mtd_blktrans_ops *tr;
|
|
struct mtd_blktrans_dev *dev, *next;
|
|
|
|
list_for_each_entry(tr, &blktrans_majors, list)
|
|
list_for_each_entry_safe(dev, next, &tr->devs, list)
|
|
if (dev->mtd == mtd)
|
|
tr->remove_dev(dev);
|
|
}
|
|
|
|
static void blktrans_notify_add(struct mtd_info *mtd)
|
|
{
|
|
struct mtd_blktrans_ops *tr;
|
|
|
|
if (mtd->type == MTD_ABSENT)
|
|
return;
|
|
|
|
list_for_each_entry(tr, &blktrans_majors, list)
|
|
tr->add_mtd(tr, mtd);
|
|
}
|
|
|
|
static struct mtd_notifier blktrans_notifier = {
|
|
.add = blktrans_notify_add,
|
|
.remove = blktrans_notify_remove,
|
|
};
|
|
|
|
int register_mtd_blktrans(struct mtd_blktrans_ops *tr)
|
|
{
|
|
struct mtd_info *mtd;
|
|
int ret;
|
|
|
|
/* Register the notifier if/when the first device type is
|
|
registered, to prevent the link/init ordering from fucking
|
|
us over. */
|
|
if (!blktrans_notifier.list.next)
|
|
register_mtd_user(&blktrans_notifier);
|
|
|
|
|
|
mutex_lock(&mtd_table_mutex);
|
|
|
|
ret = register_blkdev(tr->major, tr->name);
|
|
if (ret < 0) {
|
|
printk(KERN_WARNING "Unable to register %s block device on major %d: %d\n",
|
|
tr->name, tr->major, ret);
|
|
mutex_unlock(&mtd_table_mutex);
|
|
return ret;
|
|
}
|
|
|
|
if (ret)
|
|
tr->major = ret;
|
|
|
|
tr->blkshift = ffs(tr->blksize) - 1;
|
|
|
|
INIT_LIST_HEAD(&tr->devs);
|
|
list_add(&tr->list, &blktrans_majors);
|
|
|
|
mtd_for_each_device(mtd)
|
|
if (mtd->type != MTD_ABSENT)
|
|
tr->add_mtd(tr, mtd);
|
|
|
|
mutex_unlock(&mtd_table_mutex);
|
|
return 0;
|
|
}
|
|
|
|
int deregister_mtd_blktrans(struct mtd_blktrans_ops *tr)
|
|
{
|
|
struct mtd_blktrans_dev *dev, *next;
|
|
|
|
mutex_lock(&mtd_table_mutex);
|
|
|
|
/* Remove it from the list of active majors */
|
|
list_del(&tr->list);
|
|
|
|
list_for_each_entry_safe(dev, next, &tr->devs, list)
|
|
tr->remove_dev(dev);
|
|
|
|
unregister_blkdev(tr->major, tr->name);
|
|
mutex_unlock(&mtd_table_mutex);
|
|
|
|
BUG_ON(!list_empty(&tr->devs));
|
|
return 0;
|
|
}
|
|
|
|
static void __exit mtd_blktrans_exit(void)
|
|
{
|
|
/* No race here -- if someone's currently in register_mtd_blktrans
|
|
we're screwed anyway. */
|
|
if (blktrans_notifier.list.next)
|
|
unregister_mtd_user(&blktrans_notifier);
|
|
}
|
|
|
|
module_exit(mtd_blktrans_exit);
|
|
|
|
EXPORT_SYMBOL_GPL(register_mtd_blktrans);
|
|
EXPORT_SYMBOL_GPL(deregister_mtd_blktrans);
|
|
EXPORT_SYMBOL_GPL(add_mtd_blktrans_dev);
|
|
EXPORT_SYMBOL_GPL(del_mtd_blktrans_dev);
|
|
|
|
MODULE_AUTHOR("David Woodhouse <dwmw2@infradead.org>");
|
|
MODULE_LICENSE("GPL");
|
|
MODULE_DESCRIPTION("Common interface to block layer for MTD 'translation layers'");
|