Re: [PATCH v2] md: Fix the null-ptr-deref of 'mddev->private' while submitting IO

From: Zhihao Cheng

Date: Fri Sep 18 2026 - 02:40:08 EST


在 2026/9/18 14:30, yu kuai 写道:
Hi,

在 2026/9/18 14:19, Zhihao Cheng 写道:
在 2026/9/18 14:06, yu kuai 写道:
Hi,

在 2026/9/17 10:42, Zhihao Cheng 写道:
Concurrent processes md_stop and IO submitting could trigger a
null-ptr-deref of 'mddev->private':

   BUG: kernel NULL pointer dereference, address: 0000000000000070
   RIP: 0010:_wait_barrier+0x2f/0x250
   Call Trace:
    raid1_make_request+0x150/0xf50
    md_handle_request+0x104/0x530
    md_submit_bio+0x76/0x130
    submit_bio+0xdd/0x250
    submit_bio_wait+0x1f/0x40
    __blkdev_direct_IO_simple+0x1f6/0x370
    blkdev_write_iter+0x3b2/0x520
    ksys_write+0x7d/0x190

                P1
    fd = open(/dev/md0, O_RDWR)
                          P2 (forked from P1, fd' <= fd)
    write(fd)
     submit_bio
      md_handle_request
       raid1_make_request
        raid1_write_request
                                    ioctl(fd, STOP_ARRAY)
                           mddev_set_closing_and_sync_blockdev
                   // check passed, mddev->openers = 1,
                   // because md_open() is only called
                   // once in P1->open
                   do_md_stop
                    __md_stop
                     mddev->private = NULL

         conf = mddev->private // NULL
          wait_barrier(conf, sector) // null-ptr-deref !

It is a common problem for raid0/1/10/5, and __md_stop could be
triggered by several paths(eg. ioctl, sysfs, ->dtr). Fix it by
replacing mddev_lock() with mddev_suspend_and_lock() for all
__md_stop() callers. The caller dm_table_destroy() is guaranteed
being invoked with device suspended, so raid_dtr() could keep
using mddev_lock_nointr(). Besides, fail the submitting IO in
md_handle_request() if the 'mddev->pers' becomes NULL.

Fetch a reproducer in
https://bugzilla.kernel.org/show_bug.cgi?id=222020

Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2")
Reported-by: syzbot+3fe892ea5fc292e1353f@xxxxxxxxxxxxxxxxxxxxxxxxx
Closes: https://syzkaller.appspot.com/bug?extid=3fe892ea5fc292e1353f
Signed-off-by: Zhihao Cheng <chengzhihao1@xxxxxxxxxx>
---
   v1->v2:
    1. Add 'mddev->pers != NULL' check before make_request
    2. Delete dm-raid caller(->dtr) modifications
    3. Move memalloc_noio_restore after mddev_unlock_and_resume
   drivers/md/md.c | 32 +++++++++++++++++++++++++++++---
   1 file changed, 29 insertions(+), 3 deletions(-)

diff --git a/drivers/md/md.c b/drivers/md/md.c
index 680b34a63cb3..3dfc34caa2dc 100644
--- a/drivers/md/md.c
+++ b/drivers/md/md.c
@@ -414,6 +414,20 @@ bool md_handle_request(struct mddev *mddev,
struct bio *bio)
           if (!percpu_ref_tryget_live(&mddev->active_io))
               goto check_suspended;
       }
+    if (!mddev->pers) {
+        /*
+         * The __md_stop() sets 'mddev->private' to NULL during
+         * the IO submitting, check 'mddev->pers' before the IO
+         * being processed by specific driver to avoid the
+         * null-ptr-deref of 'mddev-><member>'. The check is
+         * safe because the IO has got the 'mddev->active_io'
+         * reference, and all __md_stop() callers will wait for
+         * the reference to be zero.
+         */
+        bio_io_error(bio);
+        percpu_ref_put(&mddev->active_io);
+        return true;
+    }
       if (!mddev->pers->make_request(mddev, bio)) {
           percpu_ref_put(&mddev->active_io);
           if (mddev_is_dm(mddev) && mddev->pers->prepare_suspend)
@@ -4657,6 +4671,8 @@ array_state_store(struct mddev *mddev, const
char *buf, size_t len)
   {
       int err = 0;
       enum array_state st = match_word(buf, array_states);
+    unsigned int noio_flags = 0;
+    bool suspend = false;
          /* No lock dependent actions */
       switch (st) {
@@ -4666,9 +4682,11 @@ array_state_store(struct mddev *mddev, const
char *buf, size_t len)
       case broken:        /* cannot be set */
       case bad_word:
           return -EINVAL;
+    case inactive:
       case clear:
+        suspend = true;
+        fallthrough;
       case readonly:
-    case inactive:
       case read_auto:
           if (!mddev->pers || !md_is_rdwr(mddev))
               break;
@@ -4702,9 +4720,11 @@ array_state_store(struct mddev *mddev, const
char *buf, size_t len)
           spin_unlock(&mddev->lock);
           return err ?: len;
       }
-    err = mddev_lock(mddev);
+    err = suspend ? mddev_suspend_and_lock(mddev) : mddev_lock(mddev);

Unlike ioctl path, where the checking of opener is just 1,
mddev_set_closing_and_sync_blockdev() here
already check there is no opener, so there can't be any IO inflight.

Hi, yukuai
I thought of a scene. If someone open '/dev/md0, write(buffer) 4K, and
close the corresponding fd. Then writeback kworker submit IO,
array_state_store will pass the mddev_set_closing_and_sync_blockdev()
check, which could lead to the similar problem.
This should not happen, that last one to close /dev/md0 will wait for
full disk flush to be done.

Makes sense, sync_blockdev will be called in close process.

       if (err)
           return err;
+    if (suspend)
+        noio_flags = memalloc_noio_save();
          switch (st) {
       case inactive:
@@ -4775,7 +4795,12 @@ array_state_store(struct mddev *mddev, const
char *buf, size_t len)
               mddev->hold_active = 0;
           sysfs_notify_dirent_safe(mddev->sysfs_state);
       }
-    mddev_unlock(mddev);
+    if (suspend) {
+        mddev_unlock_and_resume(mddev);
+        memalloc_noio_restore(noio_flags);
+    } else {
+        mddev_unlock(mddev);
+    }
          if (st == readonly || st == read_auto || st == inactive ||
           (err && st == clear))
@@ -8299,6 +8324,7 @@ static bool md_ioctl_need_suspend(unsigned int
cmd)
       case HOT_REMOVE_DISK:
       case SET_BITMAP_FILE:
       case SET_ARRAY_INFO:
+    case STOP_ARRAY:
           return true;
       default:
           return false;