nvme: reject passthrough of driver-managed Set Features

Since commit b58da2d270 ("nvme: update keep alive interval when kato
is modified"), a Set Features (KATO) passthrough command lets userspace
start keep-alive on any transport. nvme_keep_alive_work() allocates with
BLK_MQ_REQ_RESERVED, but nvme_alloc_admin_tag_set() reserves admin tags
only for fabrics, so on other transports the allocation trips
WARN_ON_ONCE() in blk_mq_get_tag() and fails:

  nvme nvme0: keep-alive failed: -11

Several Set Features change controller state the driver manages itself
and cannot react to when set behind its back. Reject these in
nvme_admin_cmd_allowed():

  - KATO on non-fabrics (keep-alive is only armed for fabrics; on PCIe
    it has no reserved tag and harms idle power states)
  - Host Behavior Support, Host Memory Buffer, Number of Queues, and
    Autonomous Power State Transition (all driver-managed)

Keep Alive on fabrics is unchanged; I/O commands are unaffected as the
check is confined to the admin path (ns == NULL).

Link: https://lore.kernel.org/linux-nvme/20260523225629.3964037-1-coshi036@gmail.com/

Fixes: b58da2d270 ("nvme: update keep alive interval when kato is modified")

Found by FuzzNvme.

Acked-by: Sungwoo Kim <iam@sung-woo.kim>
Acked-by: Dave Tian <daveti@purdue.edu>
Acked-by: Weidong Zhu <weizhu@fiu.edu>
Signed-off-by: Chao Shi <coshi036@gmail.com>
Signed-off-by: Keith Busch <kbusch@kernel.org>
This commit is contained in:
Chao Shi
2026-07-15 11:57:52 -04:00
committed by Keith Busch
parent f594863967
commit 1161be71d1

View File

@@ -14,45 +14,54 @@ enum {
NVME_IOCTL_PARTITION = (1 << 1),
};
static bool nvme_cmd_allowed(struct nvme_ns *ns, struct nvme_command *c,
unsigned int flags, bool open_for_write)
static bool nvme_admin_cmd_allowed(struct nvme_ctrl *ctrl,
struct nvme_command *c)
{
u32 effects;
/*
* Do not allow unprivileged passthrough on partitions, as that allows an
* escape from the containment of the partition.
*/
if (flags & NVME_IOCTL_PARTITION)
goto admin;
/*
* Do not allow unprivileged processes to send vendor specific or fabrics
* commands as we can't be sure about their effects.
*/
if (c->common.opcode >= nvme_cmd_vendor_start ||
c->common.opcode == nvme_fabrics_command)
goto admin;
/*
* Do not allow unprivileged passthrough of admin commands except
* for a subset of identify commands that contain information required
* to form proper I/O commands in userspace and do not expose any
* potentially sensitive information.
*/
if (!ns) {
if (c->common.opcode == nvme_admin_identify) {
switch (c->identify.cns) {
case NVME_ID_CNS_NS:
case NVME_ID_CNS_CS_NS:
case NVME_ID_CNS_NS_CS_INDEP:
case NVME_ID_CNS_CS_CTRL:
case NVME_ID_CNS_CTRL:
return true;
}
switch (c->common.opcode) {
case nvme_admin_identify:
switch (c->identify.cns) {
case NVME_ID_CNS_NS:
case NVME_ID_CNS_CS_NS:
case NVME_ID_CNS_NS_CS_INDEP:
case NVME_ID_CNS_CS_CTRL:
case NVME_ID_CNS_CTRL:
return true;
}
goto admin;
break;
case nvme_admin_set_features:
/*
* Reject Set Features that change controller state the driver
* manages itself; setting them behind the driver's back from
* userspace leaves it unable to react correctly. Keep Alive is
* only armed for fabrics - on other transports it has no
* reserved tag and harms idle power states.
*/
switch (le32_to_cpu(c->features.fid) & 0xff) {
case NVME_FEAT_KATO:
if (ctrl->ops->flags & NVME_F_FABRICS)
break;
fallthrough;
case NVME_FEAT_HOST_BEHAVIOR:
case NVME_FEAT_HOST_MEM_BUF:
case NVME_FEAT_NUM_QUEUES:
case NVME_FEAT_AUTO_PST:
return false;
}
break;
}
return capable(CAP_SYS_ADMIN);
}
static bool nvme_ns_cmd_allowed(struct nvme_ns *ns, struct nvme_command *c,
bool open_for_write)
{
u32 effects;
/*
* Check if the controller provides a Commands Supported and Effects log
@@ -61,7 +70,7 @@ static bool nvme_cmd_allowed(struct nvme_ns *ns, struct nvme_command *c,
*/
effects = nvme_command_effects(ns->ctrl, ns, c->common.opcode);
if (!(effects & NVME_CMD_EFFECTS_CSUPP))
goto admin;
return capable(CAP_SYS_ADMIN);
/*
* Don't allow passthrough for command that have intrusive (or unknown)
@@ -70,7 +79,7 @@ static bool nvme_cmd_allowed(struct nvme_ns *ns, struct nvme_command *c,
if (effects & ~(NVME_CMD_EFFECTS_CSUPP | NVME_CMD_EFFECTS_LBCC |
NVME_CMD_EFFECTS_UUID_SEL |
NVME_CMD_EFFECTS_SCOPE_MASK))
goto admin;
return capable(CAP_SYS_ADMIN);
/*
* Only allow I/O commands that transfer data to the controller or that
@@ -79,11 +88,34 @@ static bool nvme_cmd_allowed(struct nvme_ns *ns, struct nvme_command *c,
*/
if ((nvme_is_write(c) || (effects & NVME_CMD_EFFECTS_LBCC)) &&
!open_for_write)
goto admin;
return capable(CAP_SYS_ADMIN);
return true;
admin:
return capable(CAP_SYS_ADMIN);
}
static bool nvme_cmd_allowed(struct nvme_ctrl *ctrl, struct nvme_ns *ns,
struct nvme_command *c, unsigned int flags,
bool open_for_write)
{
/*
* Do not allow unprivileged passthrough on partitions, as that
* allows an escape from the containment of the partition.
*/
if (flags & NVME_IOCTL_PARTITION)
return capable(CAP_SYS_ADMIN);
/*
* Do not allow unprivileged processes to send vendor specific or
* fabrics commands as we can't be sure about their effects.
*/
if (c->common.opcode >= nvme_cmd_vendor_start ||
c->common.opcode == nvme_fabrics_command)
return capable(CAP_SYS_ADMIN);
if (!ns)
return nvme_admin_cmd_allowed(ctrl, c);
return nvme_ns_cmd_allowed(ns, c, open_for_write);
}
/*
@@ -261,7 +293,7 @@ static int nvme_submit_io(struct nvme_ns *ns, struct nvme_user_io __user *uio,
c.rw.lbat = cpu_to_le16(io.apptag);
c.rw.lbatm = cpu_to_le16(io.appmask);
if (!nvme_cmd_allowed(ns, &c, flags, open_for_write))
if (!nvme_cmd_allowed(ns->ctrl, ns, &c, flags, open_for_write))
return -EACCES;
return nvme_submit_user_cmd(ns->queue, &c, io.addr, length, metadata,
@@ -311,7 +343,7 @@ static int nvme_user_cmd(struct nvme_ctrl *ctrl, struct nvme_ns *ns,
c.common.cdw14 = cpu_to_le32(cmd.cdw14);
c.common.cdw15 = cpu_to_le32(cmd.cdw15);
if (!nvme_cmd_allowed(ns, &c, 0, open_for_write))
if (!nvme_cmd_allowed(ctrl, ns, &c, 0, open_for_write))
return -EACCES;
if (cmd.timeout_ms)
@@ -358,7 +390,7 @@ static int nvme_user_cmd64(struct nvme_ctrl *ctrl, struct nvme_ns *ns,
c.common.cdw14 = cpu_to_le32(cmd.cdw14);
c.common.cdw15 = cpu_to_le32(cmd.cdw15);
if (!nvme_cmd_allowed(ns, &c, flags, open_for_write))
if (!nvme_cmd_allowed(ctrl, ns, &c, flags, open_for_write))
return -EACCES;
if (cmd.timeout_ms)
@@ -453,6 +485,7 @@ static int nvme_uring_cmd_io(struct nvme_ctrl *ctrl, struct nvme_ns *ns,
const struct nvme_uring_cmd *cmd = io_uring_sqe128_cmd(ioucmd->sqe,
struct nvme_uring_cmd);
struct request_queue *q = ns ? ns->queue : ctrl->admin_q;
bool open_for_write = ioucmd->file->f_mode & FMODE_WRITE;
struct nvme_uring_data d;
struct nvme_command c;
struct iov_iter iter;
@@ -483,7 +516,7 @@ static int nvme_uring_cmd_io(struct nvme_ctrl *ctrl, struct nvme_ns *ns,
c.common.cdw14 = cpu_to_le32(READ_ONCE(cmd->cdw14));
c.common.cdw15 = cpu_to_le32(READ_ONCE(cmd->cdw15));
if (!nvme_cmd_allowed(ns, &c, 0, ioucmd->file->f_mode & FMODE_WRITE))
if (!nvme_cmd_allowed(ctrl, ns, &c, 0, open_for_write))
return -EACCES;
d.metadata = READ_ONCE(cmd->metadata);