Page MenuHomeFreeBSD

D60414.diff
No OneTemporary

D60414.diff

diff --git a/share/man/man4/nda.4 b/share/man/man4/nda.4
--- a/share/man/man4/nda.4
+++ b/share/man/man4/nda.4
@@ -25,7 +25,7 @@
.\" OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
.\" SUCH DAMAGE.
.\"
-.Dd October 2, 2025
+.Dd August 9, 2026
.Dt NDA 4
.Os
.Sh NAME
@@ -37,8 +37,7 @@
.Sh DESCRIPTION
The
.Nm
-driver provides support for direct access devices, implementing the
-.Tn NVMe
+driver provides support for direct access devices, implementing the NVMe
command protocol, that are attached to the system through a host adapter
supported by the CAM subsystem.
.Sh HARDWARE
@@ -165,7 +164,36 @@
Total number of
.Va BIO_DELETE
requests queued to the device.
+.It Va kern.cam.nda.N.zone_mode
+The zone mode of the namespace, either
+.Dq Host Managed
+for Zoned Namespaces or
+.Dq Not Zoned .
+.It Va kern.cam.nda.N.zone_size
+The size of each zone in logical blocks, for zoned namespaces.
+.It Va kern.cam.nda.N.max_open_zones
+The maximum number of zones that may be open at once, for zoned
+namespaces.
+Zero means the device reports no limit.
+.It Va kern.cam.nda.N.max_active_zones
+The maximum number of zones that may be active
+.Pq open or closed
+at once, for zoned namespaces.
+Zero means the device reports no limit.
+.It Va kern.cam.nda.N.read_across_zones
+Whether a single read may span more than one zone, for zoned namespaces.
.El
+.Sh ZONED NAMESPACES
+Namespaces implementing the NVMe Zoned Namespace
+.Pq ZNS
+command set are exposed as host managed zoned block devices.
+Zones can be analyzed and managed through
+.Xr zonectl 8 ,
+or programmatically via the
+.Va DIOCZONECMD
+.Xr ioctl 2 .
+Writes within a zone must be sequential, starting at the zone's write
+pointer.
.Sh NAMESPACE MAPPING
Each
.Xr nvme 4
@@ -205,7 +233,8 @@
.Xr geom 4 ,
.Xr nvd 4 ,
.Xr nvme 4 ,
-.Xr gpart 8
+.Xr gpart 8 ,
+.Xr zonectl 8
.Sh HISTORY
The
.Nm
diff --git a/sys/cam/cam_xpt.c b/sys/cam/cam_xpt.c
--- a/sys/cam/cam_xpt.c
+++ b/sys/cam/cam_xpt.c
@@ -4906,6 +4906,7 @@
free(device->serial_num, M_CAMXPT);
free(device->nvme_data, M_CAMXPT);
free(device->nvme_cdata, M_CAMXPT);
+ free(device->nvme_zns_data, M_CAMXPT);
taskqueue_enqueue(xsoftc.xpt_taskq, &device->device_destroy_task);
}
diff --git a/sys/cam/cam_xpt_internal.h b/sys/cam/cam_xpt_internal.h
--- a/sys/cam/cam_xpt_internal.h
+++ b/sys/cam/cam_xpt_internal.h
@@ -151,6 +151,7 @@
struct task device_destroy_task;
struct nvme_controller_data *nvme_cdata;
struct nvme_namespace_data *nvme_data;
+ struct nvme_zns_namespace_data *nvme_zns_data;
};
/*
diff --git a/sys/cam/nvme/nvme_all.h b/sys/cam/nvme/nvme_all.h
--- a/sys/cam/nvme/nvme_all.h
+++ b/sys/cam/nvme/nvme_all.h
@@ -48,5 +48,6 @@
int nvme_status_sbuf(struct ccb_nvmeio *nvmeio, struct sbuf *sb);
const void *nvme_get_identify_cntrl(struct cam_periph *);
const void *nvme_get_identify_ns(struct cam_periph *);
+const void *nvme_get_identify_ns_zns(struct cam_periph *);
#endif /* CAM_NVME_NVME_ALL_H */
diff --git a/sys/cam/nvme/nvme_all.c b/sys/cam/nvme/nvme_all.c
--- a/sys/cam/nvme/nvme_all.c
+++ b/sys/cam/nvme/nvme_all.c
@@ -190,4 +190,14 @@
return device->nvme_data;
}
+
+const void *
+nvme_get_identify_ns_zns(struct cam_periph *periph)
+{
+ struct cam_ed *device;
+
+ device = periph->path->device;
+
+ return device->nvme_zns_data;
+}
#endif
diff --git a/sys/cam/nvme/nvme_da.c b/sys/cam/nvme/nvme_da.c
--- a/sys/cam/nvme/nvme_da.c
+++ b/sys/cam/nvme/nvme_da.c
@@ -103,6 +103,11 @@
NDA_CCB_TYPE_MASK = 0x0F,
} nda_ccb_state;
+typedef enum {
+ NDA_ZONE_NONE = 0x00,
+ NDA_ZONE_HOST_MANAGED = 0x01,
+} nda_zone_mode;
+
/* Offsets into our private area for storing information */
#define ccb_state ccb_h.ppriv_field0
#define ccb_bp ccb_h.ppriv_ptr1 /* For NDA_CCB_BUFFER_IO */
@@ -125,6 +130,11 @@
uint64_t trim_count;
uint64_t trim_ranges;
uint64_t trim_lbas;
+ nda_zone_mode zone_mode;
+ uint64_t zone_size; /* Zone size in LBAs */
+ uint64_t max_open_zones; /* 0 == no limit */
+ uint64_t max_active_zones; /* 0 == no limit */
+ bool read_across_zones; /* OZCS.RAZB */
#ifdef CAM_TEST_FAILURE
int force_read_error;
int force_write_error;
@@ -157,6 +167,7 @@
struct cam_path *path, void *arg);
static void ndasysctlinit(void *context, int pending);
static int ndaflagssysctl(SYSCTL_HANDLER_ARGS);
+static int ndazonemodesysctl(SYSCTL_HANDLER_ARGS);
static periph_ctor_t ndaregister;
static periph_dtor_t ndacleanup;
static periph_start_t ndastart;
@@ -291,17 +302,322 @@
nvme_ns_rw_cmd(&nvmeio->cmd, rwcmd, softc->nsid, lba, count);
}
+static int
+nda_zone_bio_to_nvme(int disk_zone_cmd)
+{
+ switch (disk_zone_cmd) {
+ case DISK_ZONE_OPEN:
+ return (NVME_ZONE_SEND_OPEN);
+ case DISK_ZONE_CLOSE:
+ return (NVME_ZONE_SEND_CLOSE);
+ case DISK_ZONE_FINISH:
+ return (NVME_ZONE_SEND_FINISH);
+ case DISK_ZONE_RWP:
+ return (NVME_ZONE_SEND_RESET);
+ }
+
+ return (-1);
+}
+
+static int
+nda_zone_rep_to_nvme(uint8_t rep_options)
+{
+ switch (rep_options) {
+ case DISK_ZONE_REP_ALL:
+ return (NVME_ZONE_REPORT_ALL);
+ case DISK_ZONE_REP_EMPTY:
+ return (NVME_ZONE_REPORT_EMPTY);
+ case DISK_ZONE_REP_IMP_OPEN:
+ return (NVME_ZONE_REPORT_IMP_OPEN);
+ case DISK_ZONE_REP_EXP_OPEN:
+ return (NVME_ZONE_REPORT_EXP_OPEN);
+ case DISK_ZONE_REP_CLOSED:
+ return (NVME_ZONE_REPORT_CLOSED);
+ case DISK_ZONE_REP_FULL:
+ return (NVME_ZONE_REPORT_FULL);
+ case DISK_ZONE_REP_READONLY:
+ return (NVME_ZONE_REPORT_READONLY);
+ case DISK_ZONE_REP_OFFLINE:
+ return (NVME_ZONE_REPORT_OFFLINE);
+ }
+
+ return (-1);
+}
+
+static int
+nda_zone_cmd(struct cam_periph *periph, union ccb *ccb, struct bio *bp,
+ int *queue_ccb)
+{
+ struct nda_softc *softc;
+ int error;
+
+ error = 0;
+
+ if (bp->bio_cmd != BIO_ZONE) {
+ error = EINVAL;
+ goto bailout;
+ }
+
+ softc = periph->softc;
+
+ switch (bp->bio_zone.zone_cmd) {
+ case DISK_ZONE_OPEN:
+ case DISK_ZONE_CLOSE:
+ case DISK_ZONE_FINISH:
+ case DISK_ZONE_RWP: {
+ int send_action;
+ bool select_all;
+
+ send_action = nda_zone_bio_to_nvme(bp->bio_zone.zone_cmd);
+ if (send_action == -1) {
+ xpt_print(periph->path, "Cannot translate zone "
+ "cmd %#x to NVMe\n", bp->bio_zone.zone_cmd);
+ error = EINVAL;
+ goto bailout;
+ }
+
+ select_all = (bp->bio_zone.zone_params.rwp.flags &
+ DISK_ZONE_RWP_FLAG_ALL) != 0;
+
+ cam_fill_nvmeio(&ccb->nvmeio,
+ 0, /* retries */
+ ndadone, /* cbfcnp */
+ CAM_DIR_NONE, /* flags */
+ NULL, /* data_ptr */
+ 0, /* dxfer_len */
+ nda_default_timeout * 1000); /* timeout 30s */
+ nvme_zns_mgmt_send_cmd(&ccb->nvmeio.cmd, softc->nsid,
+ bp->bio_zone.zone_params.rwp.id, select_all,
+ send_action);
+ *queue_ccb = 1;
+
+ break;
+ }
+ case DISK_ZONE_REPORT_ZONES: {
+ uint8_t *rz_ptr;
+ uint32_t num_entries, max_entries, alloc_size;
+ struct disk_zone_report *rep;
+ int resp_option;
+
+ rep = &bp->bio_zone.zone_params.report;
+
+ num_entries = rep->entries_allocated;
+ if (num_entries == 0) {
+ xpt_print(periph->path, "No entries allocated for "
+ "Report Zones request\n");
+ error = EINVAL;
+ goto bailout;
+ }
+ resp_option = nda_zone_rep_to_nvme(rep->rep_options);
+ if (resp_option == -1) {
+ xpt_print(periph->path, "Cannot translate zone "
+ "reporting option %#x to NVMe\n",
+ rep->rep_options);
+ error = EINVAL;
+ goto bailout;
+ }
+ /*
+ * Trim the request to what the controller transfers in a
+ * single command. This keeps room for the report header and a
+ * whole number of zone descriptors.
+ */
+ max_entries = (softc->disk->d_maxsize -
+ sizeof(struct nvme_zone_report)) /
+ sizeof(struct nvme_zone_descriptor);
+ num_entries = MIN(num_entries, max_entries);
+ alloc_size = sizeof(struct nvme_zone_report) +
+ (sizeof(struct nvme_zone_descriptor) * num_entries);
+ rz_ptr = malloc(alloc_size, M_NVMEDA, M_NOWAIT | M_ZERO);
+ if (rz_ptr == NULL) {
+ xpt_print(periph->path, "Unable to allocate memory "
+ "for Report Zones request\n");
+ error = ENOMEM;
+ goto bailout;
+ }
+
+ cam_fill_nvmeio(&ccb->nvmeio,
+ 0, /* retries */
+ ndadone, /* cbfcnp */
+ CAM_DIR_IN, /* flags */
+ rz_ptr, /* data_ptr */
+ alloc_size, /* dxfer_len */
+ nda_default_timeout * 1000); /* timeout 30s */
+ /*
+ * Ask for a *full* report so that the number of zones returned
+ * in the header is the total number of zones matching the
+ * reporting options: this is what the entries_available field
+ * wants.
+ */
+ nvme_zns_mgmt_recv_cmd(&ccb->nvmeio.cmd, softc->nsid,
+ rep->starting_id, alloc_size, NVME_ZONE_RECV_REPORT,
+ resp_option, false);
+
+ /*
+ * BIO_ZONE would not normally need this. However, this is used
+ * by devstat_end_transaction_bio() to determine how much data
+ * was transferred. Because the size of the NVMe structs is
+ * different than the size of the BIO interface structs, the
+ * amount of data that is actually transferred from the drive
+ * will be different than the amount of data transferred to the
+ * user.
+ */
+ bp->bio_bcount = bp->bio_length;
+
+ *queue_ccb = 1;
+
+ break;
+ }
+ case DISK_ZONE_GET_PARAMS: {
+ struct disk_zone_disk_params *params;
+
+ params = &bp->bio_zone.zone_params.disk_params;
+ bzero(params, sizeof(*params));
+
+ switch (softc->zone_mode) {
+ case NDA_ZONE_HOST_MANAGED:
+ params->zone_mode = DISK_ZONE_MODE_HOST_MANAGED;
+ break;
+ default:
+ case NDA_ZONE_NONE:
+ params->zone_mode = DISK_ZONE_MODE_NONE;
+ break;
+ }
+
+ /*
+ * Reads are permitted anywhere in a zone that is not in the
+ * ZSO:Offline state.
+ */
+ params->flags |= DISK_ZONE_DISK_URSWRZ;
+
+ if (softc->max_open_zones != 0) {
+ params->max_seq_zones = softc->max_open_zones;
+ params->flags |= DISK_ZONE_MAX_SEQ_SET;
+ }
+
+ params->flags |= DISK_ZONE_RZ_SUP | DISK_ZONE_OPEN_SUP |
+ DISK_ZONE_CLOSE_SUP | DISK_ZONE_FINISH_SUP |
+ DISK_ZONE_RWP_SUP;
+ break;
+ }
+ default:
+ break;
+ }
+bailout:
+ return (error);
+}
+
+static void
+ndazonedone(struct cam_periph *periph, union ccb *ccb)
+{
+ struct nda_softc *softc;
+ struct bio *bp;
+
+ softc = periph->softc;
+ bp = (struct bio *)ccb->ccb_bp;
+
+ switch (bp->bio_zone.zone_cmd) {
+ case DISK_ZONE_OPEN:
+ case DISK_ZONE_CLOSE:
+ case DISK_ZONE_FINISH:
+ case DISK_ZONE_RWP:
+ break;
+ case DISK_ZONE_REPORT_ZONES: {
+ uint32_t avail_len, max_desc;
+ struct disk_zone_report *rep;
+ struct nvme_zone_report *hdr;
+ struct nvme_zone_descriptor *desc;
+ struct disk_zone_rep_entry *entry;
+ uint64_t num_avail;
+ uint32_t num_to_fill, i;
+
+ rep = &bp->bio_zone.zone_params.report;
+ avail_len = ccb->nvmeio.dxfer_len;
+ hdr = (struct nvme_zone_report *)ccb->nvmeio.data_ptr;
+
+ /*
+ * The transfer is dxfer_len bytes and the buffer was
+ * zeroed before the command, so every descriptor slot in
+ * it is safe to byte swap, whether or not the device
+ * filled it in.
+ */
+ max_desc = (avail_len - sizeof(*hdr)) / sizeof(*desc);
+ nvme_zone_report_swapbytes(hdr, max_desc);
+
+ /*
+ * Every zone has the same length (ZSZE) and the same type,
+ * which is all the SAME field describes. Zone capacity may
+ * vary per zone, but does not affect it.
+ */
+ rep->header.same = DISK_ZONE_SAME_ALL_SAME;
+ rep->header.maximum_lba = softc->disk->d_mediasize /
+ softc->disk->d_sectorsize - 1;
+ rep->entries_available = MIN(hdr->nr_zones, UINT32_MAX);
+
+ num_avail = MIN(hdr->nr_zones, max_desc);
+ num_to_fill = MIN(num_avail, rep->entries_allocated);
+ if (num_to_fill == 0) {
+ rep->entries_filled = 0;
+ bp->bio_resid = bp->bio_bcount;
+ break;
+ }
+
+ for (i = 0, desc = &hdr->zone_desc[0], entry = &rep->entries[0];
+ i < num_to_fill; i++, desc++, entry++) {
+ if (NVMEV(NVME_ZONE_DESC_ZT, desc->zt) ==
+ NVME_ZONE_TYPE_SEQUENTIAL)
+ entry->zone_type = DISK_ZONE_TYPE_SEQ_REQUIRED;
+ else
+ entry->zone_type =
+ NVMEV(NVME_ZONE_DESC_ZT, desc->zt);
+ entry->zone_condition =
+ NVMEV(NVME_ZONE_DESC_ZS, desc->zs);
+ entry->zone_flags = 0;
+ if (NVMEV(NVME_ZONE_DESC_ZA_RZR, desc->za))
+ entry->zone_flags |= DISK_ZONE_FLAG_RESET;
+ entry->zone_length = softc->zone_size;
+ entry->zone_capacity = desc->zcap;
+ entry->zone_start_lba = desc->zslba;
+ entry->write_pointer_lba = desc->wp;
+ }
+ rep->entries_filled = num_to_fill;
+ /*
+ * Note that this residual is accurate from the user's
+ * standpoint, but the amount transferred isn't accurate
+ * from the standpoint of what actually came back from the
+ * drive.
+ */
+ bp->bio_resid = bp->bio_bcount - (num_to_fill * sizeof(*entry));
+ break;
+ }
+ case DISK_ZONE_GET_PARAMS:
+ default:
+ /*
+ * In theory we should not get a GET_PARAMS bio, since it
+ * should be handled without queueing the command to the
+ * drive.
+ */
+ panic("%s: Invalid zone command %d", __func__,
+ bp->bio_zone.zone_cmd);
+ break;
+ }
+
+ if (bp->bio_zone.zone_cmd == DISK_ZONE_REPORT_ZONES)
+ free(ccb->nvmeio.data_ptr, M_NVMEDA);
+}
+
static void
ndasetgeom(struct nda_softc *softc, struct cam_periph *periph)
{
struct disk *disk = softc->disk;
const struct nvme_namespace_data *nsd;
const struct nvme_controller_data *cd;
+ const struct nvme_zns_namespace_data *znsd;
uint8_t flbas_fmt, lbads, vwc_present;
u_int flags;
nsd = nvme_get_identify_ns(periph);
cd = nvme_get_identify_cntrl(periph);
+ znsd = nvme_get_identify_ns_zns(periph);
/*
* Preserve flags we can't infer that were set before. UNMAPPED comes
@@ -322,6 +638,26 @@
if (vwc_present)
disk->d_flags |= DISKFLAG_CANFLUSHCACHE;
disk->d_flags |= flags;
+
+ if (znsd != NULL) {
+ /*
+ * Zoned namespaces are always host managed.
+ */
+ softc->zone_mode = NDA_ZONE_HOST_MANAGED;
+ softc->zone_size = znsd->lbafe[flbas_fmt].zsze;
+ /* MAR and MOR are 0's based. */
+ softc->max_active_zones =
+ znsd->mar == NVME_ZNS_NS_DATA_RESOURCES_UNLIMITED ? 0 :
+ (uint64_t)znsd->mar + 1;
+ softc->max_open_zones =
+ znsd->mor == NVME_ZNS_NS_DATA_RESOURCES_UNLIMITED ? 0 :
+ (uint64_t)znsd->mor + 1;
+ softc->read_across_zones =
+ NVMEV(NVME_ZNS_NS_DATA_OZCS_RAZB, znsd->ozcs) != 0;
+ disk->d_flags |= DISKFLAG_CANZONE;
+ } else {
+ softc->zone_mode = NDA_ZONE_NONE;
+ }
}
static void
@@ -565,6 +901,13 @@
if (bp->bio_cmd == BIO_DELETE)
softc->deletes++;
+ /*
+ * Zone cmds must be ordered, as they can depend on the effects of
+ * previously issued commands, which may affect commands after them.
+ */
+ if (bp->bio_cmd == BIO_ZONE)
+ bp->bio_flags |= BIO_ORDERED;
+
/*
* Place it in the queue of disk activities for this disk
*/
@@ -869,6 +1212,27 @@
softc, 0, ndaflagssysctl, "A",
"Flags for drive");
+ SYSCTL_ADD_PROC(&softc->sysctl_ctx, SYSCTL_CHILDREN(softc->sysctl_tree),
+ OID_AUTO, "zone_mode", CTLTYPE_STRING | CTLFLAG_RD | CTLFLAG_MPSAFE,
+ softc, 0, ndazonemodesysctl, "A",
+ "Zone Mode");
+ SYSCTL_ADD_UQUAD(&softc->sysctl_ctx,
+ SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO,
+ "zone_size", CTLFLAG_RD, &softc->zone_size,
+ "Zone size in LBAs");
+ SYSCTL_ADD_UQUAD(&softc->sysctl_ctx,
+ SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO,
+ "max_open_zones", CTLFLAG_RD, &softc->max_open_zones,
+ "Maximum number of open zones (0 = no limit)");
+ SYSCTL_ADD_UQUAD(&softc->sysctl_ctx,
+ SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO,
+ "max_active_zones", CTLFLAG_RD, &softc->max_active_zones,
+ "Maximum number of active zones (0 = no limit)");
+ SYSCTL_ADD_BOOL(&softc->sysctl_ctx,
+ SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO,
+ "read_across_zones", CTLFLAG_RD, &softc->read_across_zones, 0,
+ "Reads may span more than one zone");
+
#ifdef CAM_IO_STATS
softc->sysctl_stats_tree = SYSCTL_ADD_NODE(&softc->sysctl_stats_ctx,
SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO, "stats",
@@ -908,6 +1272,30 @@
cam_periph_release(periph);
}
+static int
+ndazonemodesysctl(SYSCTL_HANDLER_ARGS)
+{
+ char tmpbuf[24];
+ struct nda_softc *softc;
+ int error;
+
+ softc = (struct nda_softc *)arg1;
+
+ switch (softc->zone_mode) {
+ case NDA_ZONE_HOST_MANAGED:
+ snprintf(tmpbuf, sizeof(tmpbuf), "Host Managed");
+ break;
+ case NDA_ZONE_NONE:
+ default:
+ snprintf(tmpbuf, sizeof(tmpbuf), "Not Zoned");
+ break;
+ }
+
+ error = sysctl_handle_string(oidp, tmpbuf, sizeof(tmpbuf), req);
+
+ return (error);
+}
+
static int
ndaflagssysctl(SYSCTL_HANDLER_ARGS)
{
@@ -1240,6 +1628,30 @@
case BIO_FLUSH:
nda_nvme_flush(softc, nvmeio);
break;
+ case BIO_ZONE: {
+ int error, queue_ccb;
+
+ queue_ccb = 0;
+
+ error = nda_zone_cmd(periph, start_ccb, bp, &queue_ccb);
+ if ((error != 0)
+ || (queue_ccb == 0)) {
+ /*
+ * g_io_deliver will recursively call start
+ * routine for ENOMEM... drop the periph lock
+ * to allow that recursion.
+ */
+ if (error == ENOMEM)
+ cam_periph_unlock(periph);
+ biofinish(bp, NULL, error);
+ if (error == ENOMEM)
+ cam_periph_lock(periph);
+ xpt_release_ccb(start_ccb);
+ ndaschedule(periph);
+ return;
+ }
+ break;
+ }
default:
biofinish(bp, NULL, EOPNOTSUPP);
xpt_release_ccb(start_ccb);
@@ -1311,8 +1723,14 @@
if (error != 0) {
bp->bio_resid = bp->bio_bcount;
bp->bio_flags |= BIO_ERROR;
+ if (bp->bio_cmd == BIO_ZONE &&
+ bp->bio_zone.zone_cmd ==
+ DISK_ZONE_REPORT_ZONES)
+ free(nvmeio->data_ptr, M_NVMEDA);
} else {
bp->bio_resid = 0;
+ if (bp->bio_cmd == BIO_ZONE)
+ ndazonedone(periph, done_ccb);
}
softc->outstanding_cmds--;
diff --git a/sys/cam/nvme/nvme_xpt.c b/sys/cam/nvme/nvme_xpt.c
--- a/sys/cam/nvme/nvme_xpt.c
+++ b/sys/cam/nvme/nvme_xpt.c
@@ -81,6 +81,8 @@
typedef enum {
NVME_PROBE_IDENTIFY_CD,
NVME_PROBE_IDENTIFY_NS,
+ NVME_PROBE_IDENTIFY_NS_DESCS,
+ NVME_PROBE_IDENTIFY_NS_ZNS,
NVME_PROBE_DONE,
NVME_PROBE_INVALID
} nvme_probe_action;
@@ -88,6 +90,8 @@
static char *nvme_probe_action_text[] = {
"NVME_PROBE_IDENTIFY_CD",
"NVME_PROBE_IDENTIFY_NS",
+ "NVME_PROBE_IDENTIFY_NS_DESCS",
+ "NVME_PROBE_IDENTIFY_NS_ZNS",
"NVME_PROBE_DONE",
"NVME_PROBE_INVALID"
};
@@ -111,6 +115,8 @@
union {
struct nvme_controller_data cd;
struct nvme_namespace_data ns;
+ uint8_t nsdescs[NVME_NS_ID_DESC_LIST_SIZE];
+ struct nvme_zns_namespace_data zns;
};
nvme_probe_action action;
nvme_probe_flags flags;
@@ -294,6 +300,29 @@
nvme_ns_cmd(nvmeio, NVME_OPC_IDENTIFY, lun,
0, 0, 0, 0, 0, 0);
break;
+ case NVME_PROBE_IDENTIFY_NS_DESCS:
+ cam_fill_nvmeadmin(nvmeio,
+ 0, /* retries */
+ nvme_probe_done, /* cbfcnp */
+ CAM_DIR_IN, /* flags */
+ (uint8_t *)&softc->nsdescs, /* data_ptr */
+ sizeof(softc->nsdescs), /* dxfer_len */
+ 30 * 1000); /* timeout 30s */
+ nvme_ns_cmd(nvmeio, NVME_OPC_IDENTIFY, lun,
+ NVME_CNS_ID_NS_DESC_LIST, 0, 0, 0, 0, 0);
+ break;
+ case NVME_PROBE_IDENTIFY_NS_ZNS:
+ cam_fill_nvmeadmin(nvmeio,
+ 0, /* retries */
+ nvme_probe_done, /* cbfcnp */
+ CAM_DIR_IN, /* flags */
+ (uint8_t *)&softc->zns, /* data_ptr */
+ sizeof(softc->zns), /* dxfer_len */
+ 30 * 1000); /* timeout 30s */
+ nvme_ns_cmd(nvmeio, NVME_OPC_IDENTIFY, lun,
+ NVME_CNS_ID_NS_IOCS, (uint32_t)NVME_CSI_ZNS << 24,
+ 0, 0, 0, 0);
+ break;
default:
panic("nvme_probe_start: invalid action state 0x%x\n", softc->action);
}
@@ -320,6 +349,20 @@
priority = done_ccb->ccb_h.pinfo.priority;
if ((done_ccb->ccb_h.status & CAM_STATUS_MASK) != CAM_REQ_CMP) {
+ /*
+ * Older controllers may not implement the namespace
+ * indentification descriptor list and the I/O command set
+ * specific identify data, with some SIMs rejecting the higher
+ * CNS values. Announce the device without that data instead of
+ * failing the probe.
+ */
+ if (softc->action == NVME_PROBE_IDENTIFY_NS_DESCS ||
+ softc->action == NVME_PROBE_IDENTIFY_NS_ZNS) {
+ if ((done_ccb->ccb_h.status & CAM_DEV_QFRZN) != 0)
+ xpt_release_devq(path, /*count*/1,
+ /*run_queue*/TRUE);
+ goto announce;
+ }
if (cam_periph_error(done_ccb,
0, softc->restart ? (SF_NO_RECOVERY | SF_NO_RETRY) : 0
) == ERESTART) {
@@ -348,7 +391,7 @@
NVME_PROBE_SET_ACTION(softc, NVME_PROBE_INVALID);
found = 0;
goto done;
- }
+}
if (softc->restart)
goto done;
switch (softc->action) {
@@ -457,20 +500,57 @@
path->device->device_id_len = SVPD_DEVICE_ID_HDR_LEN + len;
}
- if (periph->path->device->flags & CAM_DEV_UNCONFIGURED) {
- path->device->flags &= ~CAM_DEV_UNCONFIGURED;
- xpt_acquire_device(path->device);
- done_ccb->ccb_h.func_code = XPT_GDEV_TYPE;
- xpt_action(done_ccb);
- xpt_async(AC_FOUND_DEVICE, path, done_ccb);
- } else {
- xpt_async(AC_GETDEV_CHANGED, path, NULL);
+ NVME_PROBE_SET_ACTION(softc, NVME_PROBE_IDENTIFY_NS_DESCS);
+ xpt_release_ccb(done_ccb);
+ xpt_schedule(periph, priority);
+ goto out;
+ case NVME_PROBE_IDENTIFY_NS_DESCS: {
+ uint8_t csi;
+
+ csi = nvme_ns_id_desc_list_csi(softc->nsdescs,
+ sizeof(softc->nsdescs));
+ if (csi == NVME_CSI_ZNS) {
+ NVME_PROBE_SET_ACTION(softc, NVME_PROBE_IDENTIFY_NS_ZNS);
+ xpt_release_ccb(done_ccb);
+ xpt_schedule(periph, priority);
+ goto out;
}
- NVME_PROBE_SET_ACTION(softc, NVME_PROBE_DONE);
+ free(path->device->nvme_zns_data, M_CAMXPT);
+ path->device->nvme_zns_data = NULL;
break;
+ }
+ case NVME_PROBE_IDENTIFY_NS_ZNS: {
+ struct nvme_zns_namespace_data *zns_data;
+
+ nvme_zns_namespace_data_swapbytes(&softc->zns);
+
+ zns_data = path->device->nvme_zns_data;
+ if (zns_data == NULL) {
+ zns_data = malloc(sizeof(*zns_data), M_CAMXPT,
+ M_NOWAIT);
+ if (zns_data == NULL) {
+ xpt_print(path, "Can't allocate memory");
+ goto device_fail;
+ }
+ }
+ bcopy(&softc->zns, zns_data, sizeof(*zns_data));
+ path->device->nvme_zns_data = zns_data;
+ break;
+ }
default:
panic("nvme_probe_done: invalid action state 0x%x\n", softc->action);
}
+announce:
+ if (periph->path->device->flags & CAM_DEV_UNCONFIGURED) {
+ path->device->flags &= ~CAM_DEV_UNCONFIGURED;
+ xpt_acquire_device(path->device);
+ done_ccb->ccb_h.func_code = XPT_GDEV_TYPE;
+ xpt_action(done_ccb);
+ xpt_async(AC_FOUND_DEVICE, path, done_ccb);
+ } else {
+ xpt_async(AC_GETDEV_CHANGED, path, NULL);
+ }
+ NVME_PROBE_SET_ACTION(softc, NVME_PROBE_DONE);
done:
if (softc->restart) {
softc->restart = false;
diff --git a/sys/dev/nvd/nvd.c b/sys/dev/nvd/nvd.c
--- a/sys/dev/nvd/nvd.c
+++ b/sys/dev/nvd/nvd.c
@@ -474,6 +474,17 @@
device_t pdev = nvd_ctrlr->ctrlr->dev;
int unit;
+ /*
+ * Zoned namespaces require the host to follow the zone write rules,
+ * which nvd(4) does not implement; exposing them as regular disks
+ * would just produce I/O errors. Use nda(4) for these instead.
+ */
+ if (nvme_ns_get_flags(ns) & NVME_NS_ZONED) {
+ device_printf(pdev, "zoned namespaces are not supported by "
+ "nvd(4), use nda(4) instead\n");
+ return (0);
+ }
+
ndisk = malloc(sizeof(struct nvd_disk), M_NVD, M_ZERO | M_WAITOK);
ndisk->ctrlr = nvd_ctrlr;
ndisk->ns = ns;
@@ -571,8 +582,9 @@
struct nvd_controller *nvd_ctrlr = device_get_softc(dev);
struct nvd_disk *ndisk = nvd_ns_to_disk(nvd_ctrlr, ns);
+ /* Nothing to tear down for a namespace that was never attached. */
if (ndisk == NULL)
- panic("nvdc: no namespace found for ns %p", ns);
+ return (0);
nvd_gone(ndisk);
/* gonecb removes it from the list -- no need to wait */
return (0);
@@ -586,8 +598,13 @@
struct disk *disk;
struct nvme_namespace *ns;
+ /*
+ * Namespaces without a disk, such as the zoned ones declined above
+ * and the inactive ones the controller skips when it announces
+ * namespaces, have nothing to resize.
+ */
if (ndisk == NULL)
- panic("nvdc: no namespace found for %d", nsid);
+ return (0);
disk = ndisk->disk;
ns = ndisk->ns;
diff --git a/sys/dev/nvme/nvme.h b/sys/dev/nvme/nvme.h
--- a/sys/dev/nvme/nvme.h
+++ b/sys/dev/nvme/nvme.h
@@ -103,6 +103,10 @@
#define NVME_CAP_HI_REG_CSS_MASK (0xff)
#define NVME_CAP_HI_REG_CSS_NVM_SHIFT (5)
#define NVME_CAP_HI_REG_CSS_NVM_MASK (0x1)
+#define NVME_CAP_HI_REG_CSS_IOCS_SHIFT (11)
+#define NVME_CAP_HI_REG_CSS_IOCS_MASK (0x1)
+#define NVME_CAP_HI_REG_CSS_NOIO_SHIFT (12)
+#define NVME_CAP_HI_REG_CSS_NOIO_MASK (0x1)
/* CAP.CSS command set support flags */
#define NVME_CAP_CSS_NVM (0x01)
@@ -134,6 +138,10 @@
NVMEV(NVME_CAP_HI_REG_CSS, x)
#define NVME_CAP_HI_CSS_NVM(x) \
NVMEV(NVME_CAP_HI_REG_CSS_NVM, x)
+#define NVME_CAP_HI_CSS_IOCS(x) \
+ NVMEV(NVME_CAP_HI_REG_CSS_IOCS, x)
+#define NVME_CAP_HI_CSS_NOIO(x) \
+ NVMEV(NVME_CAP_HI_REG_CSS_NOIO, x)
#define NVME_CAP_HI_BPS(x) \
NVMEV(NVME_CAP_HI_REG_BPS, x)
#define NVME_CAP_HI_CPS(x) \
@@ -1030,6 +1038,16 @@
NVME_SC_CONFLICTING_ATTRIBUTES = 0x80,
NVME_SC_INVALID_PROTECTION_INFO = 0x81,
NVME_SC_ATTEMPTED_WRITE_TO_RO_PAGE = 0x82,
+
+ /* Zoned Namespace Command Set */
+ NVME_SC_ZONE_BOUNDARY_ERROR = 0xb8,
+ NVME_SC_ZONE_IS_FULL = 0xb9,
+ NVME_SC_ZONE_IS_READONLY = 0xba,
+ NVME_SC_ZONE_IS_OFFLINE = 0xbb,
+ NVME_SC_ZONE_INVALID_WRITE = 0xbc,
+ NVME_SC_TOO_MANY_ACTIVE_ZONES = 0xbd,
+ NVME_SC_TOO_MANY_OPEN_ZONES = 0xbe,
+ NVME_SC_INVALID_ZONE_STATE_TRANSITION = 0xbf,
};
/* media error status codes */
@@ -1103,6 +1121,25 @@
NVME_OPC_GET_LBA_STATUS = 0x86,
};
+/* command set identifiers */
+enum nvme_csi {
+ NVME_CSI_NVM = 0x00,
+ NVME_CSI_KV = 0x01,
+ NVME_CSI_ZNS = 0x02,
+};
+
+/* identify controller or namespace structure (CNS) values */
+enum nvme_cns {
+ NVME_CNS_ID_NS = 0x00,
+ NVME_CNS_ID_CTRLR = 0x01,
+ NVME_CNS_ACTIVE_NS_LIST = 0x02,
+ NVME_CNS_ID_NS_DESC_LIST = 0x03,
+ /* 0x04 - NVM set list */
+ NVME_CNS_ID_NS_IOCS = 0x05,
+ NVME_CNS_ID_CTRLR_IOCS = 0x06,
+ NVME_CNS_ACTIVE_NS_LIST_IOCS = 0x07,
+};
+
/* nvme nvm opcodes */
enum nvme_nvm_opcode {
NVME_OPC_FLUSH = 0x00,
@@ -1124,6 +1161,11 @@
NVME_OPC_RESERVATION_RELEASE = 0x15,
/* 0x16-0x18 - reserved */
NVME_OPC_COPY = 0x19,
+
+ /* Zoned Namespace Command Set */
+ NVME_OPC_ZONE_MGMT_SEND = 0x79,
+ NVME_OPC_ZONE_MGMT_RECV = 0x7a,
+ NVME_OPC_ZONE_APPEND = 0x7d,
};
enum nvme_feature {
@@ -1715,6 +1757,254 @@
_Static_assert(sizeof(struct nvme_ns_list) == 4096, "bad size for nvme_ns_list");
+/* Namespace Identification Descriptor (CNS 03h) */
+struct nvme_ns_id_descriptor {
+ /** namespace identifier type */
+ uint8_t nidt;
+
+ /** namespace identifier length */
+ uint8_t nidl;
+
+ uint8_t reserved2[2];
+
+ /** namespace identifier */
+ uint8_t nid[];
+} __packed;
+
+enum nvme_nidt {
+ NVME_NIDT_EUI64 = 0x01,
+ NVME_NIDT_NGUID = 0x02,
+ NVME_NIDT_UUID = 0x03,
+ NVME_NIDT_CSI = 0x04,
+};
+
+/* Size of the Namespace Identification Descriptor list. */
+#define NVME_NS_ID_DESC_LIST_SIZE 4096
+
+/*
+ * Return the I/O command set a namespace is associated with, given its
+ * Namespace Identification Descriptor list. A list with no Command Set
+ * Identifier descriptor describes an NVM Command Set namespace.
+ */
+static inline uint8_t
+nvme_ns_id_desc_list_csi(const void *list, size_t len)
+{
+ const struct nvme_ns_id_descriptor *desc;
+ size_t off;
+
+ off = 0;
+ while (len - off >= sizeof(*desc)) {
+ desc = (const struct nvme_ns_id_descriptor *)
+ ((const uint8_t *)list + off);
+ off += sizeof(*desc);
+ /* A zeroed or overlong descriptor ends the list. */
+ if (desc->nidt == 0 || desc->nidl == 0 ||
+ desc->nidl > len - off)
+ break;
+ if (desc->nidt == NVME_NIDT_CSI)
+ return (desc->nid[0]);
+ off += desc->nidl;
+ }
+
+ return (NVME_CSI_NVM);
+}
+
+/*
+ * Zoned Namespace Command Set (CSI 02h) definitions.
+ */
+
+/* Zone Management Send action (cdw13 bits 7:0) */
+enum nvme_zone_send_action {
+ NVME_ZONE_SEND_CLOSE = 0x01,
+ NVME_ZONE_SEND_FINISH = 0x02,
+ NVME_ZONE_SEND_OPEN = 0x03,
+ NVME_ZONE_SEND_RESET = 0x04,
+ NVME_ZONE_SEND_OFFLINE = 0x05,
+ NVME_ZONE_SEND_SET_ZDE = 0x10,
+};
+
+/* Zone Management Send: select all zones (cdw13 bit 8) */
+#define NVME_ZONE_SEND_SELECT_ALL (1 << 8)
+
+/* Zone Management Receive action (cdw13 bits 7:0) */
+enum nvme_zone_recv_action {
+ NVME_ZONE_RECV_REPORT = 0x00,
+ NVME_ZONE_RECV_EXT_REPORT = 0x01,
+};
+
+/* Zone Management Receive: reporting options (cdw13 bits 15:8) */
+enum nvme_zone_report_option {
+ NVME_ZONE_REPORT_ALL = 0x00,
+ NVME_ZONE_REPORT_EMPTY = 0x01,
+ NVME_ZONE_REPORT_IMP_OPEN = 0x02,
+ NVME_ZONE_REPORT_EXP_OPEN = 0x03,
+ NVME_ZONE_REPORT_CLOSED = 0x04,
+ NVME_ZONE_REPORT_FULL = 0x05,
+ NVME_ZONE_REPORT_READONLY = 0x06,
+ NVME_ZONE_REPORT_OFFLINE = 0x07,
+};
+
+/* Zone Management Receive: return partial report (cdw13 bit 16) */
+#define NVME_ZONE_RECV_PARTIAL (1 << 16)
+
+enum nvme_zone_type {
+ NVME_ZONE_TYPE_SEQUENTIAL = 0x02,
+};
+
+enum nvme_zone_state {
+ NVME_ZONE_STATE_EMPTY = 0x01,
+ NVME_ZONE_STATE_IMP_OPEN = 0x02,
+ NVME_ZONE_STATE_EXP_OPEN = 0x03,
+ NVME_ZONE_STATE_CLOSED = 0x04,
+ NVME_ZONE_STATE_READONLY = 0x0d,
+ NVME_ZONE_STATE_FULL = 0x0e,
+ NVME_ZONE_STATE_OFFLINE = 0x0f,
+};
+
+struct nvme_zone_descriptor {
+ /** zone type */
+ uint8_t zt;
+#define NVME_ZONE_DESC_ZT_SHIFT (0)
+#define NVME_ZONE_DESC_ZT_MASK (0xF)
+
+ /** zone state */
+ uint8_t zs;
+#define NVME_ZONE_DESC_ZS_SHIFT (4)
+#define NVME_ZONE_DESC_ZS_MASK (0xF)
+
+ /** zone attributes */
+ uint8_t za;
+/* Zone Finished by Controller */
+#define NVME_ZONE_DESC_ZA_ZFC_SHIFT (0)
+#define NVME_ZONE_DESC_ZA_ZFC_MASK (0x1)
+/* Finish Zone Recommended */
+#define NVME_ZONE_DESC_ZA_FZR_SHIFT (1)
+#define NVME_ZONE_DESC_ZA_FZR_MASK (0x1)
+/* Reset Zone Recommended */
+#define NVME_ZONE_DESC_ZA_RZR_SHIFT (2)
+#define NVME_ZONE_DESC_ZA_RZR_MASK (0x1)
+/* ZRWA Valid */
+#define NVME_ZONE_DESC_ZA_ZRWAV_SHIFT (3)
+#define NVME_ZONE_DESC_ZA_ZRWAV_MASK (0x1)
+/* Zone Descriptor Extension Valid */
+#define NVME_ZONE_DESC_ZA_ZDEV_SHIFT (7)
+#define NVME_ZONE_DESC_ZA_ZDEV_MASK (0x1)
+
+ /** zone attributes information */
+ uint8_t zai;
+
+ /* bytes 4-7: Reserved */
+ uint8_t reserved1[4];
+
+ /** zone capacity */
+ uint64_t zcap;
+
+ /** zone start logical block address */
+ uint64_t zslba;
+
+ /** write pointer */
+ uint64_t wp;
+
+ /* bytes 32-63: Reserved */
+ uint8_t reserved2[32];
+} __packed __aligned(4);
+
+_Static_assert(sizeof(struct nvme_zone_descriptor) == 64,
+ "bad size for nvme_zone_descriptor");
+
+/* Report Zones data structure (Zone Management Receive) */
+struct nvme_zone_report {
+ /** number of zones matching the reporting options */
+ uint64_t nr_zones;
+
+ /* bytes 8-63: Reserved */
+ uint8_t reserved1[56];
+
+ struct nvme_zone_descriptor zone_desc[];
+} __packed __aligned(4);
+
+_Static_assert(sizeof(struct nvme_zone_report) == 64,
+ "bad size for nvme_zone_report");
+
+/* ZNS LBA Format Extension */
+struct nvme_zns_lbafe {
+ /** zone size (in logical blocks) */
+ uint64_t zsze;
+
+ /** zone descriptor extension size (in units of 64 bytes) */
+ uint8_t zdes;
+
+ /* bytes 9-15: Reserved */
+ uint8_t reserved1[7];
+} __packed;
+
+_Static_assert(sizeof(struct nvme_zns_lbafe) == 16,
+ "bad size for nvme_zns_lbafe");
+
+/* I/O Command Set specific Identify Namespace for ZNS (CNS 05h, CSI 02h) */
+struct nvme_zns_namespace_data {
+ /** zone operation characteristics */
+ uint16_t zoc;
+#define NVME_ZNS_NS_DATA_ZOC_VZC_SHIFT (0)
+#define NVME_ZNS_NS_DATA_ZOC_VZC_MASK (0x1)
+#define NVME_ZNS_NS_DATA_ZOC_ZAE_SHIFT (1)
+#define NVME_ZNS_NS_DATA_ZOC_ZAE_MASK (0x1)
+
+ /** optional zoned command support */
+ uint16_t ozcs;
+#define NVME_ZNS_NS_DATA_OZCS_RAZB_SHIFT (0)
+#define NVME_ZNS_NS_DATA_OZCS_RAZB_MASK (0x1)
+#define NVME_ZNS_NS_DATA_OZCS_ZRWASUP_SHIFT (1)
+#define NVME_ZNS_NS_DATA_OZCS_ZRWASUP_MASK (0x1)
+
+ /** maximum active resources (0's based) */
+ uint32_t mar;
+
+ /** maximum open resources (0's based) */
+ uint32_t mor;
+/* Value of mar and mor for a namespace that imposes no limit. */
+#define NVME_ZNS_NS_DATA_RESOURCES_UNLIMITED (0xffffffff)
+
+ /** reset recommended limit */
+ uint32_t rrl;
+
+ /** finish recommended limit */
+ uint32_t frl;
+
+ /** reset recommended limit 1-3 */
+ uint32_t rrl1;
+ uint32_t rrl2;
+ uint32_t rrl3;
+
+ /** finish recommended limit 1-3 */
+ uint32_t frl1;
+ uint32_t frl2;
+ uint32_t frl3;
+
+ /** number of ZRWA resources */
+ uint32_t numzrwa;
+
+ /** ZRWA flush granularity */
+ uint16_t zrwafg;
+
+ /** ZRWA size */
+ uint16_t zrwasz;
+
+ /** ZRWA capability */
+ uint8_t zrwacap;
+
+ /* bytes 53-2815: Reserved */
+ uint8_t reserved1[2763];
+
+ /** zns lba format extension support */
+ struct nvme_zns_lbafe lbafe[64];
+
+ uint8_t vendor_specific[256];
+} __packed __aligned(4);
+
+_Static_assert(sizeof(struct nvme_zns_namespace_data) == 4096,
+ "bad size for nvme_zns_namespace_data");
+
struct nvme_command_effects_page {
uint32_t acs[256];
uint32_t iocs[256];
@@ -2007,6 +2297,7 @@
NVME_NS_ALIVE = 0x04,
NVME_NS_DELTA = 0x08,
NVME_NS_GONE = 0x10,
+ NVME_NS_ZONED = 0x20,
};
int nvme_ctrlr_passthrough_cmd(struct nvme_controller *ctrlr,
@@ -2149,6 +2440,34 @@
cmd->cdw11 = htole32(NVME_DSM_ATTR_DEALLOCATE);
}
+static inline
+void nvme_zns_mgmt_send_cmd(struct nvme_command *cmd, uint32_t nsid,
+ uint64_t slba, bool select_all, uint8_t zsa)
+{
+ cmd->opc = NVME_OPC_ZONE_MGMT_SEND;
+ cmd->nsid = htole32(nsid);
+ cmd->cdw10 = htole32(slba & 0xffffffffu);
+ cmd->cdw11 = htole32(slba >> 32);
+ cmd->cdw13 = htole32(zsa |
+ (select_all ? NVME_ZONE_SEND_SELECT_ALL : 0));
+}
+
+/* dxfer_len must be a non-zero multiple of four. */
+static inline
+void nvme_zns_mgmt_recv_cmd(struct nvme_command *cmd, uint32_t nsid,
+ uint64_t slba, uint32_t dxfer_len, uint8_t zra, uint8_t zrasf,
+ bool partial)
+{
+ cmd->opc = NVME_OPC_ZONE_MGMT_RECV;
+ cmd->nsid = htole32(nsid);
+ cmd->cdw10 = htole32(slba & 0xffffffffu);
+ cmd->cdw11 = htole32(slba >> 32);
+ /* Number of dwords to transfer, 0's based */
+ cmd->cdw12 = htole32(dxfer_len / 4 - 1);
+ cmd->cdw13 = htole32(zra | (zrasf << 8) |
+ (partial ? NVME_ZONE_RECV_PARTIAL : 0));
+}
+
extern int nvme_use_nvd;
#endif /* _KERNEL */
@@ -2352,6 +2671,59 @@
#endif
}
+static inline
+void nvme_zns_namespace_data_swapbytes(
+ struct nvme_zns_namespace_data *s __unused)
+{
+#if _BYTE_ORDER != _LITTLE_ENDIAN
+ s->zoc = le16toh(s->zoc);
+ s->ozcs = le16toh(s->ozcs);
+ s->mar = le32toh(s->mar);
+ s->mor = le32toh(s->mor);
+ s->rrl = le32toh(s->rrl);
+ s->frl = le32toh(s->frl);
+ s->rrl1 = le32toh(s->rrl1);
+ s->rrl2 = le32toh(s->rrl2);
+ s->rrl3 = le32toh(s->rrl3);
+ s->frl1 = le32toh(s->frl1);
+ s->frl2 = le32toh(s->frl2);
+ s->frl3 = le32toh(s->frl3);
+ s->numzrwa = le32toh(s->numzrwa);
+ s->zrwafg = le16toh(s->zrwafg);
+ s->zrwasz = le16toh(s->zrwasz);
+ for (unsigned int i = 0; i < nitems(s->lbafe); i++)
+ s->lbafe[i].zsze = le64toh(s->lbafe[i].zsze);
+#endif
+}
+
+
+static inline
+void nvme_zone_descriptor_swapbytes(struct nvme_zone_descriptor *s __unused)
+{
+#if _BYTE_ORDER != _LITTLE_ENDIAN
+ s->zcap = le64toh(s->zcap);
+ s->zslba = le64toh(s->zslba);
+ s->wp = le64toh(s->wp);
+#endif
+}
+
+/*
+ * Swap the report header and the first num_zones zone descriptors, which
+ * must all be within the buffer the report was read into.
+ */
+static inline
+void nvme_zone_report_swapbytes(struct nvme_zone_report *s __unused,
+ uint64_t num_zones __unused)
+{
+#if _BYTE_ORDER != _LITTLE_ENDIAN
+ uint64_t i;
+
+ s->nr_zones = le64toh(s->nr_zones);
+ for (i = 0; i < num_zones; i++)
+ nvme_zone_descriptor_swapbytes(&s->zone_desc[i]);
+#endif
+}
+
static inline
void nvme_command_effects_page_swapbytes(
struct nvme_command_effects_page *s __unused)
diff --git a/sys/dev/nvme/nvme_ctrlr.c b/sys/dev/nvme/nvme_ctrlr.c
--- a/sys/dev/nvme/nvme_ctrlr.c
+++ b/sys/dev/nvme/nvme_ctrlr.c
@@ -400,16 +400,20 @@
/* Initialization values for CC */
cc = 0;
cc |= NVMEF(NVME_CC_REG_EN, 1);
- /* No CSI support; prefer the NVM command set when present. */
+ /*
+ * Select all supported I/O command sets when the controller
+ * implements them (CAP.CSS bit 6) so that namespaces using a
+ * command set other than NVM (e.g. ZNS) are active. Otherwise
+ * fall back to the NVM command set, or to no I/O command set at
+ * all for administrative controllers.
+ */
css = NVME_CAP_HI_CSS(ctrlr->cap_hi);
- if ((css & NVME_CAP_CSS_NVM) != 0)
+ if ((css & NVME_CAP_CSS_IOCSS) != 0)
+ cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_IOCSS);
+ else if ((css & NVME_CAP_CSS_NVM) != 0)
cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_NVM);
else if ((css & NVME_CAP_CSS_NOIOCSS) != 0)
cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_ADMIN);
- else if ((css & NVME_CAP_CSS_IOCSS) != 0)
- cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_IOCSS);
- else
- cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_NVM);
cc |= NVMEF(NVME_CC_REG_AMS, 0);
cc |= NVMEF(NVME_CC_REG_SHN, 0);
cc |= NVMEF(NVME_CC_REG_IOSQES, ctrlr->io_sqes);
diff --git a/sys/dev/nvme/nvme_ctrlr_cmd.c b/sys/dev/nvme/nvme_ctrlr_cmd.c
--- a/sys/dev/nvme/nvme_ctrlr_cmd.c
+++ b/sys/dev/nvme/nvme_ctrlr_cmd.c
@@ -30,46 +30,41 @@
#include "nvme_private.h"
void
-nvme_ctrlr_cmd_identify_controller(struct nvme_controller *ctrlr, void *payload,
- nvme_cb_fn_t cb_fn, void *cb_arg)
+nvme_ctrlr_cmd_identify(struct nvme_controller *ctrlr, uint8_t cns,
+ uint16_t cntid, uint32_t nsid, uint8_t csi, void *payload,
+ uint32_t payload_size, nvme_cb_fn_t cb_fn, void *cb_arg)
{
struct nvme_request *req;
struct nvme_command *cmd;
- req = nvme_allocate_request_vaddr(payload,
- sizeof(struct nvme_controller_data), M_WAITOK, cb_fn, cb_arg);
+ req = nvme_allocate_request_vaddr(payload, payload_size, M_WAITOK,
+ cb_fn, cb_arg);
cmd = &req->cmd;
cmd->opc = NVME_OPC_IDENTIFY;
-
- /*
- * TODO: create an identify command data structure, which
- * includes this CNS bit in cdw10.
- */
- cmd->cdw10 = htole32(1);
+ cmd->nsid = htole32(nsid);
+ cmd->cdw10 = htole32((uint32_t)cntid << 16 | cns);
+ cmd->cdw11 = htole32((uint32_t)csi << 24);
nvme_ctrlr_submit_admin_request(ctrlr, req);
}
void
-nvme_ctrlr_cmd_identify_namespace(struct nvme_controller *ctrlr, uint32_t nsid,
- void *payload, nvme_cb_fn_t cb_fn, void *cb_arg)
+nvme_ctrlr_cmd_identify_controller(struct nvme_controller *ctrlr, void *payload,
+ nvme_cb_fn_t cb_fn, void *cb_arg)
{
- struct nvme_request *req;
- struct nvme_command *cmd;
- req = nvme_allocate_request_vaddr(payload,
- sizeof(struct nvme_namespace_data), M_WAITOK, cb_fn, cb_arg);
-
- cmd = &req->cmd;
- cmd->opc = NVME_OPC_IDENTIFY;
+ nvme_ctrlr_cmd_identify(ctrlr, NVME_CNS_ID_CTRLR, 0, 0, 0, payload,
+ sizeof(struct nvme_controller_data), cb_fn, cb_arg);
+}
- /*
- * TODO: create an identify command data structure
- */
- cmd->nsid = htole32(nsid);
+void
+nvme_ctrlr_cmd_identify_namespace(struct nvme_controller *ctrlr, uint32_t nsid,
+ void *payload, nvme_cb_fn_t cb_fn, void *cb_arg)
+{
- nvme_ctrlr_submit_admin_request(ctrlr, req);
+ nvme_ctrlr_cmd_identify(ctrlr, NVME_CNS_ID_NS, 0, nsid, 0, payload,
+ sizeof(struct nvme_namespace_data), cb_fn, cb_arg);
}
void
diff --git a/sys/dev/nvme/nvme_ns.c b/sys/dev/nvme/nvme_ns.c
--- a/sys/dev/nvme/nvme_ns.c
+++ b/sys/dev/nvme/nvme_ns.c
@@ -564,6 +564,36 @@
return ((ns->flags & NVME_NS_ALIVE) ? 0 : ENXIO);
}
+ /*
+ * Determine which I/O command set the namespace is associated with.
+ * Namespaces using a command set other than NVM (e.g. ZNS)
+ * can only be active when the controller supports multiple command sets
+ * (CAP.CSS bit 6) and CC.CSS selected them. The CSI is reported in the
+ * Namespace Identification Descriptor list (CNS 03h); any failure to
+ * retrieve it is treated as the NVM command set.
+ */
+ ns->csi = NVME_CSI_NVM;
+ if (NVME_CAP_HI_CSS_IOCS(ctrlr->cap_hi) &&
+ (ctrlr->max_identify_cns == 0 ||
+ ctrlr->max_identify_cns >= NVME_CNS_ID_NS_DESC_LIST)) {
+ uint8_t *desclist;
+
+ desclist = malloc(NVME_NS_ID_DESC_LIST_SIZE, M_NVME,
+ M_WAITOK | M_ZERO);
+ status.done = 0;
+ nvme_ctrlr_cmd_identify(ctrlr, NVME_CNS_ID_NS_DESC_LIST, 0, id,
+ 0, desclist, NVME_NS_ID_DESC_LIST_SIZE,
+ nvme_completion_poll_cb, &status);
+ nvme_completion_poll(&status);
+ if (!nvme_completion_is_error(&status.cpl))
+ ns->csi = nvme_ns_id_desc_list_csi(desclist,
+ NVME_NS_ID_DESC_LIST_SIZE);
+ free(desclist, M_NVME);
+ }
+ ns->flags &= ~NVME_NS_ZONED;
+ if (ns->csi == NVME_CSI_ZNS)
+ ns->flags |= NVME_NS_ZONED;
+
/*
* Check the validity of the format specified. Note: format is a 0-based
* value, so > is appropriate here, not >=.
diff --git a/sys/dev/nvme/nvme_private.h b/sys/dev/nvme/nvme_private.h
--- a/sys/dev/nvme/nvme_private.h
+++ b/sys/dev/nvme/nvme_private.h
@@ -216,6 +216,7 @@
uint32_t flags;
struct cdev *cdev;
uint32_t boundary;
+ uint8_t csi;
struct mtx lock;
};
@@ -401,6 +402,10 @@
void nvme_ctrlr_cmd_identify_namespace(struct nvme_controller *ctrlr,
uint32_t nsid, void *payload,
nvme_cb_fn_t cb_fn, void *cb_arg);
+void nvme_ctrlr_cmd_identify(struct nvme_controller *ctrlr, uint8_t cns,
+ uint16_t cntid, uint32_t nsid, uint8_t csi,
+ void *payload, uint32_t payload_size,
+ nvme_cb_fn_t cb_fn, void *cb_arg);
void nvme_ctrlr_cmd_set_interrupt_coalescing(struct nvme_controller *ctrlr,
uint32_t microseconds,
uint32_t threshold,
diff --git a/sys/sys/disk_zone.h b/sys/sys/disk_zone.h
--- a/sys/sys/disk_zone.h
+++ b/sys/sys/disk_zone.h
@@ -151,8 +151,15 @@
(cond) == DISK_ZONE_COND_READONLY || \
(cond) == DISK_ZONE_COND_FULL || \
(cond) == DISK_ZONE_COND_OFFLINE)
+ /*
+ * Number of usable logical blocks in the zone, for devices (e.g.
+ * NVMe Zoned Namespaces) where this may be smaller than the zone
+ * length. Zero means the capacity was not reported; assume it is
+ * equal to the zone length.
+ */
+ uint64_t zone_capacity;
/* XXX KDM padding space may not be a good idea inside the bio */
- uint8_t reserved[32];
+ uint8_t reserved[24];
};
struct disk_zone_report {

File Metadata

Mime Type
text/plain
Expires
Sun, Oct 11, 4:22 PM (19 h, 55 m)
Storage Engine
blob
Storage Format
Raw Data
Storage Handle
40590893
Default Alt Text
D60414.diff (41 KB)

Event Timeline