Page Menu
Home
FreeBSD
Search
Configure Global Search
Log In
Files
F175410828
D60414.diff
No One
Temporary
Actions
View File
Edit File
Delete File
View Transforms
Subscribe
Mute Notifications
Flag For Later
Award Token
Size
41 KB
Referenced Files
None
Subscribers
None
D60414.diff
View Options
diff --git a/share/man/man4/nda.4 b/share/man/man4/nda.4
--- a/share/man/man4/nda.4
+++ b/share/man/man4/nda.4
@@ -25,7 +25,7 @@
.\" OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
.\" SUCH DAMAGE.
.\"
-.Dd October 2, 2025
+.Dd August 9, 2026
.Dt NDA 4
.Os
.Sh NAME
@@ -37,8 +37,7 @@
.Sh DESCRIPTION
The
.Nm
-driver provides support for direct access devices, implementing the
-.Tn NVMe
+driver provides support for direct access devices, implementing the NVMe
command protocol, that are attached to the system through a host adapter
supported by the CAM subsystem.
.Sh HARDWARE
@@ -165,7 +164,36 @@
Total number of
.Va BIO_DELETE
requests queued to the device.
+.It Va kern.cam.nda.N.zone_mode
+The zone mode of the namespace, either
+.Dq Host Managed
+for Zoned Namespaces or
+.Dq Not Zoned .
+.It Va kern.cam.nda.N.zone_size
+The size of each zone in logical blocks, for zoned namespaces.
+.It Va kern.cam.nda.N.max_open_zones
+The maximum number of zones that may be open at once, for zoned
+namespaces.
+Zero means the device reports no limit.
+.It Va kern.cam.nda.N.max_active_zones
+The maximum number of zones that may be active
+.Pq open or closed
+at once, for zoned namespaces.
+Zero means the device reports no limit.
+.It Va kern.cam.nda.N.read_across_zones
+Whether a single read may span more than one zone, for zoned namespaces.
.El
+.Sh ZONED NAMESPACES
+Namespaces implementing the NVMe Zoned Namespace
+.Pq ZNS
+command set are exposed as host managed zoned block devices.
+Zones can be analyzed and managed through
+.Xr zonectl 8 ,
+or programmatically via the
+.Va DIOCZONECMD
+.Xr ioctl 2 .
+Writes within a zone must be sequential, starting at the zone's write
+pointer.
.Sh NAMESPACE MAPPING
Each
.Xr nvme 4
@@ -205,7 +233,8 @@
.Xr geom 4 ,
.Xr nvd 4 ,
.Xr nvme 4 ,
-.Xr gpart 8
+.Xr gpart 8 ,
+.Xr zonectl 8
.Sh HISTORY
The
.Nm
diff --git a/sys/cam/cam_xpt.c b/sys/cam/cam_xpt.c
--- a/sys/cam/cam_xpt.c
+++ b/sys/cam/cam_xpt.c
@@ -4906,6 +4906,7 @@
free(device->serial_num, M_CAMXPT);
free(device->nvme_data, M_CAMXPT);
free(device->nvme_cdata, M_CAMXPT);
+ free(device->nvme_zns_data, M_CAMXPT);
taskqueue_enqueue(xsoftc.xpt_taskq, &device->device_destroy_task);
}
diff --git a/sys/cam/cam_xpt_internal.h b/sys/cam/cam_xpt_internal.h
--- a/sys/cam/cam_xpt_internal.h
+++ b/sys/cam/cam_xpt_internal.h
@@ -151,6 +151,7 @@
struct task device_destroy_task;
struct nvme_controller_data *nvme_cdata;
struct nvme_namespace_data *nvme_data;
+ struct nvme_zns_namespace_data *nvme_zns_data;
};
/*
diff --git a/sys/cam/nvme/nvme_all.h b/sys/cam/nvme/nvme_all.h
--- a/sys/cam/nvme/nvme_all.h
+++ b/sys/cam/nvme/nvme_all.h
@@ -48,5 +48,6 @@
int nvme_status_sbuf(struct ccb_nvmeio *nvmeio, struct sbuf *sb);
const void *nvme_get_identify_cntrl(struct cam_periph *);
const void *nvme_get_identify_ns(struct cam_periph *);
+const void *nvme_get_identify_ns_zns(struct cam_periph *);
#endif /* CAM_NVME_NVME_ALL_H */
diff --git a/sys/cam/nvme/nvme_all.c b/sys/cam/nvme/nvme_all.c
--- a/sys/cam/nvme/nvme_all.c
+++ b/sys/cam/nvme/nvme_all.c
@@ -190,4 +190,14 @@
return device->nvme_data;
}
+
+const void *
+nvme_get_identify_ns_zns(struct cam_periph *periph)
+{
+ struct cam_ed *device;
+
+ device = periph->path->device;
+
+ return device->nvme_zns_data;
+}
#endif
diff --git a/sys/cam/nvme/nvme_da.c b/sys/cam/nvme/nvme_da.c
--- a/sys/cam/nvme/nvme_da.c
+++ b/sys/cam/nvme/nvme_da.c
@@ -103,6 +103,11 @@
NDA_CCB_TYPE_MASK = 0x0F,
} nda_ccb_state;
+typedef enum {
+ NDA_ZONE_NONE = 0x00,
+ NDA_ZONE_HOST_MANAGED = 0x01,
+} nda_zone_mode;
+
/* Offsets into our private area for storing information */
#define ccb_state ccb_h.ppriv_field0
#define ccb_bp ccb_h.ppriv_ptr1 /* For NDA_CCB_BUFFER_IO */
@@ -125,6 +130,11 @@
uint64_t trim_count;
uint64_t trim_ranges;
uint64_t trim_lbas;
+ nda_zone_mode zone_mode;
+ uint64_t zone_size; /* Zone size in LBAs */
+ uint64_t max_open_zones; /* 0 == no limit */
+ uint64_t max_active_zones; /* 0 == no limit */
+ bool read_across_zones; /* OZCS.RAZB */
#ifdef CAM_TEST_FAILURE
int force_read_error;
int force_write_error;
@@ -157,6 +167,7 @@
struct cam_path *path, void *arg);
static void ndasysctlinit(void *context, int pending);
static int ndaflagssysctl(SYSCTL_HANDLER_ARGS);
+static int ndazonemodesysctl(SYSCTL_HANDLER_ARGS);
static periph_ctor_t ndaregister;
static periph_dtor_t ndacleanup;
static periph_start_t ndastart;
@@ -291,17 +302,322 @@
nvme_ns_rw_cmd(&nvmeio->cmd, rwcmd, softc->nsid, lba, count);
}
+static int
+nda_zone_bio_to_nvme(int disk_zone_cmd)
+{
+ switch (disk_zone_cmd) {
+ case DISK_ZONE_OPEN:
+ return (NVME_ZONE_SEND_OPEN);
+ case DISK_ZONE_CLOSE:
+ return (NVME_ZONE_SEND_CLOSE);
+ case DISK_ZONE_FINISH:
+ return (NVME_ZONE_SEND_FINISH);
+ case DISK_ZONE_RWP:
+ return (NVME_ZONE_SEND_RESET);
+ }
+
+ return (-1);
+}
+
+static int
+nda_zone_rep_to_nvme(uint8_t rep_options)
+{
+ switch (rep_options) {
+ case DISK_ZONE_REP_ALL:
+ return (NVME_ZONE_REPORT_ALL);
+ case DISK_ZONE_REP_EMPTY:
+ return (NVME_ZONE_REPORT_EMPTY);
+ case DISK_ZONE_REP_IMP_OPEN:
+ return (NVME_ZONE_REPORT_IMP_OPEN);
+ case DISK_ZONE_REP_EXP_OPEN:
+ return (NVME_ZONE_REPORT_EXP_OPEN);
+ case DISK_ZONE_REP_CLOSED:
+ return (NVME_ZONE_REPORT_CLOSED);
+ case DISK_ZONE_REP_FULL:
+ return (NVME_ZONE_REPORT_FULL);
+ case DISK_ZONE_REP_READONLY:
+ return (NVME_ZONE_REPORT_READONLY);
+ case DISK_ZONE_REP_OFFLINE:
+ return (NVME_ZONE_REPORT_OFFLINE);
+ }
+
+ return (-1);
+}
+
+static int
+nda_zone_cmd(struct cam_periph *periph, union ccb *ccb, struct bio *bp,
+ int *queue_ccb)
+{
+ struct nda_softc *softc;
+ int error;
+
+ error = 0;
+
+ if (bp->bio_cmd != BIO_ZONE) {
+ error = EINVAL;
+ goto bailout;
+ }
+
+ softc = periph->softc;
+
+ switch (bp->bio_zone.zone_cmd) {
+ case DISK_ZONE_OPEN:
+ case DISK_ZONE_CLOSE:
+ case DISK_ZONE_FINISH:
+ case DISK_ZONE_RWP: {
+ int send_action;
+ bool select_all;
+
+ send_action = nda_zone_bio_to_nvme(bp->bio_zone.zone_cmd);
+ if (send_action == -1) {
+ xpt_print(periph->path, "Cannot translate zone "
+ "cmd %#x to NVMe\n", bp->bio_zone.zone_cmd);
+ error = EINVAL;
+ goto bailout;
+ }
+
+ select_all = (bp->bio_zone.zone_params.rwp.flags &
+ DISK_ZONE_RWP_FLAG_ALL) != 0;
+
+ cam_fill_nvmeio(&ccb->nvmeio,
+ 0, /* retries */
+ ndadone, /* cbfcnp */
+ CAM_DIR_NONE, /* flags */
+ NULL, /* data_ptr */
+ 0, /* dxfer_len */
+ nda_default_timeout * 1000); /* timeout 30s */
+ nvme_zns_mgmt_send_cmd(&ccb->nvmeio.cmd, softc->nsid,
+ bp->bio_zone.zone_params.rwp.id, select_all,
+ send_action);
+ *queue_ccb = 1;
+
+ break;
+ }
+ case DISK_ZONE_REPORT_ZONES: {
+ uint8_t *rz_ptr;
+ uint32_t num_entries, max_entries, alloc_size;
+ struct disk_zone_report *rep;
+ int resp_option;
+
+ rep = &bp->bio_zone.zone_params.report;
+
+ num_entries = rep->entries_allocated;
+ if (num_entries == 0) {
+ xpt_print(periph->path, "No entries allocated for "
+ "Report Zones request\n");
+ error = EINVAL;
+ goto bailout;
+ }
+ resp_option = nda_zone_rep_to_nvme(rep->rep_options);
+ if (resp_option == -1) {
+ xpt_print(periph->path, "Cannot translate zone "
+ "reporting option %#x to NVMe\n",
+ rep->rep_options);
+ error = EINVAL;
+ goto bailout;
+ }
+ /*
+ * Trim the request to what the controller transfers in a
+ * single command. This keeps room for the report header and a
+ * whole number of zone descriptors.
+ */
+ max_entries = (softc->disk->d_maxsize -
+ sizeof(struct nvme_zone_report)) /
+ sizeof(struct nvme_zone_descriptor);
+ num_entries = MIN(num_entries, max_entries);
+ alloc_size = sizeof(struct nvme_zone_report) +
+ (sizeof(struct nvme_zone_descriptor) * num_entries);
+ rz_ptr = malloc(alloc_size, M_NVMEDA, M_NOWAIT | M_ZERO);
+ if (rz_ptr == NULL) {
+ xpt_print(periph->path, "Unable to allocate memory "
+ "for Report Zones request\n");
+ error = ENOMEM;
+ goto bailout;
+ }
+
+ cam_fill_nvmeio(&ccb->nvmeio,
+ 0, /* retries */
+ ndadone, /* cbfcnp */
+ CAM_DIR_IN, /* flags */
+ rz_ptr, /* data_ptr */
+ alloc_size, /* dxfer_len */
+ nda_default_timeout * 1000); /* timeout 30s */
+ /*
+ * Ask for a *full* report so that the number of zones returned
+ * in the header is the total number of zones matching the
+ * reporting options: this is what the entries_available field
+ * wants.
+ */
+ nvme_zns_mgmt_recv_cmd(&ccb->nvmeio.cmd, softc->nsid,
+ rep->starting_id, alloc_size, NVME_ZONE_RECV_REPORT,
+ resp_option, false);
+
+ /*
+ * BIO_ZONE would not normally need this. However, this is used
+ * by devstat_end_transaction_bio() to determine how much data
+ * was transferred. Because the size of the NVMe structs is
+ * different than the size of the BIO interface structs, the
+ * amount of data that is actually transferred from the drive
+ * will be different than the amount of data transferred to the
+ * user.
+ */
+ bp->bio_bcount = bp->bio_length;
+
+ *queue_ccb = 1;
+
+ break;
+ }
+ case DISK_ZONE_GET_PARAMS: {
+ struct disk_zone_disk_params *params;
+
+ params = &bp->bio_zone.zone_params.disk_params;
+ bzero(params, sizeof(*params));
+
+ switch (softc->zone_mode) {
+ case NDA_ZONE_HOST_MANAGED:
+ params->zone_mode = DISK_ZONE_MODE_HOST_MANAGED;
+ break;
+ default:
+ case NDA_ZONE_NONE:
+ params->zone_mode = DISK_ZONE_MODE_NONE;
+ break;
+ }
+
+ /*
+ * Reads are permitted anywhere in a zone that is not in the
+ * ZSO:Offline state.
+ */
+ params->flags |= DISK_ZONE_DISK_URSWRZ;
+
+ if (softc->max_open_zones != 0) {
+ params->max_seq_zones = softc->max_open_zones;
+ params->flags |= DISK_ZONE_MAX_SEQ_SET;
+ }
+
+ params->flags |= DISK_ZONE_RZ_SUP | DISK_ZONE_OPEN_SUP |
+ DISK_ZONE_CLOSE_SUP | DISK_ZONE_FINISH_SUP |
+ DISK_ZONE_RWP_SUP;
+ break;
+ }
+ default:
+ break;
+ }
+bailout:
+ return (error);
+}
+
+static void
+ndazonedone(struct cam_periph *periph, union ccb *ccb)
+{
+ struct nda_softc *softc;
+ struct bio *bp;
+
+ softc = periph->softc;
+ bp = (struct bio *)ccb->ccb_bp;
+
+ switch (bp->bio_zone.zone_cmd) {
+ case DISK_ZONE_OPEN:
+ case DISK_ZONE_CLOSE:
+ case DISK_ZONE_FINISH:
+ case DISK_ZONE_RWP:
+ break;
+ case DISK_ZONE_REPORT_ZONES: {
+ uint32_t avail_len, max_desc;
+ struct disk_zone_report *rep;
+ struct nvme_zone_report *hdr;
+ struct nvme_zone_descriptor *desc;
+ struct disk_zone_rep_entry *entry;
+ uint64_t num_avail;
+ uint32_t num_to_fill, i;
+
+ rep = &bp->bio_zone.zone_params.report;
+ avail_len = ccb->nvmeio.dxfer_len;
+ hdr = (struct nvme_zone_report *)ccb->nvmeio.data_ptr;
+
+ /*
+ * The transfer is dxfer_len bytes and the buffer was
+ * zeroed before the command, so every descriptor slot in
+ * it is safe to byte swap, whether or not the device
+ * filled it in.
+ */
+ max_desc = (avail_len - sizeof(*hdr)) / sizeof(*desc);
+ nvme_zone_report_swapbytes(hdr, max_desc);
+
+ /*
+ * Every zone has the same length (ZSZE) and the same type,
+ * which is all the SAME field describes. Zone capacity may
+ * vary per zone, but does not affect it.
+ */
+ rep->header.same = DISK_ZONE_SAME_ALL_SAME;
+ rep->header.maximum_lba = softc->disk->d_mediasize /
+ softc->disk->d_sectorsize - 1;
+ rep->entries_available = MIN(hdr->nr_zones, UINT32_MAX);
+
+ num_avail = MIN(hdr->nr_zones, max_desc);
+ num_to_fill = MIN(num_avail, rep->entries_allocated);
+ if (num_to_fill == 0) {
+ rep->entries_filled = 0;
+ bp->bio_resid = bp->bio_bcount;
+ break;
+ }
+
+ for (i = 0, desc = &hdr->zone_desc[0], entry = &rep->entries[0];
+ i < num_to_fill; i++, desc++, entry++) {
+ if (NVMEV(NVME_ZONE_DESC_ZT, desc->zt) ==
+ NVME_ZONE_TYPE_SEQUENTIAL)
+ entry->zone_type = DISK_ZONE_TYPE_SEQ_REQUIRED;
+ else
+ entry->zone_type =
+ NVMEV(NVME_ZONE_DESC_ZT, desc->zt);
+ entry->zone_condition =
+ NVMEV(NVME_ZONE_DESC_ZS, desc->zs);
+ entry->zone_flags = 0;
+ if (NVMEV(NVME_ZONE_DESC_ZA_RZR, desc->za))
+ entry->zone_flags |= DISK_ZONE_FLAG_RESET;
+ entry->zone_length = softc->zone_size;
+ entry->zone_capacity = desc->zcap;
+ entry->zone_start_lba = desc->zslba;
+ entry->write_pointer_lba = desc->wp;
+ }
+ rep->entries_filled = num_to_fill;
+ /*
+ * Note that this residual is accurate from the user's
+ * standpoint, but the amount transferred isn't accurate
+ * from the standpoint of what actually came back from the
+ * drive.
+ */
+ bp->bio_resid = bp->bio_bcount - (num_to_fill * sizeof(*entry));
+ break;
+ }
+ case DISK_ZONE_GET_PARAMS:
+ default:
+ /*
+ * In theory we should not get a GET_PARAMS bio, since it
+ * should be handled without queueing the command to the
+ * drive.
+ */
+ panic("%s: Invalid zone command %d", __func__,
+ bp->bio_zone.zone_cmd);
+ break;
+ }
+
+ if (bp->bio_zone.zone_cmd == DISK_ZONE_REPORT_ZONES)
+ free(ccb->nvmeio.data_ptr, M_NVMEDA);
+}
+
static void
ndasetgeom(struct nda_softc *softc, struct cam_periph *periph)
{
struct disk *disk = softc->disk;
const struct nvme_namespace_data *nsd;
const struct nvme_controller_data *cd;
+ const struct nvme_zns_namespace_data *znsd;
uint8_t flbas_fmt, lbads, vwc_present;
u_int flags;
nsd = nvme_get_identify_ns(periph);
cd = nvme_get_identify_cntrl(periph);
+ znsd = nvme_get_identify_ns_zns(periph);
/*
* Preserve flags we can't infer that were set before. UNMAPPED comes
@@ -322,6 +638,26 @@
if (vwc_present)
disk->d_flags |= DISKFLAG_CANFLUSHCACHE;
disk->d_flags |= flags;
+
+ if (znsd != NULL) {
+ /*
+ * Zoned namespaces are always host managed.
+ */
+ softc->zone_mode = NDA_ZONE_HOST_MANAGED;
+ softc->zone_size = znsd->lbafe[flbas_fmt].zsze;
+ /* MAR and MOR are 0's based. */
+ softc->max_active_zones =
+ znsd->mar == NVME_ZNS_NS_DATA_RESOURCES_UNLIMITED ? 0 :
+ (uint64_t)znsd->mar + 1;
+ softc->max_open_zones =
+ znsd->mor == NVME_ZNS_NS_DATA_RESOURCES_UNLIMITED ? 0 :
+ (uint64_t)znsd->mor + 1;
+ softc->read_across_zones =
+ NVMEV(NVME_ZNS_NS_DATA_OZCS_RAZB, znsd->ozcs) != 0;
+ disk->d_flags |= DISKFLAG_CANZONE;
+ } else {
+ softc->zone_mode = NDA_ZONE_NONE;
+ }
}
static void
@@ -565,6 +901,13 @@
if (bp->bio_cmd == BIO_DELETE)
softc->deletes++;
+ /*
+ * Zone cmds must be ordered, as they can depend on the effects of
+ * previously issued commands, which may affect commands after them.
+ */
+ if (bp->bio_cmd == BIO_ZONE)
+ bp->bio_flags |= BIO_ORDERED;
+
/*
* Place it in the queue of disk activities for this disk
*/
@@ -869,6 +1212,27 @@
softc, 0, ndaflagssysctl, "A",
"Flags for drive");
+ SYSCTL_ADD_PROC(&softc->sysctl_ctx, SYSCTL_CHILDREN(softc->sysctl_tree),
+ OID_AUTO, "zone_mode", CTLTYPE_STRING | CTLFLAG_RD | CTLFLAG_MPSAFE,
+ softc, 0, ndazonemodesysctl, "A",
+ "Zone Mode");
+ SYSCTL_ADD_UQUAD(&softc->sysctl_ctx,
+ SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO,
+ "zone_size", CTLFLAG_RD, &softc->zone_size,
+ "Zone size in LBAs");
+ SYSCTL_ADD_UQUAD(&softc->sysctl_ctx,
+ SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO,
+ "max_open_zones", CTLFLAG_RD, &softc->max_open_zones,
+ "Maximum number of open zones (0 = no limit)");
+ SYSCTL_ADD_UQUAD(&softc->sysctl_ctx,
+ SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO,
+ "max_active_zones", CTLFLAG_RD, &softc->max_active_zones,
+ "Maximum number of active zones (0 = no limit)");
+ SYSCTL_ADD_BOOL(&softc->sysctl_ctx,
+ SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO,
+ "read_across_zones", CTLFLAG_RD, &softc->read_across_zones, 0,
+ "Reads may span more than one zone");
+
#ifdef CAM_IO_STATS
softc->sysctl_stats_tree = SYSCTL_ADD_NODE(&softc->sysctl_stats_ctx,
SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO, "stats",
@@ -908,6 +1272,30 @@
cam_periph_release(periph);
}
+static int
+ndazonemodesysctl(SYSCTL_HANDLER_ARGS)
+{
+ char tmpbuf[24];
+ struct nda_softc *softc;
+ int error;
+
+ softc = (struct nda_softc *)arg1;
+
+ switch (softc->zone_mode) {
+ case NDA_ZONE_HOST_MANAGED:
+ snprintf(tmpbuf, sizeof(tmpbuf), "Host Managed");
+ break;
+ case NDA_ZONE_NONE:
+ default:
+ snprintf(tmpbuf, sizeof(tmpbuf), "Not Zoned");
+ break;
+ }
+
+ error = sysctl_handle_string(oidp, tmpbuf, sizeof(tmpbuf), req);
+
+ return (error);
+}
+
static int
ndaflagssysctl(SYSCTL_HANDLER_ARGS)
{
@@ -1240,6 +1628,30 @@
case BIO_FLUSH:
nda_nvme_flush(softc, nvmeio);
break;
+ case BIO_ZONE: {
+ int error, queue_ccb;
+
+ queue_ccb = 0;
+
+ error = nda_zone_cmd(periph, start_ccb, bp, &queue_ccb);
+ if ((error != 0)
+ || (queue_ccb == 0)) {
+ /*
+ * g_io_deliver will recursively call start
+ * routine for ENOMEM... drop the periph lock
+ * to allow that recursion.
+ */
+ if (error == ENOMEM)
+ cam_periph_unlock(periph);
+ biofinish(bp, NULL, error);
+ if (error == ENOMEM)
+ cam_periph_lock(periph);
+ xpt_release_ccb(start_ccb);
+ ndaschedule(periph);
+ return;
+ }
+ break;
+ }
default:
biofinish(bp, NULL, EOPNOTSUPP);
xpt_release_ccb(start_ccb);
@@ -1311,8 +1723,14 @@
if (error != 0) {
bp->bio_resid = bp->bio_bcount;
bp->bio_flags |= BIO_ERROR;
+ if (bp->bio_cmd == BIO_ZONE &&
+ bp->bio_zone.zone_cmd ==
+ DISK_ZONE_REPORT_ZONES)
+ free(nvmeio->data_ptr, M_NVMEDA);
} else {
bp->bio_resid = 0;
+ if (bp->bio_cmd == BIO_ZONE)
+ ndazonedone(periph, done_ccb);
}
softc->outstanding_cmds--;
diff --git a/sys/cam/nvme/nvme_xpt.c b/sys/cam/nvme/nvme_xpt.c
--- a/sys/cam/nvme/nvme_xpt.c
+++ b/sys/cam/nvme/nvme_xpt.c
@@ -81,6 +81,8 @@
typedef enum {
NVME_PROBE_IDENTIFY_CD,
NVME_PROBE_IDENTIFY_NS,
+ NVME_PROBE_IDENTIFY_NS_DESCS,
+ NVME_PROBE_IDENTIFY_NS_ZNS,
NVME_PROBE_DONE,
NVME_PROBE_INVALID
} nvme_probe_action;
@@ -88,6 +90,8 @@
static char *nvme_probe_action_text[] = {
"NVME_PROBE_IDENTIFY_CD",
"NVME_PROBE_IDENTIFY_NS",
+ "NVME_PROBE_IDENTIFY_NS_DESCS",
+ "NVME_PROBE_IDENTIFY_NS_ZNS",
"NVME_PROBE_DONE",
"NVME_PROBE_INVALID"
};
@@ -111,6 +115,8 @@
union {
struct nvme_controller_data cd;
struct nvme_namespace_data ns;
+ uint8_t nsdescs[NVME_NS_ID_DESC_LIST_SIZE];
+ struct nvme_zns_namespace_data zns;
};
nvme_probe_action action;
nvme_probe_flags flags;
@@ -294,6 +300,29 @@
nvme_ns_cmd(nvmeio, NVME_OPC_IDENTIFY, lun,
0, 0, 0, 0, 0, 0);
break;
+ case NVME_PROBE_IDENTIFY_NS_DESCS:
+ cam_fill_nvmeadmin(nvmeio,
+ 0, /* retries */
+ nvme_probe_done, /* cbfcnp */
+ CAM_DIR_IN, /* flags */
+ (uint8_t *)&softc->nsdescs, /* data_ptr */
+ sizeof(softc->nsdescs), /* dxfer_len */
+ 30 * 1000); /* timeout 30s */
+ nvme_ns_cmd(nvmeio, NVME_OPC_IDENTIFY, lun,
+ NVME_CNS_ID_NS_DESC_LIST, 0, 0, 0, 0, 0);
+ break;
+ case NVME_PROBE_IDENTIFY_NS_ZNS:
+ cam_fill_nvmeadmin(nvmeio,
+ 0, /* retries */
+ nvme_probe_done, /* cbfcnp */
+ CAM_DIR_IN, /* flags */
+ (uint8_t *)&softc->zns, /* data_ptr */
+ sizeof(softc->zns), /* dxfer_len */
+ 30 * 1000); /* timeout 30s */
+ nvme_ns_cmd(nvmeio, NVME_OPC_IDENTIFY, lun,
+ NVME_CNS_ID_NS_IOCS, (uint32_t)NVME_CSI_ZNS << 24,
+ 0, 0, 0, 0);
+ break;
default:
panic("nvme_probe_start: invalid action state 0x%x\n", softc->action);
}
@@ -320,6 +349,20 @@
priority = done_ccb->ccb_h.pinfo.priority;
if ((done_ccb->ccb_h.status & CAM_STATUS_MASK) != CAM_REQ_CMP) {
+ /*
+ * Older controllers may not implement the namespace
+ * indentification descriptor list and the I/O command set
+ * specific identify data, with some SIMs rejecting the higher
+ * CNS values. Announce the device without that data instead of
+ * failing the probe.
+ */
+ if (softc->action == NVME_PROBE_IDENTIFY_NS_DESCS ||
+ softc->action == NVME_PROBE_IDENTIFY_NS_ZNS) {
+ if ((done_ccb->ccb_h.status & CAM_DEV_QFRZN) != 0)
+ xpt_release_devq(path, /*count*/1,
+ /*run_queue*/TRUE);
+ goto announce;
+ }
if (cam_periph_error(done_ccb,
0, softc->restart ? (SF_NO_RECOVERY | SF_NO_RETRY) : 0
) == ERESTART) {
@@ -348,7 +391,7 @@
NVME_PROBE_SET_ACTION(softc, NVME_PROBE_INVALID);
found = 0;
goto done;
- }
+}
if (softc->restart)
goto done;
switch (softc->action) {
@@ -457,20 +500,57 @@
path->device->device_id_len = SVPD_DEVICE_ID_HDR_LEN + len;
}
- if (periph->path->device->flags & CAM_DEV_UNCONFIGURED) {
- path->device->flags &= ~CAM_DEV_UNCONFIGURED;
- xpt_acquire_device(path->device);
- done_ccb->ccb_h.func_code = XPT_GDEV_TYPE;
- xpt_action(done_ccb);
- xpt_async(AC_FOUND_DEVICE, path, done_ccb);
- } else {
- xpt_async(AC_GETDEV_CHANGED, path, NULL);
+ NVME_PROBE_SET_ACTION(softc, NVME_PROBE_IDENTIFY_NS_DESCS);
+ xpt_release_ccb(done_ccb);
+ xpt_schedule(periph, priority);
+ goto out;
+ case NVME_PROBE_IDENTIFY_NS_DESCS: {
+ uint8_t csi;
+
+ csi = nvme_ns_id_desc_list_csi(softc->nsdescs,
+ sizeof(softc->nsdescs));
+ if (csi == NVME_CSI_ZNS) {
+ NVME_PROBE_SET_ACTION(softc, NVME_PROBE_IDENTIFY_NS_ZNS);
+ xpt_release_ccb(done_ccb);
+ xpt_schedule(periph, priority);
+ goto out;
}
- NVME_PROBE_SET_ACTION(softc, NVME_PROBE_DONE);
+ free(path->device->nvme_zns_data, M_CAMXPT);
+ path->device->nvme_zns_data = NULL;
break;
+ }
+ case NVME_PROBE_IDENTIFY_NS_ZNS: {
+ struct nvme_zns_namespace_data *zns_data;
+
+ nvme_zns_namespace_data_swapbytes(&softc->zns);
+
+ zns_data = path->device->nvme_zns_data;
+ if (zns_data == NULL) {
+ zns_data = malloc(sizeof(*zns_data), M_CAMXPT,
+ M_NOWAIT);
+ if (zns_data == NULL) {
+ xpt_print(path, "Can't allocate memory");
+ goto device_fail;
+ }
+ }
+ bcopy(&softc->zns, zns_data, sizeof(*zns_data));
+ path->device->nvme_zns_data = zns_data;
+ break;
+ }
default:
panic("nvme_probe_done: invalid action state 0x%x\n", softc->action);
}
+announce:
+ if (periph->path->device->flags & CAM_DEV_UNCONFIGURED) {
+ path->device->flags &= ~CAM_DEV_UNCONFIGURED;
+ xpt_acquire_device(path->device);
+ done_ccb->ccb_h.func_code = XPT_GDEV_TYPE;
+ xpt_action(done_ccb);
+ xpt_async(AC_FOUND_DEVICE, path, done_ccb);
+ } else {
+ xpt_async(AC_GETDEV_CHANGED, path, NULL);
+ }
+ NVME_PROBE_SET_ACTION(softc, NVME_PROBE_DONE);
done:
if (softc->restart) {
softc->restart = false;
diff --git a/sys/dev/nvd/nvd.c b/sys/dev/nvd/nvd.c
--- a/sys/dev/nvd/nvd.c
+++ b/sys/dev/nvd/nvd.c
@@ -474,6 +474,17 @@
device_t pdev = nvd_ctrlr->ctrlr->dev;
int unit;
+ /*
+ * Zoned namespaces require the host to follow the zone write rules,
+ * which nvd(4) does not implement; exposing them as regular disks
+ * would just produce I/O errors. Use nda(4) for these instead.
+ */
+ if (nvme_ns_get_flags(ns) & NVME_NS_ZONED) {
+ device_printf(pdev, "zoned namespaces are not supported by "
+ "nvd(4), use nda(4) instead\n");
+ return (0);
+ }
+
ndisk = malloc(sizeof(struct nvd_disk), M_NVD, M_ZERO | M_WAITOK);
ndisk->ctrlr = nvd_ctrlr;
ndisk->ns = ns;
@@ -571,8 +582,9 @@
struct nvd_controller *nvd_ctrlr = device_get_softc(dev);
struct nvd_disk *ndisk = nvd_ns_to_disk(nvd_ctrlr, ns);
+ /* Nothing to tear down for a namespace that was never attached. */
if (ndisk == NULL)
- panic("nvdc: no namespace found for ns %p", ns);
+ return (0);
nvd_gone(ndisk);
/* gonecb removes it from the list -- no need to wait */
return (0);
@@ -586,8 +598,13 @@
struct disk *disk;
struct nvme_namespace *ns;
+ /*
+ * Namespaces without a disk, such as the zoned ones declined above
+ * and the inactive ones the controller skips when it announces
+ * namespaces, have nothing to resize.
+ */
if (ndisk == NULL)
- panic("nvdc: no namespace found for %d", nsid);
+ return (0);
disk = ndisk->disk;
ns = ndisk->ns;
diff --git a/sys/dev/nvme/nvme.h b/sys/dev/nvme/nvme.h
--- a/sys/dev/nvme/nvme.h
+++ b/sys/dev/nvme/nvme.h
@@ -103,6 +103,10 @@
#define NVME_CAP_HI_REG_CSS_MASK (0xff)
#define NVME_CAP_HI_REG_CSS_NVM_SHIFT (5)
#define NVME_CAP_HI_REG_CSS_NVM_MASK (0x1)
+#define NVME_CAP_HI_REG_CSS_IOCS_SHIFT (11)
+#define NVME_CAP_HI_REG_CSS_IOCS_MASK (0x1)
+#define NVME_CAP_HI_REG_CSS_NOIO_SHIFT (12)
+#define NVME_CAP_HI_REG_CSS_NOIO_MASK (0x1)
/* CAP.CSS command set support flags */
#define NVME_CAP_CSS_NVM (0x01)
@@ -134,6 +138,10 @@
NVMEV(NVME_CAP_HI_REG_CSS, x)
#define NVME_CAP_HI_CSS_NVM(x) \
NVMEV(NVME_CAP_HI_REG_CSS_NVM, x)
+#define NVME_CAP_HI_CSS_IOCS(x) \
+ NVMEV(NVME_CAP_HI_REG_CSS_IOCS, x)
+#define NVME_CAP_HI_CSS_NOIO(x) \
+ NVMEV(NVME_CAP_HI_REG_CSS_NOIO, x)
#define NVME_CAP_HI_BPS(x) \
NVMEV(NVME_CAP_HI_REG_BPS, x)
#define NVME_CAP_HI_CPS(x) \
@@ -1030,6 +1038,16 @@
NVME_SC_CONFLICTING_ATTRIBUTES = 0x80,
NVME_SC_INVALID_PROTECTION_INFO = 0x81,
NVME_SC_ATTEMPTED_WRITE_TO_RO_PAGE = 0x82,
+
+ /* Zoned Namespace Command Set */
+ NVME_SC_ZONE_BOUNDARY_ERROR = 0xb8,
+ NVME_SC_ZONE_IS_FULL = 0xb9,
+ NVME_SC_ZONE_IS_READONLY = 0xba,
+ NVME_SC_ZONE_IS_OFFLINE = 0xbb,
+ NVME_SC_ZONE_INVALID_WRITE = 0xbc,
+ NVME_SC_TOO_MANY_ACTIVE_ZONES = 0xbd,
+ NVME_SC_TOO_MANY_OPEN_ZONES = 0xbe,
+ NVME_SC_INVALID_ZONE_STATE_TRANSITION = 0xbf,
};
/* media error status codes */
@@ -1103,6 +1121,25 @@
NVME_OPC_GET_LBA_STATUS = 0x86,
};
+/* command set identifiers */
+enum nvme_csi {
+ NVME_CSI_NVM = 0x00,
+ NVME_CSI_KV = 0x01,
+ NVME_CSI_ZNS = 0x02,
+};
+
+/* identify controller or namespace structure (CNS) values */
+enum nvme_cns {
+ NVME_CNS_ID_NS = 0x00,
+ NVME_CNS_ID_CTRLR = 0x01,
+ NVME_CNS_ACTIVE_NS_LIST = 0x02,
+ NVME_CNS_ID_NS_DESC_LIST = 0x03,
+ /* 0x04 - NVM set list */
+ NVME_CNS_ID_NS_IOCS = 0x05,
+ NVME_CNS_ID_CTRLR_IOCS = 0x06,
+ NVME_CNS_ACTIVE_NS_LIST_IOCS = 0x07,
+};
+
/* nvme nvm opcodes */
enum nvme_nvm_opcode {
NVME_OPC_FLUSH = 0x00,
@@ -1124,6 +1161,11 @@
NVME_OPC_RESERVATION_RELEASE = 0x15,
/* 0x16-0x18 - reserved */
NVME_OPC_COPY = 0x19,
+
+ /* Zoned Namespace Command Set */
+ NVME_OPC_ZONE_MGMT_SEND = 0x79,
+ NVME_OPC_ZONE_MGMT_RECV = 0x7a,
+ NVME_OPC_ZONE_APPEND = 0x7d,
};
enum nvme_feature {
@@ -1715,6 +1757,254 @@
_Static_assert(sizeof(struct nvme_ns_list) == 4096, "bad size for nvme_ns_list");
+/* Namespace Identification Descriptor (CNS 03h) */
+struct nvme_ns_id_descriptor {
+ /** namespace identifier type */
+ uint8_t nidt;
+
+ /** namespace identifier length */
+ uint8_t nidl;
+
+ uint8_t reserved2[2];
+
+ /** namespace identifier */
+ uint8_t nid[];
+} __packed;
+
+enum nvme_nidt {
+ NVME_NIDT_EUI64 = 0x01,
+ NVME_NIDT_NGUID = 0x02,
+ NVME_NIDT_UUID = 0x03,
+ NVME_NIDT_CSI = 0x04,
+};
+
+/* Size of the Namespace Identification Descriptor list. */
+#define NVME_NS_ID_DESC_LIST_SIZE 4096
+
+/*
+ * Return the I/O command set a namespace is associated with, given its
+ * Namespace Identification Descriptor list. A list with no Command Set
+ * Identifier descriptor describes an NVM Command Set namespace.
+ */
+static inline uint8_t
+nvme_ns_id_desc_list_csi(const void *list, size_t len)
+{
+ const struct nvme_ns_id_descriptor *desc;
+ size_t off;
+
+ off = 0;
+ while (len - off >= sizeof(*desc)) {
+ desc = (const struct nvme_ns_id_descriptor *)
+ ((const uint8_t *)list + off);
+ off += sizeof(*desc);
+ /* A zeroed or overlong descriptor ends the list. */
+ if (desc->nidt == 0 || desc->nidl == 0 ||
+ desc->nidl > len - off)
+ break;
+ if (desc->nidt == NVME_NIDT_CSI)
+ return (desc->nid[0]);
+ off += desc->nidl;
+ }
+
+ return (NVME_CSI_NVM);
+}
+
+/*
+ * Zoned Namespace Command Set (CSI 02h) definitions.
+ */
+
+/* Zone Management Send action (cdw13 bits 7:0) */
+enum nvme_zone_send_action {
+ NVME_ZONE_SEND_CLOSE = 0x01,
+ NVME_ZONE_SEND_FINISH = 0x02,
+ NVME_ZONE_SEND_OPEN = 0x03,
+ NVME_ZONE_SEND_RESET = 0x04,
+ NVME_ZONE_SEND_OFFLINE = 0x05,
+ NVME_ZONE_SEND_SET_ZDE = 0x10,
+};
+
+/* Zone Management Send: select all zones (cdw13 bit 8) */
+#define NVME_ZONE_SEND_SELECT_ALL (1 << 8)
+
+/* Zone Management Receive action (cdw13 bits 7:0) */
+enum nvme_zone_recv_action {
+ NVME_ZONE_RECV_REPORT = 0x00,
+ NVME_ZONE_RECV_EXT_REPORT = 0x01,
+};
+
+/* Zone Management Receive: reporting options (cdw13 bits 15:8) */
+enum nvme_zone_report_option {
+ NVME_ZONE_REPORT_ALL = 0x00,
+ NVME_ZONE_REPORT_EMPTY = 0x01,
+ NVME_ZONE_REPORT_IMP_OPEN = 0x02,
+ NVME_ZONE_REPORT_EXP_OPEN = 0x03,
+ NVME_ZONE_REPORT_CLOSED = 0x04,
+ NVME_ZONE_REPORT_FULL = 0x05,
+ NVME_ZONE_REPORT_READONLY = 0x06,
+ NVME_ZONE_REPORT_OFFLINE = 0x07,
+};
+
+/* Zone Management Receive: return partial report (cdw13 bit 16) */
+#define NVME_ZONE_RECV_PARTIAL (1 << 16)
+
+enum nvme_zone_type {
+ NVME_ZONE_TYPE_SEQUENTIAL = 0x02,
+};
+
+enum nvme_zone_state {
+ NVME_ZONE_STATE_EMPTY = 0x01,
+ NVME_ZONE_STATE_IMP_OPEN = 0x02,
+ NVME_ZONE_STATE_EXP_OPEN = 0x03,
+ NVME_ZONE_STATE_CLOSED = 0x04,
+ NVME_ZONE_STATE_READONLY = 0x0d,
+ NVME_ZONE_STATE_FULL = 0x0e,
+ NVME_ZONE_STATE_OFFLINE = 0x0f,
+};
+
+struct nvme_zone_descriptor {
+ /** zone type */
+ uint8_t zt;
+#define NVME_ZONE_DESC_ZT_SHIFT (0)
+#define NVME_ZONE_DESC_ZT_MASK (0xF)
+
+ /** zone state */
+ uint8_t zs;
+#define NVME_ZONE_DESC_ZS_SHIFT (4)
+#define NVME_ZONE_DESC_ZS_MASK (0xF)
+
+ /** zone attributes */
+ uint8_t za;
+/* Zone Finished by Controller */
+#define NVME_ZONE_DESC_ZA_ZFC_SHIFT (0)
+#define NVME_ZONE_DESC_ZA_ZFC_MASK (0x1)
+/* Finish Zone Recommended */
+#define NVME_ZONE_DESC_ZA_FZR_SHIFT (1)
+#define NVME_ZONE_DESC_ZA_FZR_MASK (0x1)
+/* Reset Zone Recommended */
+#define NVME_ZONE_DESC_ZA_RZR_SHIFT (2)
+#define NVME_ZONE_DESC_ZA_RZR_MASK (0x1)
+/* ZRWA Valid */
+#define NVME_ZONE_DESC_ZA_ZRWAV_SHIFT (3)
+#define NVME_ZONE_DESC_ZA_ZRWAV_MASK (0x1)
+/* Zone Descriptor Extension Valid */
+#define NVME_ZONE_DESC_ZA_ZDEV_SHIFT (7)
+#define NVME_ZONE_DESC_ZA_ZDEV_MASK (0x1)
+
+ /** zone attributes information */
+ uint8_t zai;
+
+ /* bytes 4-7: Reserved */
+ uint8_t reserved1[4];
+
+ /** zone capacity */
+ uint64_t zcap;
+
+ /** zone start logical block address */
+ uint64_t zslba;
+
+ /** write pointer */
+ uint64_t wp;
+
+ /* bytes 32-63: Reserved */
+ uint8_t reserved2[32];
+} __packed __aligned(4);
+
+_Static_assert(sizeof(struct nvme_zone_descriptor) == 64,
+ "bad size for nvme_zone_descriptor");
+
+/* Report Zones data structure (Zone Management Receive) */
+struct nvme_zone_report {
+ /** number of zones matching the reporting options */
+ uint64_t nr_zones;
+
+ /* bytes 8-63: Reserved */
+ uint8_t reserved1[56];
+
+ struct nvme_zone_descriptor zone_desc[];
+} __packed __aligned(4);
+
+_Static_assert(sizeof(struct nvme_zone_report) == 64,
+ "bad size for nvme_zone_report");
+
+/* ZNS LBA Format Extension */
+struct nvme_zns_lbafe {
+ /** zone size (in logical blocks) */
+ uint64_t zsze;
+
+ /** zone descriptor extension size (in units of 64 bytes) */
+ uint8_t zdes;
+
+ /* bytes 9-15: Reserved */
+ uint8_t reserved1[7];
+} __packed;
+
+_Static_assert(sizeof(struct nvme_zns_lbafe) == 16,
+ "bad size for nvme_zns_lbafe");
+
+/* I/O Command Set specific Identify Namespace for ZNS (CNS 05h, CSI 02h) */
+struct nvme_zns_namespace_data {
+ /** zone operation characteristics */
+ uint16_t zoc;
+#define NVME_ZNS_NS_DATA_ZOC_VZC_SHIFT (0)
+#define NVME_ZNS_NS_DATA_ZOC_VZC_MASK (0x1)
+#define NVME_ZNS_NS_DATA_ZOC_ZAE_SHIFT (1)
+#define NVME_ZNS_NS_DATA_ZOC_ZAE_MASK (0x1)
+
+ /** optional zoned command support */
+ uint16_t ozcs;
+#define NVME_ZNS_NS_DATA_OZCS_RAZB_SHIFT (0)
+#define NVME_ZNS_NS_DATA_OZCS_RAZB_MASK (0x1)
+#define NVME_ZNS_NS_DATA_OZCS_ZRWASUP_SHIFT (1)
+#define NVME_ZNS_NS_DATA_OZCS_ZRWASUP_MASK (0x1)
+
+ /** maximum active resources (0's based) */
+ uint32_t mar;
+
+ /** maximum open resources (0's based) */
+ uint32_t mor;
+/* Value of mar and mor for a namespace that imposes no limit. */
+#define NVME_ZNS_NS_DATA_RESOURCES_UNLIMITED (0xffffffff)
+
+ /** reset recommended limit */
+ uint32_t rrl;
+
+ /** finish recommended limit */
+ uint32_t frl;
+
+ /** reset recommended limit 1-3 */
+ uint32_t rrl1;
+ uint32_t rrl2;
+ uint32_t rrl3;
+
+ /** finish recommended limit 1-3 */
+ uint32_t frl1;
+ uint32_t frl2;
+ uint32_t frl3;
+
+ /** number of ZRWA resources */
+ uint32_t numzrwa;
+
+ /** ZRWA flush granularity */
+ uint16_t zrwafg;
+
+ /** ZRWA size */
+ uint16_t zrwasz;
+
+ /** ZRWA capability */
+ uint8_t zrwacap;
+
+ /* bytes 53-2815: Reserved */
+ uint8_t reserved1[2763];
+
+ /** zns lba format extension support */
+ struct nvme_zns_lbafe lbafe[64];
+
+ uint8_t vendor_specific[256];
+} __packed __aligned(4);
+
+_Static_assert(sizeof(struct nvme_zns_namespace_data) == 4096,
+ "bad size for nvme_zns_namespace_data");
+
struct nvme_command_effects_page {
uint32_t acs[256];
uint32_t iocs[256];
@@ -2007,6 +2297,7 @@
NVME_NS_ALIVE = 0x04,
NVME_NS_DELTA = 0x08,
NVME_NS_GONE = 0x10,
+ NVME_NS_ZONED = 0x20,
};
int nvme_ctrlr_passthrough_cmd(struct nvme_controller *ctrlr,
@@ -2149,6 +2440,34 @@
cmd->cdw11 = htole32(NVME_DSM_ATTR_DEALLOCATE);
}
+static inline
+void nvme_zns_mgmt_send_cmd(struct nvme_command *cmd, uint32_t nsid,
+ uint64_t slba, bool select_all, uint8_t zsa)
+{
+ cmd->opc = NVME_OPC_ZONE_MGMT_SEND;
+ cmd->nsid = htole32(nsid);
+ cmd->cdw10 = htole32(slba & 0xffffffffu);
+ cmd->cdw11 = htole32(slba >> 32);
+ cmd->cdw13 = htole32(zsa |
+ (select_all ? NVME_ZONE_SEND_SELECT_ALL : 0));
+}
+
+/* dxfer_len must be a non-zero multiple of four. */
+static inline
+void nvme_zns_mgmt_recv_cmd(struct nvme_command *cmd, uint32_t nsid,
+ uint64_t slba, uint32_t dxfer_len, uint8_t zra, uint8_t zrasf,
+ bool partial)
+{
+ cmd->opc = NVME_OPC_ZONE_MGMT_RECV;
+ cmd->nsid = htole32(nsid);
+ cmd->cdw10 = htole32(slba & 0xffffffffu);
+ cmd->cdw11 = htole32(slba >> 32);
+ /* Number of dwords to transfer, 0's based */
+ cmd->cdw12 = htole32(dxfer_len / 4 - 1);
+ cmd->cdw13 = htole32(zra | (zrasf << 8) |
+ (partial ? NVME_ZONE_RECV_PARTIAL : 0));
+}
+
extern int nvme_use_nvd;
#endif /* _KERNEL */
@@ -2352,6 +2671,59 @@
#endif
}
+static inline
+void nvme_zns_namespace_data_swapbytes(
+ struct nvme_zns_namespace_data *s __unused)
+{
+#if _BYTE_ORDER != _LITTLE_ENDIAN
+ s->zoc = le16toh(s->zoc);
+ s->ozcs = le16toh(s->ozcs);
+ s->mar = le32toh(s->mar);
+ s->mor = le32toh(s->mor);
+ s->rrl = le32toh(s->rrl);
+ s->frl = le32toh(s->frl);
+ s->rrl1 = le32toh(s->rrl1);
+ s->rrl2 = le32toh(s->rrl2);
+ s->rrl3 = le32toh(s->rrl3);
+ s->frl1 = le32toh(s->frl1);
+ s->frl2 = le32toh(s->frl2);
+ s->frl3 = le32toh(s->frl3);
+ s->numzrwa = le32toh(s->numzrwa);
+ s->zrwafg = le16toh(s->zrwafg);
+ s->zrwasz = le16toh(s->zrwasz);
+ for (unsigned int i = 0; i < nitems(s->lbafe); i++)
+ s->lbafe[i].zsze = le64toh(s->lbafe[i].zsze);
+#endif
+}
+
+
+static inline
+void nvme_zone_descriptor_swapbytes(struct nvme_zone_descriptor *s __unused)
+{
+#if _BYTE_ORDER != _LITTLE_ENDIAN
+ s->zcap = le64toh(s->zcap);
+ s->zslba = le64toh(s->zslba);
+ s->wp = le64toh(s->wp);
+#endif
+}
+
+/*
+ * Swap the report header and the first num_zones zone descriptors, which
+ * must all be within the buffer the report was read into.
+ */
+static inline
+void nvme_zone_report_swapbytes(struct nvme_zone_report *s __unused,
+ uint64_t num_zones __unused)
+{
+#if _BYTE_ORDER != _LITTLE_ENDIAN
+ uint64_t i;
+
+ s->nr_zones = le64toh(s->nr_zones);
+ for (i = 0; i < num_zones; i++)
+ nvme_zone_descriptor_swapbytes(&s->zone_desc[i]);
+#endif
+}
+
static inline
void nvme_command_effects_page_swapbytes(
struct nvme_command_effects_page *s __unused)
diff --git a/sys/dev/nvme/nvme_ctrlr.c b/sys/dev/nvme/nvme_ctrlr.c
--- a/sys/dev/nvme/nvme_ctrlr.c
+++ b/sys/dev/nvme/nvme_ctrlr.c
@@ -400,16 +400,20 @@
/* Initialization values for CC */
cc = 0;
cc |= NVMEF(NVME_CC_REG_EN, 1);
- /* No CSI support; prefer the NVM command set when present. */
+ /*
+ * Select all supported I/O command sets when the controller
+ * implements them (CAP.CSS bit 6) so that namespaces using a
+ * command set other than NVM (e.g. ZNS) are active. Otherwise
+ * fall back to the NVM command set, or to no I/O command set at
+ * all for administrative controllers.
+ */
css = NVME_CAP_HI_CSS(ctrlr->cap_hi);
- if ((css & NVME_CAP_CSS_NVM) != 0)
+ if ((css & NVME_CAP_CSS_IOCSS) != 0)
+ cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_IOCSS);
+ else if ((css & NVME_CAP_CSS_NVM) != 0)
cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_NVM);
else if ((css & NVME_CAP_CSS_NOIOCSS) != 0)
cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_ADMIN);
- else if ((css & NVME_CAP_CSS_IOCSS) != 0)
- cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_IOCSS);
- else
- cc |= NVMEF(NVME_CC_REG_CSS, NVME_CC_CSS_NVM);
cc |= NVMEF(NVME_CC_REG_AMS, 0);
cc |= NVMEF(NVME_CC_REG_SHN, 0);
cc |= NVMEF(NVME_CC_REG_IOSQES, ctrlr->io_sqes);
diff --git a/sys/dev/nvme/nvme_ctrlr_cmd.c b/sys/dev/nvme/nvme_ctrlr_cmd.c
--- a/sys/dev/nvme/nvme_ctrlr_cmd.c
+++ b/sys/dev/nvme/nvme_ctrlr_cmd.c
@@ -30,46 +30,41 @@
#include "nvme_private.h"
void
-nvme_ctrlr_cmd_identify_controller(struct nvme_controller *ctrlr, void *payload,
- nvme_cb_fn_t cb_fn, void *cb_arg)
+nvme_ctrlr_cmd_identify(struct nvme_controller *ctrlr, uint8_t cns,
+ uint16_t cntid, uint32_t nsid, uint8_t csi, void *payload,
+ uint32_t payload_size, nvme_cb_fn_t cb_fn, void *cb_arg)
{
struct nvme_request *req;
struct nvme_command *cmd;
- req = nvme_allocate_request_vaddr(payload,
- sizeof(struct nvme_controller_data), M_WAITOK, cb_fn, cb_arg);
+ req = nvme_allocate_request_vaddr(payload, payload_size, M_WAITOK,
+ cb_fn, cb_arg);
cmd = &req->cmd;
cmd->opc = NVME_OPC_IDENTIFY;
-
- /*
- * TODO: create an identify command data structure, which
- * includes this CNS bit in cdw10.
- */
- cmd->cdw10 = htole32(1);
+ cmd->nsid = htole32(nsid);
+ cmd->cdw10 = htole32((uint32_t)cntid << 16 | cns);
+ cmd->cdw11 = htole32((uint32_t)csi << 24);
nvme_ctrlr_submit_admin_request(ctrlr, req);
}
void
-nvme_ctrlr_cmd_identify_namespace(struct nvme_controller *ctrlr, uint32_t nsid,
- void *payload, nvme_cb_fn_t cb_fn, void *cb_arg)
+nvme_ctrlr_cmd_identify_controller(struct nvme_controller *ctrlr, void *payload,
+ nvme_cb_fn_t cb_fn, void *cb_arg)
{
- struct nvme_request *req;
- struct nvme_command *cmd;
- req = nvme_allocate_request_vaddr(payload,
- sizeof(struct nvme_namespace_data), M_WAITOK, cb_fn, cb_arg);
-
- cmd = &req->cmd;
- cmd->opc = NVME_OPC_IDENTIFY;
+ nvme_ctrlr_cmd_identify(ctrlr, NVME_CNS_ID_CTRLR, 0, 0, 0, payload,
+ sizeof(struct nvme_controller_data), cb_fn, cb_arg);
+}
- /*
- * TODO: create an identify command data structure
- */
- cmd->nsid = htole32(nsid);
+void
+nvme_ctrlr_cmd_identify_namespace(struct nvme_controller *ctrlr, uint32_t nsid,
+ void *payload, nvme_cb_fn_t cb_fn, void *cb_arg)
+{
- nvme_ctrlr_submit_admin_request(ctrlr, req);
+ nvme_ctrlr_cmd_identify(ctrlr, NVME_CNS_ID_NS, 0, nsid, 0, payload,
+ sizeof(struct nvme_namespace_data), cb_fn, cb_arg);
}
void
diff --git a/sys/dev/nvme/nvme_ns.c b/sys/dev/nvme/nvme_ns.c
--- a/sys/dev/nvme/nvme_ns.c
+++ b/sys/dev/nvme/nvme_ns.c
@@ -564,6 +564,36 @@
return ((ns->flags & NVME_NS_ALIVE) ? 0 : ENXIO);
}
+ /*
+ * Determine which I/O command set the namespace is associated with.
+ * Namespaces using a command set other than NVM (e.g. ZNS)
+ * can only be active when the controller supports multiple command sets
+ * (CAP.CSS bit 6) and CC.CSS selected them. The CSI is reported in the
+ * Namespace Identification Descriptor list (CNS 03h); any failure to
+ * retrieve it is treated as the NVM command set.
+ */
+ ns->csi = NVME_CSI_NVM;
+ if (NVME_CAP_HI_CSS_IOCS(ctrlr->cap_hi) &&
+ (ctrlr->max_identify_cns == 0 ||
+ ctrlr->max_identify_cns >= NVME_CNS_ID_NS_DESC_LIST)) {
+ uint8_t *desclist;
+
+ desclist = malloc(NVME_NS_ID_DESC_LIST_SIZE, M_NVME,
+ M_WAITOK | M_ZERO);
+ status.done = 0;
+ nvme_ctrlr_cmd_identify(ctrlr, NVME_CNS_ID_NS_DESC_LIST, 0, id,
+ 0, desclist, NVME_NS_ID_DESC_LIST_SIZE,
+ nvme_completion_poll_cb, &status);
+ nvme_completion_poll(&status);
+ if (!nvme_completion_is_error(&status.cpl))
+ ns->csi = nvme_ns_id_desc_list_csi(desclist,
+ NVME_NS_ID_DESC_LIST_SIZE);
+ free(desclist, M_NVME);
+ }
+ ns->flags &= ~NVME_NS_ZONED;
+ if (ns->csi == NVME_CSI_ZNS)
+ ns->flags |= NVME_NS_ZONED;
+
/*
* Check the validity of the format specified. Note: format is a 0-based
* value, so > is appropriate here, not >=.
diff --git a/sys/dev/nvme/nvme_private.h b/sys/dev/nvme/nvme_private.h
--- a/sys/dev/nvme/nvme_private.h
+++ b/sys/dev/nvme/nvme_private.h
@@ -216,6 +216,7 @@
uint32_t flags;
struct cdev *cdev;
uint32_t boundary;
+ uint8_t csi;
struct mtx lock;
};
@@ -401,6 +402,10 @@
void nvme_ctrlr_cmd_identify_namespace(struct nvme_controller *ctrlr,
uint32_t nsid, void *payload,
nvme_cb_fn_t cb_fn, void *cb_arg);
+void nvme_ctrlr_cmd_identify(struct nvme_controller *ctrlr, uint8_t cns,
+ uint16_t cntid, uint32_t nsid, uint8_t csi,
+ void *payload, uint32_t payload_size,
+ nvme_cb_fn_t cb_fn, void *cb_arg);
void nvme_ctrlr_cmd_set_interrupt_coalescing(struct nvme_controller *ctrlr,
uint32_t microseconds,
uint32_t threshold,
diff --git a/sys/sys/disk_zone.h b/sys/sys/disk_zone.h
--- a/sys/sys/disk_zone.h
+++ b/sys/sys/disk_zone.h
@@ -151,8 +151,15 @@
(cond) == DISK_ZONE_COND_READONLY || \
(cond) == DISK_ZONE_COND_FULL || \
(cond) == DISK_ZONE_COND_OFFLINE)
+ /*
+ * Number of usable logical blocks in the zone, for devices (e.g.
+ * NVMe Zoned Namespaces) where this may be smaller than the zone
+ * length. Zero means the capacity was not reported; assume it is
+ * equal to the zone length.
+ */
+ uint64_t zone_capacity;
/* XXX KDM padding space may not be a good idea inside the bio */
- uint8_t reserved[32];
+ uint8_t reserved[24];
};
struct disk_zone_report {
File Metadata
Details
Attached
Mime Type
text/plain
Expires
Sun, Oct 11, 4:22 PM (19 h, 55 m)
Storage Engine
blob
Storage Format
Raw Data
Storage Handle
40590893
Default Alt Text
D60414.diff (41 KB)
Attached To
Mode
D60414: nvme: Add ZNS support (WIP)
Attached
Detach File
Event Timeline
Log In to Comment