Index: head/sys/dev/xen/blkfront/blkfront.c
===================================================================
--- head/sys/dev/xen/blkfront/blkfront.c	(revision 314839)
+++ head/sys/dev/xen/blkfront/blkfront.c	(revision 314840)
@@ -1,1608 +1,1613 @@
 /*
  * XenBSD block device driver
  *
  * Copyright (c) 2010-2013 Spectra Logic Corporation
  * Copyright (c) 2009 Scott Long, Yahoo!
  * Copyright (c) 2009 Frank Suchomel, Citrix
  * Copyright (c) 2009 Doug F. Rabson, Citrix
  * Copyright (c) 2005 Kip Macy
  * Copyright (c) 2003-2004, Keir Fraser & Steve Hand
  * Modifications by Mark A. Williamson are (c) Intel Research Cambridge
  *
  *
  * Permission is hereby granted, free of charge, to any person obtaining a copy
  * of this software and associated documentation files (the "Software"), to
  * deal in the Software without restriction, including without limitation the
  * rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
  * sell copies of the Software, and to permit persons to whom the Software is
  * furnished to do so, subject to the following conditions:
  *
  * The above copyright notice and this permission notice shall be included in
  * all copies or substantial portions of the Software.
  * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
  * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
  * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
  * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
  * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
  * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
  * DEALINGS IN THE SOFTWARE.
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/systm.h>
 #include <sys/malloc.h>
 #include <sys/kernel.h>
 #include <vm/vm.h>
 #include <vm/pmap.h>
 
 #include <sys/bio.h>
 #include <sys/bus.h>
 #include <sys/conf.h>
 #include <sys/module.h>
 #include <sys/sysctl.h>
 
 #include <machine/bus.h>
 #include <sys/rman.h>
 #include <machine/resource.h>
 #include <machine/intr_machdep.h>
 #include <machine/vmparam.h>
 #include <sys/bus_dma.h>
 
 #include <xen/xen-os.h>
 #include <xen/hypervisor.h>
 #include <xen/xen_intr.h>
 #include <xen/gnttab.h>
 #include <xen/interface/grant_table.h>
 #include <xen/interface/io/protocols.h>
 #include <xen/xenbus/xenbusvar.h>
 
 #include <machine/_inttypes.h>
 
 #include <geom/geom_disk.h>
 
 #include <dev/xen/blkfront/block.h>
 
 #include "xenbus_if.h"
 
 /*--------------------------- Forward Declarations ---------------------------*/
 static void xbd_closing(device_t);
 static void xbd_startio(struct xbd_softc *sc);
 
 /*---------------------------------- Macros ----------------------------------*/
 #if 0
 #define DPRINTK(fmt, args...) printf("[XEN] %s:%d: " fmt ".\n", __func__, __LINE__, ##args)
 #else
 #define DPRINTK(fmt, args...) 
 #endif
 
 #define XBD_SECTOR_SHFT		9
 
 /*---------------------------- Global Static Data ----------------------------*/
 static MALLOC_DEFINE(M_XENBLOCKFRONT, "xbd", "Xen Block Front driver data");
 
 static int xbd_enable_indirect = 1;
 SYSCTL_NODE(_hw, OID_AUTO, xbd, CTLFLAG_RD, 0, "xbd driver parameters");
 SYSCTL_INT(_hw_xbd, OID_AUTO, xbd_enable_indirect, CTLFLAG_RDTUN,
     &xbd_enable_indirect, 0, "Enable xbd indirect segments");
 
 /*---------------------------- Command Processing ----------------------------*/
 static void
 xbd_freeze(struct xbd_softc *sc, xbd_flag_t xbd_flag)
 {
 	if (xbd_flag != XBDF_NONE && (sc->xbd_flags & xbd_flag) != 0)
 		return;
 
 	sc->xbd_flags |= xbd_flag;
 	sc->xbd_qfrozen_cnt++;
 }
 
 static void
 xbd_thaw(struct xbd_softc *sc, xbd_flag_t xbd_flag)
 {
 	if (xbd_flag != XBDF_NONE && (sc->xbd_flags & xbd_flag) == 0)
 		return;
 
 	if (sc->xbd_qfrozen_cnt == 0)
 		panic("%s: Thaw with flag 0x%x while not frozen.",
 		    __func__, xbd_flag);
 
 	sc->xbd_flags &= ~xbd_flag;
 	sc->xbd_qfrozen_cnt--;
 }
 
 static void
 xbd_cm_freeze(struct xbd_softc *sc, struct xbd_command *cm, xbdc_flag_t cm_flag)
 {
 	if ((cm->cm_flags & XBDCF_FROZEN) != 0)
 		return;
 
 	cm->cm_flags |= XBDCF_FROZEN|cm_flag;
 	xbd_freeze(sc, XBDF_NONE);
 }
 
 static void
 xbd_cm_thaw(struct xbd_softc *sc, struct xbd_command *cm)
 {
 	if ((cm->cm_flags & XBDCF_FROZEN) == 0)
 		return;
 
 	cm->cm_flags &= ~XBDCF_FROZEN;
 	xbd_thaw(sc, XBDF_NONE);
 }
 
 static inline void 
 xbd_flush_requests(struct xbd_softc *sc)
 {
 	int notify;
 
 	RING_PUSH_REQUESTS_AND_CHECK_NOTIFY(&sc->xbd_ring, notify);
 
 	if (notify)
 		xen_intr_signal(sc->xen_intr_handle);
 }
 
 static void
 xbd_free_command(struct xbd_command *cm)
 {
 
 	KASSERT((cm->cm_flags & XBDCF_Q_MASK) == XBD_Q_NONE,
 	    ("Freeing command that is still on queue %d.",
 	    cm->cm_flags & XBDCF_Q_MASK));
 
 	cm->cm_flags = XBDCF_INITIALIZER;
 	cm->cm_bp = NULL;
 	cm->cm_complete = NULL;
 	xbd_enqueue_cm(cm, XBD_Q_FREE);
 	xbd_thaw(cm->cm_sc, XBDF_CM_SHORTAGE);
 }
 
 static void
 xbd_mksegarray(bus_dma_segment_t *segs, int nsegs,
     grant_ref_t * gref_head, int otherend_id, int readonly,
     grant_ref_t * sg_ref, struct blkif_request_segment *sg)
 {
 	struct blkif_request_segment *last_block_sg = sg + nsegs;
 	vm_paddr_t buffer_ma;
 	uint64_t fsect, lsect;
 	int ref;
 
 	while (sg < last_block_sg) {
 		KASSERT(segs->ds_addr % (1 << XBD_SECTOR_SHFT) == 0,
 		    ("XEN disk driver I/O must be sector aligned"));
 		KASSERT(segs->ds_len % (1 << XBD_SECTOR_SHFT) == 0,
 		    ("XEN disk driver I/Os must be a multiple of "
 		    "the sector length"));
 		buffer_ma = segs->ds_addr;
 		fsect = (buffer_ma & PAGE_MASK) >> XBD_SECTOR_SHFT;
 		lsect = fsect + (segs->ds_len  >> XBD_SECTOR_SHFT) - 1;
 
 		KASSERT(lsect <= 7, ("XEN disk driver data cannot "
 		    "cross a page boundary"));
 
 		/* install a grant reference. */
 		ref = gnttab_claim_grant_reference(gref_head);
 
 		/*
 		 * GNTTAB_LIST_END == 0xffffffff, but it is private
 		 * to gnttab.c.
 		 */
 		KASSERT(ref != ~0, ("grant_reference failed"));
 
 		gnttab_grant_foreign_access_ref(
 		    ref,
 		    otherend_id,
 		    buffer_ma >> PAGE_SHIFT,
 		    readonly);
 
 		*sg_ref = ref;
 		*sg = (struct blkif_request_segment) {
 			.gref       = ref,
 			.first_sect = fsect, 
 			.last_sect  = lsect
 		};
 		sg++;
 		sg_ref++;
 		segs++;
 	}
 }
 
 static void
 xbd_queue_cb(void *arg, bus_dma_segment_t *segs, int nsegs, int error)
 {
 	struct xbd_softc *sc;
 	struct xbd_command *cm;
 	int op;
 
 	cm = arg;
 	sc = cm->cm_sc;
 
 	if (error) {
 		cm->cm_bp->bio_error = EIO;
 		biodone(cm->cm_bp);
 		xbd_free_command(cm);
 		return;
 	}
 
 	KASSERT(nsegs <= sc->xbd_max_request_segments,
 	    ("Too many segments in a blkfront I/O"));
 
 	if (nsegs <= BLKIF_MAX_SEGMENTS_PER_REQUEST) {
 		blkif_request_t	*ring_req;
 
 		/* Fill out a blkif_request_t structure. */
 		ring_req = (blkif_request_t *)
 		    RING_GET_REQUEST(&sc->xbd_ring, sc->xbd_ring.req_prod_pvt);
 		sc->xbd_ring.req_prod_pvt++;
 		ring_req->id = cm->cm_id;
 		ring_req->operation = cm->cm_operation;
 		ring_req->sector_number = cm->cm_sector_number;
 		ring_req->handle = (blkif_vdev_t)(uintptr_t)sc->xbd_disk;
 		ring_req->nr_segments = nsegs;
 		cm->cm_nseg = nsegs;
 		xbd_mksegarray(segs, nsegs, &cm->cm_gref_head,
 		    xenbus_get_otherend_id(sc->xbd_dev),
 		    cm->cm_operation == BLKIF_OP_WRITE,
 		    cm->cm_sg_refs, ring_req->seg);
 	} else {
 		blkif_request_indirect_t *ring_req;
 
 		/* Fill out a blkif_request_indirect_t structure. */
 		ring_req = (blkif_request_indirect_t *)
 		    RING_GET_REQUEST(&sc->xbd_ring, sc->xbd_ring.req_prod_pvt);
 		sc->xbd_ring.req_prod_pvt++;
 		ring_req->id = cm->cm_id;
 		ring_req->operation = BLKIF_OP_INDIRECT;
 		ring_req->indirect_op = cm->cm_operation;
 		ring_req->sector_number = cm->cm_sector_number;
 		ring_req->handle = (blkif_vdev_t)(uintptr_t)sc->xbd_disk;
 		ring_req->nr_segments = nsegs;
 		cm->cm_nseg = nsegs;
 		xbd_mksegarray(segs, nsegs, &cm->cm_gref_head,
 		    xenbus_get_otherend_id(sc->xbd_dev),
 		    cm->cm_operation == BLKIF_OP_WRITE,
 		    cm->cm_sg_refs, cm->cm_indirectionpages);
 		memcpy(ring_req->indirect_grefs, &cm->cm_indirectionrefs,
 		    sizeof(grant_ref_t) * sc->xbd_max_request_indirectpages);
 	}
 
 	if (cm->cm_operation == BLKIF_OP_READ)
 		op = BUS_DMASYNC_PREREAD;
 	else if (cm->cm_operation == BLKIF_OP_WRITE)
 		op = BUS_DMASYNC_PREWRITE;
 	else
 		op = 0;
 	bus_dmamap_sync(sc->xbd_io_dmat, cm->cm_map, op);
 
 	gnttab_free_grant_references(cm->cm_gref_head);
 
 	xbd_enqueue_cm(cm, XBD_Q_BUSY);
 
 	/*
 	 * If bus dma had to asynchronously call us back to dispatch
 	 * this command, we are no longer executing in the context of 
 	 * xbd_startio().  Thus we cannot rely on xbd_startio()'s call to
 	 * xbd_flush_requests() to publish this command to the backend
 	 * along with any other commands that it could batch.
 	 */
 	if ((cm->cm_flags & XBDCF_ASYNC_MAPPING) != 0)
 		xbd_flush_requests(sc);
 
 	return;
 }
 
 static int
 xbd_queue_request(struct xbd_softc *sc, struct xbd_command *cm)
 {
 	int error;
 
 	if (cm->cm_bp != NULL)
 		error = bus_dmamap_load_bio(sc->xbd_io_dmat, cm->cm_map,
 		    cm->cm_bp, xbd_queue_cb, cm, 0);
 	else
 		error = bus_dmamap_load(sc->xbd_io_dmat, cm->cm_map,
 		    cm->cm_data, cm->cm_datalen, xbd_queue_cb, cm, 0);
 	if (error == EINPROGRESS) {
 		/*
 		 * Maintain queuing order by freezing the queue.  The next
 		 * command may not require as many resources as the command
 		 * we just attempted to map, so we can't rely on bus dma
 		 * blocking for it too.
 		 */
 		xbd_cm_freeze(sc, cm, XBDCF_ASYNC_MAPPING);
 		return (0);
 	}
 
 	return (error);
 }
 
 static void
 xbd_restart_queue_callback(void *arg)
 {
 	struct xbd_softc *sc = arg;
 
 	mtx_lock(&sc->xbd_io_lock);
 
 	xbd_thaw(sc, XBDF_GNT_SHORTAGE);
 
 	xbd_startio(sc);
 
 	mtx_unlock(&sc->xbd_io_lock);
 }
 
 static struct xbd_command *
 xbd_bio_command(struct xbd_softc *sc)
 {
 	struct xbd_command *cm;
 	struct bio *bp;
 
 	if (__predict_false(sc->xbd_state != XBD_STATE_CONNECTED))
 		return (NULL);
 
 	bp = xbd_dequeue_bio(sc);
 	if (bp == NULL)
 		return (NULL);
 
 	if ((cm = xbd_dequeue_cm(sc, XBD_Q_FREE)) == NULL) {
 		xbd_freeze(sc, XBDF_CM_SHORTAGE);
 		xbd_requeue_bio(sc, bp);
 		return (NULL);
 	}
 
 	if (gnttab_alloc_grant_references(sc->xbd_max_request_segments,
 	    &cm->cm_gref_head) != 0) {
 		gnttab_request_free_callback(&sc->xbd_callback,
 		    xbd_restart_queue_callback, sc,
 		    sc->xbd_max_request_segments);
 		xbd_freeze(sc, XBDF_GNT_SHORTAGE);
 		xbd_requeue_bio(sc, bp);
 		xbd_enqueue_cm(cm, XBD_Q_FREE);
 		return (NULL);
 	}
 
 	cm->cm_bp = bp;
 	cm->cm_sector_number = (blkif_sector_t)bp->bio_pblkno;
 
 	switch (bp->bio_cmd) {
 	case BIO_READ:
 		cm->cm_operation = BLKIF_OP_READ;
 		break;
 	case BIO_WRITE:
 		cm->cm_operation = BLKIF_OP_WRITE;
 		if ((bp->bio_flags & BIO_ORDERED) != 0) {
 			if ((sc->xbd_flags & XBDF_BARRIER) != 0) {
 				cm->cm_operation = BLKIF_OP_WRITE_BARRIER;
 			} else {
 				/*
 				 * Single step this command.
 				 */
 				cm->cm_flags |= XBDCF_Q_FREEZE;
 				if (xbd_queue_length(sc, XBD_Q_BUSY) != 0) {
 					/*
 					 * Wait for in-flight requests to
 					 * finish.
 					 */
 					xbd_freeze(sc, XBDF_WAIT_IDLE);
 					xbd_requeue_cm(cm, XBD_Q_READY);
 					return (NULL);
 				}
 			}
 		}
 		break;
 	case BIO_FLUSH:
 		if ((sc->xbd_flags & XBDF_FLUSH) != 0)
 			cm->cm_operation = BLKIF_OP_FLUSH_DISKCACHE;
 		else if ((sc->xbd_flags & XBDF_BARRIER) != 0)
 			cm->cm_operation = BLKIF_OP_WRITE_BARRIER;
 		else
 			panic("flush request, but no flush support available");
 		break;
 	default:
 		panic("unknown bio command %d", bp->bio_cmd);
 	}
 
 	return (cm);
 }
 
 /*
  * Dequeue buffers and place them in the shared communication ring.
  * Return when no more requests can be accepted or all buffers have 
  * been queued.
  *
  * Signal XEN once the ring has been filled out.
  */
 static void
 xbd_startio(struct xbd_softc *sc)
 {
 	struct xbd_command *cm;
 	int error, queued = 0;
 
 	mtx_assert(&sc->xbd_io_lock, MA_OWNED);
 
 	if (sc->xbd_state != XBD_STATE_CONNECTED)
 		return;
 
 	while (!RING_FULL(&sc->xbd_ring)) {
 
 		if (sc->xbd_qfrozen_cnt != 0)
 			break;
 
 		cm = xbd_dequeue_cm(sc, XBD_Q_READY);
 
 		if (cm == NULL)
 		    cm = xbd_bio_command(sc);
 
 		if (cm == NULL)
 			break;
 
 		if ((cm->cm_flags & XBDCF_Q_FREEZE) != 0) {
 			/*
 			 * Single step command.  Future work is
 			 * held off until this command completes.
 			 */
 			xbd_cm_freeze(sc, cm, XBDCF_Q_FREEZE);
 		}
 
 		if ((error = xbd_queue_request(sc, cm)) != 0) {
 			printf("xbd_queue_request returned %d\n", error);
 			break;
 		}
 		queued++;
 	}
 
 	if (queued != 0) 
 		xbd_flush_requests(sc);
 }
 
 static void
 xbd_bio_complete(struct xbd_softc *sc, struct xbd_command *cm)
 {
 	struct bio *bp;
 
 	bp = cm->cm_bp;
 
 	if (__predict_false(cm->cm_status != BLKIF_RSP_OKAY)) {
 		disk_err(bp, "disk error" , -1, 0);
 		printf(" status: %x\n", cm->cm_status);
 		bp->bio_flags |= BIO_ERROR;
 	}
 
 	if (bp->bio_flags & BIO_ERROR)
 		bp->bio_error = EIO;
 	else
 		bp->bio_resid = 0;
 
 	xbd_free_command(cm);
 	biodone(bp);
 }
 
 static void
 xbd_int(void *xsc)
 {
 	struct xbd_softc *sc = xsc;
 	struct xbd_command *cm;
 	blkif_response_t *bret;
 	RING_IDX i, rp;
 	int op;
 
 	mtx_lock(&sc->xbd_io_lock);
 
 	if (__predict_false(sc->xbd_state == XBD_STATE_DISCONNECTED)) {
 		mtx_unlock(&sc->xbd_io_lock);
 		return;
 	}
 
  again:
 	rp = sc->xbd_ring.sring->rsp_prod;
 	rmb(); /* Ensure we see queued responses up to 'rp'. */
 
 	for (i = sc->xbd_ring.rsp_cons; i != rp;) {
 		bret = RING_GET_RESPONSE(&sc->xbd_ring, i);
 		cm   = &sc->xbd_shadow[bret->id];
 
 		xbd_remove_cm(cm, XBD_Q_BUSY);
 		gnttab_end_foreign_access_references(cm->cm_nseg,
 		    cm->cm_sg_refs);
 		i++;
 
 		if (cm->cm_operation == BLKIF_OP_READ)
 			op = BUS_DMASYNC_POSTREAD;
 		else if (cm->cm_operation == BLKIF_OP_WRITE ||
 		    cm->cm_operation == BLKIF_OP_WRITE_BARRIER)
 			op = BUS_DMASYNC_POSTWRITE;
 		else
 			op = 0;
 		bus_dmamap_sync(sc->xbd_io_dmat, cm->cm_map, op);
 		bus_dmamap_unload(sc->xbd_io_dmat, cm->cm_map);
 
 		/*
 		 * Release any hold this command has on future command
 		 * dispatch. 
 		 */
 		xbd_cm_thaw(sc, cm);
 
 		/*
 		 * Directly call the i/o complete routine to save an
 		 * an indirection in the common case.
 		 */
 		cm->cm_status = bret->status;
 		if (cm->cm_bp)
 			xbd_bio_complete(sc, cm);
 		else if (cm->cm_complete != NULL)
 			cm->cm_complete(cm);
 		else
 			xbd_free_command(cm);
 	}
 
 	sc->xbd_ring.rsp_cons = i;
 
 	if (i != sc->xbd_ring.req_prod_pvt) {
 		int more_to_do;
 		RING_FINAL_CHECK_FOR_RESPONSES(&sc->xbd_ring, more_to_do);
 		if (more_to_do)
 			goto again;
 	} else {
 		sc->xbd_ring.sring->rsp_event = i + 1;
 	}
 
 	if (xbd_queue_length(sc, XBD_Q_BUSY) == 0)
 		xbd_thaw(sc, XBDF_WAIT_IDLE);
 
 	xbd_startio(sc);
 
 	if (__predict_false(sc->xbd_state == XBD_STATE_SUSPENDED))
 		wakeup(&sc->xbd_cm_q[XBD_Q_BUSY]);
 
 	mtx_unlock(&sc->xbd_io_lock);
 }
 
 /*------------------------------- Dump Support -------------------------------*/
 /**
  * Quiesce the disk writes for a dump file before allowing the next buffer.
  */
 static void
 xbd_quiesce(struct xbd_softc *sc)
 {
 	int mtd;
 
 	// While there are outstanding requests
 	while (xbd_queue_length(sc, XBD_Q_BUSY) != 0) {
 		RING_FINAL_CHECK_FOR_RESPONSES(&sc->xbd_ring, mtd);
 		if (mtd) {
 			/* Received request completions, update queue. */
 			xbd_int(sc);
 		}
 		if (xbd_queue_length(sc, XBD_Q_BUSY) != 0) {
 			/*
 			 * Still pending requests, wait for the disk i/o
 			 * to complete.
 			 */
 			HYPERVISOR_yield();
 		}
 	}
 }
 
 /* Kernel dump function for a paravirtualized disk device */
 static void
 xbd_dump_complete(struct xbd_command *cm)
 {
 
 	xbd_enqueue_cm(cm, XBD_Q_COMPLETE);
 }
 
 static int
 xbd_dump(void *arg, void *virtual, vm_offset_t physical, off_t offset,
     size_t length)
 {
 	struct disk *dp = arg;
 	struct xbd_softc *sc = dp->d_drv1;
 	struct xbd_command *cm;
 	size_t chunk;
 	int sbp;
 	int rc = 0;
 
 	if (length <= 0)
 		return (rc);
 
 	xbd_quiesce(sc);	/* All quiet on the western front. */
 
 	/*
 	 * If this lock is held, then this module is failing, and a
 	 * successful kernel dump is highly unlikely anyway.
 	 */
 	mtx_lock(&sc->xbd_io_lock);
 
 	/* Split the 64KB block as needed */
 	for (sbp=0; length > 0; sbp++) {
 		cm = xbd_dequeue_cm(sc, XBD_Q_FREE);
 		if (cm == NULL) {
 			mtx_unlock(&sc->xbd_io_lock);
 			device_printf(sc->xbd_dev, "dump: no more commands?\n");
 			return (EBUSY);
 		}
 
 		if (gnttab_alloc_grant_references(sc->xbd_max_request_segments,
 		    &cm->cm_gref_head) != 0) {
 			xbd_free_command(cm);
 			mtx_unlock(&sc->xbd_io_lock);
 			device_printf(sc->xbd_dev, "no more grant allocs?\n");
 			return (EBUSY);
 		}
 
 		chunk = length > sc->xbd_max_request_size ?
 		    sc->xbd_max_request_size : length;
 		cm->cm_data = virtual;
 		cm->cm_datalen = chunk;
 		cm->cm_operation = BLKIF_OP_WRITE;
 		cm->cm_sector_number = offset / dp->d_sectorsize;
 		cm->cm_complete = xbd_dump_complete;
 
 		xbd_enqueue_cm(cm, XBD_Q_READY);
 
 		length -= chunk;
 		offset += chunk;
 		virtual = (char *) virtual + chunk;
 	}
 
 	/* Tell DOM0 to do the I/O */
 	xbd_startio(sc);
 	mtx_unlock(&sc->xbd_io_lock);
 
 	/* Poll for the completion. */
 	xbd_quiesce(sc);	/* All quite on the eastern front */
 
 	/* If there were any errors, bail out... */
 	while ((cm = xbd_dequeue_cm(sc, XBD_Q_COMPLETE)) != NULL) {
 		if (cm->cm_status != BLKIF_RSP_OKAY) {
 			device_printf(sc->xbd_dev,
 			    "Dump I/O failed at sector %jd\n",
 			    cm->cm_sector_number);
 			rc = EIO;
 		}
 		xbd_free_command(cm);
 	}
 
 	return (rc);
 }
 
 /*----------------------------- Disk Entrypoints -----------------------------*/
 static int
 xbd_open(struct disk *dp)
 {
 	struct xbd_softc *sc = dp->d_drv1;
 
 	if (sc == NULL) {
 		printf("xbd%d: not found", dp->d_unit);
 		return (ENXIO);
 	}
 
 	sc->xbd_flags |= XBDF_OPEN;
 	sc->xbd_users++;
 	return (0);
 }
 
 static int
 xbd_close(struct disk *dp)
 {
 	struct xbd_softc *sc = dp->d_drv1;
 
 	if (sc == NULL)
 		return (ENXIO);
 	sc->xbd_flags &= ~XBDF_OPEN;
 	if (--(sc->xbd_users) == 0) {
 		/*
 		 * Check whether we have been instructed to close.  We will
 		 * have ignored this request initially, as the device was
 		 * still mounted.
 		 */
 		if (xenbus_get_otherend_state(sc->xbd_dev) ==
 		    XenbusStateClosing)
 			xbd_closing(sc->xbd_dev);
 	}
 	return (0);
 }
 
 static int
 xbd_ioctl(struct disk *dp, u_long cmd, void *addr, int flag, struct thread *td)
 {
 	struct xbd_softc *sc = dp->d_drv1;
 
 	if (sc == NULL)
 		return (ENXIO);
 
 	return (ENOTTY);
 }
 
 /*
  * Read/write routine for a buffer.  Finds the proper unit, place it on
  * the sortq and kick the controller.
  */
 static void
 xbd_strategy(struct bio *bp)
 {
 	struct xbd_softc *sc = bp->bio_disk->d_drv1;
 
 	/* bogus disk? */
 	if (sc == NULL) {
 		bp->bio_error = EINVAL;
 		bp->bio_flags |= BIO_ERROR;
 		bp->bio_resid = bp->bio_bcount;
 		biodone(bp);
 		return;
 	}
 
 	/*
 	 * Place it in the queue of disk activities for this disk
 	 */
 	mtx_lock(&sc->xbd_io_lock);
 
 	xbd_enqueue_bio(sc, bp);
 	xbd_startio(sc);
 
 	mtx_unlock(&sc->xbd_io_lock);
 	return;
 }
 
 /*------------------------------ Ring Management -----------------------------*/
 static int 
 xbd_alloc_ring(struct xbd_softc *sc)
 {
 	blkif_sring_t *sring;
 	uintptr_t sring_page_addr;
 	int error;
 	int i;
 
 	sring = malloc(sc->xbd_ring_pages * PAGE_SIZE, M_XENBLOCKFRONT,
 	    M_NOWAIT|M_ZERO);
 	if (sring == NULL) {
 		xenbus_dev_fatal(sc->xbd_dev, ENOMEM, "allocating shared ring");
 		return (ENOMEM);
 	}
 	SHARED_RING_INIT(sring);
 	FRONT_RING_INIT(&sc->xbd_ring, sring, sc->xbd_ring_pages * PAGE_SIZE);
 
 	for (i = 0, sring_page_addr = (uintptr_t)sring;
 	     i < sc->xbd_ring_pages;
 	     i++, sring_page_addr += PAGE_SIZE) {
 
 		error = xenbus_grant_ring(sc->xbd_dev,
 		    (vtophys(sring_page_addr) >> PAGE_SHIFT),
 		    &sc->xbd_ring_ref[i]);
 		if (error) {
 			xenbus_dev_fatal(sc->xbd_dev, error,
 			    "granting ring_ref(%d)", i);
 			return (error);
 		}
 	}
 	if (sc->xbd_ring_pages == 1) {
 		error = xs_printf(XST_NIL, xenbus_get_node(sc->xbd_dev),
 		    "ring-ref", "%u", sc->xbd_ring_ref[0]);
 		if (error) {
 			xenbus_dev_fatal(sc->xbd_dev, error,
 			    "writing %s/ring-ref",
 			    xenbus_get_node(sc->xbd_dev));
 			return (error);
 		}
 	} else {
 		for (i = 0; i < sc->xbd_ring_pages; i++) {
 			char ring_ref_name[]= "ring_refXX";
 
 			snprintf(ring_ref_name, sizeof(ring_ref_name),
 			    "ring-ref%u", i);
 			error = xs_printf(XST_NIL, xenbus_get_node(sc->xbd_dev),
 			     ring_ref_name, "%u", sc->xbd_ring_ref[i]);
 			if (error) {
 				xenbus_dev_fatal(sc->xbd_dev, error,
 				    "writing %s/%s",
 				    xenbus_get_node(sc->xbd_dev),
 				    ring_ref_name);
 				return (error);
 			}
 		}
 	}
 
 	error = xen_intr_alloc_and_bind_local_port(sc->xbd_dev,
 	    xenbus_get_otherend_id(sc->xbd_dev), NULL, xbd_int, sc,
 	    INTR_TYPE_BIO | INTR_MPSAFE, &sc->xen_intr_handle);
 	if (error) {
 		xenbus_dev_fatal(sc->xbd_dev, error,
 		    "xen_intr_alloc_and_bind_local_port failed");
 		return (error);
 	}
 
 	return (0);
 }
 
 static void
 xbd_free_ring(struct xbd_softc *sc)
 {
 	int i;
 
 	if (sc->xbd_ring.sring == NULL)
 		return;
 
 	for (i = 0; i < sc->xbd_ring_pages; i++) {
 		if (sc->xbd_ring_ref[i] != GRANT_REF_INVALID) {
 			gnttab_end_foreign_access_ref(sc->xbd_ring_ref[i]);
 			sc->xbd_ring_ref[i] = GRANT_REF_INVALID;
 		}
 	}
 	free(sc->xbd_ring.sring, M_XENBLOCKFRONT);
 	sc->xbd_ring.sring = NULL;
 }
 
 /*-------------------------- Initialization/Teardown -------------------------*/
 static int
 xbd_feature_string(struct xbd_softc *sc, char *features, size_t len)
 {
 	struct sbuf sb;
 	int feature_cnt;
 
 	sbuf_new(&sb, features, len, SBUF_FIXEDLEN);
 
 	feature_cnt = 0;
 	if ((sc->xbd_flags & XBDF_FLUSH) != 0) {
 		sbuf_printf(&sb, "flush");
 		feature_cnt++;
 	}
 
 	if ((sc->xbd_flags & XBDF_BARRIER) != 0) {
 		if (feature_cnt != 0)
 			sbuf_printf(&sb, ", ");
 		sbuf_printf(&sb, "write_barrier");
 		feature_cnt++;
 	}
 
 	if ((sc->xbd_flags & XBDF_DISCARD) != 0) {
 		if (feature_cnt != 0)
 			sbuf_printf(&sb, ", ");
 		sbuf_printf(&sb, "discard");
 		feature_cnt++;
 	}
 
 	if ((sc->xbd_flags & XBDF_PERSISTENT) != 0) {
 		if (feature_cnt != 0)
 			sbuf_printf(&sb, ", ");
 		sbuf_printf(&sb, "persistent_grants");
 		feature_cnt++;
 	}
 
 	(void) sbuf_finish(&sb);
 	return (sbuf_len(&sb));
 }
 
 static int
 xbd_sysctl_features(SYSCTL_HANDLER_ARGS)
 {
 	char features[80];
 	struct xbd_softc *sc = arg1;
 	int error;
 	int len;
 
 	error = sysctl_wire_old_buffer(req, 0);
 	if (error != 0)
 		return (error);
 
 	len = xbd_feature_string(sc, features, sizeof(features));
 
 	/* len is -1 on error, which will make the SYSCTL_OUT a no-op. */
 	return (SYSCTL_OUT(req, features, len + 1/*NUL*/));
 }
 
 static void
 xbd_setup_sysctl(struct xbd_softc *xbd)
 {
 	struct sysctl_ctx_list *sysctl_ctx = NULL;
 	struct sysctl_oid *sysctl_tree = NULL;
 	struct sysctl_oid_list *children;
 	
 	sysctl_ctx = device_get_sysctl_ctx(xbd->xbd_dev);
 	if (sysctl_ctx == NULL)
 		return;
 
 	sysctl_tree = device_get_sysctl_tree(xbd->xbd_dev);
 	if (sysctl_tree == NULL)
 		return;
 
 	children = SYSCTL_CHILDREN(sysctl_tree);
 	SYSCTL_ADD_UINT(sysctl_ctx, children, OID_AUTO,
 	    "max_requests", CTLFLAG_RD, &xbd->xbd_max_requests, -1,
 	    "maximum outstanding requests (negotiated)");
 
 	SYSCTL_ADD_UINT(sysctl_ctx, children, OID_AUTO,
 	    "max_request_segments", CTLFLAG_RD,
 	    &xbd->xbd_max_request_segments, 0,
 	    "maximum number of pages per requests (negotiated)");
 
 	SYSCTL_ADD_UINT(sysctl_ctx, children, OID_AUTO,
 	    "max_request_size", CTLFLAG_RD, &xbd->xbd_max_request_size, 0,
 	    "maximum size in bytes of a request (negotiated)");
 
 	SYSCTL_ADD_UINT(sysctl_ctx, children, OID_AUTO,
 	    "ring_pages", CTLFLAG_RD, &xbd->xbd_ring_pages, 0,
 	    "communication channel pages (negotiated)");
 
 	SYSCTL_ADD_PROC(sysctl_ctx, children, OID_AUTO,
 	    "features", CTLTYPE_STRING|CTLFLAG_RD, xbd, 0,
 	    xbd_sysctl_features, "A", "protocol features (negotiated)");
 }
 
 /*
  * Translate Linux major/minor to an appropriate name and unit
  * number. For HVM guests, this allows us to use the same drive names
  * with blkfront as the emulated drives, easing transition slightly.
  */
 static void
 xbd_vdevice_to_unit(uint32_t vdevice, int *unit, const char **name)
 {
 	static struct vdev_info {
 		int major;
 		int shift;
 		int base;
 		const char *name;
 	} info[] = {
 		{3,	6,	0,	"ada"},	/* ide0 */
 		{22,	6,	2,	"ada"},	/* ide1 */
 		{33,	6,	4,	"ada"},	/* ide2 */
 		{34,	6,	6,	"ada"},	/* ide3 */
 		{56,	6,	8,	"ada"},	/* ide4 */
 		{57,	6,	10,	"ada"},	/* ide5 */
 		{88,	6,	12,	"ada"},	/* ide6 */
 		{89,	6,	14,	"ada"},	/* ide7 */
 		{90,	6,	16,	"ada"},	/* ide8 */
 		{91,	6,	18,	"ada"},	/* ide9 */
 
 		{8,	4,	0,	"da"},	/* scsi disk0 */
 		{65,	4,	16,	"da"},	/* scsi disk1 */
 		{66,	4,	32,	"da"},	/* scsi disk2 */
 		{67,	4,	48,	"da"},	/* scsi disk3 */
 		{68,	4,	64,	"da"},	/* scsi disk4 */
 		{69,	4,	80,	"da"},	/* scsi disk5 */
 		{70,	4,	96,	"da"},	/* scsi disk6 */
 		{71,	4,	112,	"da"},	/* scsi disk7 */
 		{128,	4,	128,	"da"},	/* scsi disk8 */
 		{129,	4,	144,	"da"},	/* scsi disk9 */
 		{130,	4,	160,	"da"},	/* scsi disk10 */
 		{131,	4,	176,	"da"},	/* scsi disk11 */
 		{132,	4,	192,	"da"},	/* scsi disk12 */
 		{133,	4,	208,	"da"},	/* scsi disk13 */
 		{134,	4,	224,	"da"},	/* scsi disk14 */
 		{135,	4,	240,	"da"},	/* scsi disk15 */
 
 		{202,	4,	0,	"xbd"},	/* xbd */
 
 		{0,	0,	0,	NULL},
 	};
 	int major = vdevice >> 8;
 	int minor = vdevice & 0xff;
 	int i;
 
 	if (vdevice & (1 << 28)) {
 		*unit = (vdevice & ((1 << 28) - 1)) >> 8;
 		*name = "xbd";
 		return;
 	}
 
 	for (i = 0; info[i].major; i++) {
 		if (info[i].major == major) {
 			*unit = info[i].base + (minor >> info[i].shift);
 			*name = info[i].name;
 			return;
 		}
 	}
 
 	*unit = minor >> 4;
 	*name = "xbd";
 }
 
 int
 xbd_instance_create(struct xbd_softc *sc, blkif_sector_t sectors,
     int vdevice, uint16_t vdisk_info, unsigned long sector_size,
     unsigned long phys_sector_size)
 {
 	char features[80];
 	int unit, error = 0;
 	const char *name;
 
 	xbd_vdevice_to_unit(vdevice, &unit, &name);
 
 	sc->xbd_unit = unit;
 
 	if (strcmp(name, "xbd") != 0)
 		device_printf(sc->xbd_dev, "attaching as %s%d\n", name, unit);
 
 	if (xbd_feature_string(sc, features, sizeof(features)) > 0) {
 		device_printf(sc->xbd_dev, "features: %s\n",
 		    features);
 	}
 
 	sc->xbd_disk = disk_alloc();
 	sc->xbd_disk->d_unit = sc->xbd_unit;
 	sc->xbd_disk->d_open = xbd_open;
 	sc->xbd_disk->d_close = xbd_close;
 	sc->xbd_disk->d_ioctl = xbd_ioctl;
 	sc->xbd_disk->d_strategy = xbd_strategy;
 	sc->xbd_disk->d_dump = xbd_dump;
 	sc->xbd_disk->d_name = name;
 	sc->xbd_disk->d_drv1 = sc;
 	sc->xbd_disk->d_sectorsize = sector_size;
 	sc->xbd_disk->d_stripesize = phys_sector_size;
 	sc->xbd_disk->d_stripeoffset = 0;
 
 	sc->xbd_disk->d_mediasize = sectors * sector_size;
 	sc->xbd_disk->d_maxsize = sc->xbd_max_request_size;
 	sc->xbd_disk->d_flags = DISKFLAG_UNMAPPED_BIO;
 	if ((sc->xbd_flags & (XBDF_FLUSH|XBDF_BARRIER)) != 0) {
 		sc->xbd_disk->d_flags |= DISKFLAG_CANFLUSHCACHE;
 		device_printf(sc->xbd_dev,
 		    "synchronize cache commands enabled.\n");
 	}
 	disk_create(sc->xbd_disk, DISK_VERSION);
 
 	return error;
 }
 
 static void 
 xbd_free(struct xbd_softc *sc)
 {
 	int i;
 	
 	/* Prevent new requests being issued until we fix things up. */
 	mtx_lock(&sc->xbd_io_lock);
 	sc->xbd_state = XBD_STATE_DISCONNECTED; 
 	mtx_unlock(&sc->xbd_io_lock);
 
 	/* Free resources associated with old device channel. */
 	xbd_free_ring(sc);
 	if (sc->xbd_shadow) {
 
 		for (i = 0; i < sc->xbd_max_requests; i++) {
 			struct xbd_command *cm;
 
 			cm = &sc->xbd_shadow[i];
 			if (cm->cm_sg_refs != NULL) {
 				free(cm->cm_sg_refs, M_XENBLOCKFRONT);
 				cm->cm_sg_refs = NULL;
 			}
 
 			if (cm->cm_indirectionpages != NULL) {
 				gnttab_end_foreign_access_references(
 				    sc->xbd_max_request_indirectpages,
 				    &cm->cm_indirectionrefs[0]);
 				contigfree(cm->cm_indirectionpages, PAGE_SIZE *
 				    sc->xbd_max_request_indirectpages,
 				    M_XENBLOCKFRONT);
 				cm->cm_indirectionpages = NULL;
 			}
 
 			bus_dmamap_destroy(sc->xbd_io_dmat, cm->cm_map);
 		}
 		free(sc->xbd_shadow, M_XENBLOCKFRONT);
 		sc->xbd_shadow = NULL;
 
 		bus_dma_tag_destroy(sc->xbd_io_dmat);
 		
 		xbd_initq_cm(sc, XBD_Q_FREE);
 		xbd_initq_cm(sc, XBD_Q_READY);
 		xbd_initq_cm(sc, XBD_Q_COMPLETE);
 	}
 		
 	xen_intr_unbind(&sc->xen_intr_handle);
 
 }
 
 /*--------------------------- State Change Handlers --------------------------*/
 static void
 xbd_initialize(struct xbd_softc *sc)
 {
 	const char *otherend_path;
 	const char *node_path;
 	uint32_t max_ring_page_order;
 	int error;
 
 	if (xenbus_get_state(sc->xbd_dev) != XenbusStateInitialising) {
 		/* Initialization has already been performed. */
 		return;
 	}
 
 	/*
 	 * Protocol defaults valid even if negotiation for a
 	 * setting fails.
 	 */
 	max_ring_page_order = 0;
 	sc->xbd_ring_pages = 1;
 
 	/*
 	 * Protocol negotiation.
 	 *
 	 * \note xs_gather() returns on the first encountered error, so
 	 *       we must use independent calls in order to guarantee
 	 *       we don't miss information in a sparsly populated back-end
 	 *       tree.
 	 *
 	 * \note xs_scanf() does not update variables for unmatched
 	 *	 fields.
 	 */
 	otherend_path = xenbus_get_otherend_path(sc->xbd_dev);
 	node_path = xenbus_get_node(sc->xbd_dev);
 
 	/* Support both backend schemes for relaying ring page limits. */
 	(void)xs_scanf(XST_NIL, otherend_path,
 	    "max-ring-page-order", NULL, "%" PRIu32,
 	    &max_ring_page_order);
 	sc->xbd_ring_pages = 1 << max_ring_page_order;
 	(void)xs_scanf(XST_NIL, otherend_path,
 	    "max-ring-pages", NULL, "%" PRIu32,
 	    &sc->xbd_ring_pages);
 	if (sc->xbd_ring_pages < 1)
 		sc->xbd_ring_pages = 1;
 
 	if (sc->xbd_ring_pages > XBD_MAX_RING_PAGES) {
 		device_printf(sc->xbd_dev,
 		    "Back-end specified ring-pages of %u "
 		    "limited to front-end limit of %u.\n",
 		    sc->xbd_ring_pages, XBD_MAX_RING_PAGES);
 		sc->xbd_ring_pages = XBD_MAX_RING_PAGES;
 	}
 
 	if (powerof2(sc->xbd_ring_pages) == 0) {
 		uint32_t new_page_limit;
 
 		new_page_limit = 0x01 << (fls(sc->xbd_ring_pages) - 1);
 		device_printf(sc->xbd_dev,
 		    "Back-end specified ring-pages of %u "
 		    "is not a power of 2. Limited to %u.\n",
 		    sc->xbd_ring_pages, new_page_limit);
 		sc->xbd_ring_pages = new_page_limit;
 	}
 
 	sc->xbd_max_requests =
 	    BLKIF_MAX_RING_REQUESTS(sc->xbd_ring_pages * PAGE_SIZE);
 	if (sc->xbd_max_requests > XBD_MAX_REQUESTS) {
 		device_printf(sc->xbd_dev,
 		    "Back-end specified max_requests of %u "
 		    "limited to front-end limit of %zu.\n",
 		    sc->xbd_max_requests, XBD_MAX_REQUESTS);
 		sc->xbd_max_requests = XBD_MAX_REQUESTS;
 	}
 
 	if (xbd_alloc_ring(sc) != 0)
 		return;
 
 	/* Support both backend schemes for relaying ring page limits. */
 	if (sc->xbd_ring_pages > 1) {
 		error = xs_printf(XST_NIL, node_path,
 		    "num-ring-pages","%u",
 		    sc->xbd_ring_pages);
 		if (error) {
 			xenbus_dev_fatal(sc->xbd_dev, error,
 			    "writing %s/num-ring-pages",
 			    node_path);
 			return;
 		}
 
 		error = xs_printf(XST_NIL, node_path,
 		    "ring-page-order", "%u",
 		    fls(sc->xbd_ring_pages) - 1);
 		if (error) {
 			xenbus_dev_fatal(sc->xbd_dev, error,
 			    "writing %s/ring-page-order",
 			    node_path);
 			return;
 		}
 	}
 
 	error = xs_printf(XST_NIL, node_path, "event-channel",
 	    "%u", xen_intr_port(sc->xen_intr_handle));
 	if (error) {
 		xenbus_dev_fatal(sc->xbd_dev, error,
 		    "writing %s/event-channel",
 		    node_path);
 		return;
 	}
 
 	error = xs_printf(XST_NIL, node_path, "protocol",
 	    "%s", XEN_IO_PROTO_ABI_NATIVE);
 	if (error) {
 		xenbus_dev_fatal(sc->xbd_dev, error,
 		    "writing %s/protocol",
 		    node_path);
 		return;
 	}
 
 	xenbus_set_state(sc->xbd_dev, XenbusStateInitialised);
 }
 
 /* 
  * Invoked when the backend is finally 'ready' (and has published
  * the details about the physical device - #sectors, size, etc). 
  */
 static void 
 xbd_connect(struct xbd_softc *sc)
 {
 	device_t dev = sc->xbd_dev;
 	unsigned long sectors, sector_size, phys_sector_size;
 	unsigned int binfo;
 	int err, feature_barrier, feature_flush;
 	int i, j;
 
 	if (sc->xbd_state == XBD_STATE_CONNECTED || 
 	    sc->xbd_state == XBD_STATE_SUSPENDED)
 		return;
 
 	DPRINTK("blkfront.c:connect:%s.\n", xenbus_get_otherend_path(dev));
 
 	err = xs_gather(XST_NIL, xenbus_get_otherend_path(dev),
 	    "sectors", "%lu", &sectors,
 	    "info", "%u", &binfo,
 	    "sector-size", "%lu", &sector_size,
 	    NULL);
 	if (err) {
 		xenbus_dev_fatal(dev, err,
 		    "reading backend fields at %s",
 		    xenbus_get_otherend_path(dev));
 		return;
 	}
 	if ((sectors == 0) || (sector_size == 0)) {
 		xenbus_dev_fatal(dev, 0,
 		    "invalid parameters from %s:"
 		    " sectors = %lu, sector_size = %lu",
 		    xenbus_get_otherend_path(dev),
 		    sectors, sector_size);
 		return;
 	}
 	err = xs_gather(XST_NIL, xenbus_get_otherend_path(dev),
 	     "physical-sector-size", "%lu", &phys_sector_size,
 	     NULL);
 	if (err || phys_sector_size <= sector_size)
 		phys_sector_size = 0;
 	err = xs_gather(XST_NIL, xenbus_get_otherend_path(dev),
 	     "feature-barrier", "%d", &feature_barrier,
 	     NULL);
 	if (err == 0 && feature_barrier != 0)
 		sc->xbd_flags |= XBDF_BARRIER;
 
 	err = xs_gather(XST_NIL, xenbus_get_otherend_path(dev),
 	     "feature-flush-cache", "%d", &feature_flush,
 	     NULL);
 	if (err == 0 && feature_flush != 0)
 		sc->xbd_flags |= XBDF_FLUSH;
 
 	err = xs_gather(XST_NIL, xenbus_get_otherend_path(dev),
 	    "feature-max-indirect-segments", "%" PRIu32,
 	    &sc->xbd_max_request_segments, NULL);
 	if ((err != 0) || (xbd_enable_indirect == 0))
 		sc->xbd_max_request_segments = 0;
 	if (sc->xbd_max_request_segments > XBD_MAX_INDIRECT_SEGMENTS)
 		sc->xbd_max_request_segments = XBD_MAX_INDIRECT_SEGMENTS;
 	if (sc->xbd_max_request_segments > XBD_SIZE_TO_SEGS(MAXPHYS))
 		sc->xbd_max_request_segments = XBD_SIZE_TO_SEGS(MAXPHYS);
 	sc->xbd_max_request_indirectpages =
 	    XBD_INDIRECT_SEGS_TO_PAGES(sc->xbd_max_request_segments);
 	if (sc->xbd_max_request_segments < BLKIF_MAX_SEGMENTS_PER_REQUEST)
 		sc->xbd_max_request_segments = BLKIF_MAX_SEGMENTS_PER_REQUEST;
 	sc->xbd_max_request_size =
 	    XBD_SEGS_TO_SIZE(sc->xbd_max_request_segments);
 
 	/* Allocate datastructures based on negotiated values. */
 	err = bus_dma_tag_create(
 	    bus_get_dma_tag(sc->xbd_dev),	/* parent */
 	    512, PAGE_SIZE,			/* algnmnt, boundary */
 	    BUS_SPACE_MAXADDR,			/* lowaddr */
 	    BUS_SPACE_MAXADDR,			/* highaddr */
 	    NULL, NULL,				/* filter, filterarg */
 	    sc->xbd_max_request_size,
 	    sc->xbd_max_request_segments,
 	    PAGE_SIZE,				/* maxsegsize */
 	    BUS_DMA_ALLOCNOW,			/* flags */
 	    busdma_lock_mutex,			/* lockfunc */
 	    &sc->xbd_io_lock,			/* lockarg */
 	    &sc->xbd_io_dmat);
 	if (err != 0) {
 		xenbus_dev_fatal(sc->xbd_dev, err,
 		    "Cannot allocate parent DMA tag\n");
 		return;
 	}
 
 	/* Per-transaction data allocation. */
 	sc->xbd_shadow = malloc(sizeof(*sc->xbd_shadow) * sc->xbd_max_requests,
 	    M_XENBLOCKFRONT, M_NOWAIT|M_ZERO);
 	if (sc->xbd_shadow == NULL) {
 		bus_dma_tag_destroy(sc->xbd_io_dmat);
 		xenbus_dev_fatal(sc->xbd_dev, ENOMEM,
 		    "Cannot allocate request structures\n");
 		return;
 	}
 
 	for (i = 0; i < sc->xbd_max_requests; i++) {
 		struct xbd_command *cm;
 		void * indirectpages;
 
 		cm = &sc->xbd_shadow[i];
 		cm->cm_sg_refs = malloc(
 		    sizeof(grant_ref_t) * sc->xbd_max_request_segments,
 		    M_XENBLOCKFRONT, M_NOWAIT);
 		if (cm->cm_sg_refs == NULL)
 			break;
 		cm->cm_id = i;
 		cm->cm_flags = XBDCF_INITIALIZER;
 		cm->cm_sc = sc;
 		if (bus_dmamap_create(sc->xbd_io_dmat, 0, &cm->cm_map) != 0)
 			break;
 		if (sc->xbd_max_request_indirectpages > 0) {
 			indirectpages = contigmalloc(
 			    PAGE_SIZE * sc->xbd_max_request_indirectpages,
 			    M_XENBLOCKFRONT, M_ZERO, 0, ~0, PAGE_SIZE, 0);
 		} else {
 			indirectpages = NULL;
 		}
 		for (j = 0; j < sc->xbd_max_request_indirectpages; j++) {
 			if (gnttab_grant_foreign_access(
 			    xenbus_get_otherend_id(sc->xbd_dev),
 			    (vtophys(indirectpages) >> PAGE_SHIFT) + j,
 			    1 /* grant read-only access */,
 			    &cm->cm_indirectionrefs[j]))
 				break;
 		}
 		if (j < sc->xbd_max_request_indirectpages)
 			break;
 		cm->cm_indirectionpages = indirectpages;
 		xbd_free_command(cm);
 	}
 
 	if (sc->xbd_disk == NULL) {
 		device_printf(dev, "%juMB <%s> at %s",
 		    (uintmax_t) sectors / (1048576 / sector_size),
 		    device_get_desc(dev),
 		    xenbus_get_node(dev));
 		bus_print_child_footer(device_get_parent(dev), dev);
 
 		xbd_instance_create(sc, sectors, sc->xbd_vdevice, binfo,
 		    sector_size, phys_sector_size);
 	}
 
 	(void)xenbus_set_state(dev, XenbusStateConnected); 
 
 	/* Kick pending requests. */
 	mtx_lock(&sc->xbd_io_lock);
 	sc->xbd_state = XBD_STATE_CONNECTED;
 	xbd_startio(sc);
 	sc->xbd_flags |= XBDF_READY;
 	mtx_unlock(&sc->xbd_io_lock);
 }
 
 /**
  * Handle the change of state of the backend to Closing.  We must delete our
  * device-layer structures now, to ensure that writes are flushed through to
  * the backend.  Once this is done, we can switch to Closed in
  * acknowledgement.
  */
 static void
 xbd_closing(device_t dev)
 {
 	struct xbd_softc *sc = device_get_softc(dev);
 
 	xenbus_set_state(dev, XenbusStateClosing);
 
 	DPRINTK("xbd_closing: %s removed\n", xenbus_get_node(dev));
 
 	if (sc->xbd_disk != NULL) {
 		disk_destroy(sc->xbd_disk);
 		sc->xbd_disk = NULL;
 	}
 
 	xenbus_set_state(dev, XenbusStateClosed); 
 }
 
 /*---------------------------- NewBus Entrypoints ----------------------------*/
 static int
 xbd_probe(device_t dev)
 {
 	if (strcmp(xenbus_get_type(dev), "vbd") != 0)
 		return (ENXIO);
 
 	if (xen_hvm_domain() && xen_disable_pv_disks != 0)
 		return (ENXIO);
 
 	if (xen_hvm_domain()) {
 		int error;
 		char *type;
 
 		/*
 		 * When running in an HVM domain, IDE disk emulation is
 		 * disabled early in boot so that native drivers will
 		 * not see emulated hardware.  However, CDROM device
 		 * emulation cannot be disabled.
 		 *
 		 * Through use of FreeBSD's vm_guest and xen_hvm_domain()
 		 * APIs, we could modify the native CDROM driver to fail its
 		 * probe when running under Xen.  Unfortunatlely, the PV
 		 * CDROM support in XenServer (up through at least version
 		 * 6.2) isn't functional, so we instead rely on the emulated
 		 * CDROM instance, and fail to attach the PV one here in
 		 * the blkfront driver.
 		 */
 		error = xs_read(XST_NIL, xenbus_get_node(dev),
 		    "device-type", NULL, (void **) &type);
 		if (error)
 			return (ENXIO);
 
 		if (strncmp(type, "cdrom", 5) == 0) {
 			free(type, M_XENSTORE);
 			return (ENXIO);
 		}
 		free(type, M_XENSTORE);
 	}
 
 	device_set_desc(dev, "Virtual Block Device");
 	device_quiet(dev);
 	return (0);
 }
 
 /*
  * Setup supplies the backend dir, virtual device.  We place an event
  * channel and shared frame entries.  We watch backend to wait if it's
  * ok.
  */
 static int
 xbd_attach(device_t dev)
 {
 	struct xbd_softc *sc;
 	const char *name;
 	uint32_t vdevice;
 	int error;
 	int i;
 	int unit;
 
 	/* FIXME: Use dynamic device id if this is not set. */
 	error = xs_scanf(XST_NIL, xenbus_get_node(dev),
 	    "virtual-device", NULL, "%" PRIu32, &vdevice);
 	if (error)
 		error = xs_scanf(XST_NIL, xenbus_get_node(dev),
 		    "virtual-device-ext", NULL, "%" PRIu32, &vdevice);
 	if (error) {
 		xenbus_dev_fatal(dev, error, "reading virtual-device");
 		device_printf(dev, "Couldn't determine virtual device.\n");
 		return (error);
 	}
 
 	xbd_vdevice_to_unit(vdevice, &unit, &name);
 	if (!strcmp(name, "xbd"))
 		device_set_unit(dev, unit);
 
 	sc = device_get_softc(dev);
 	mtx_init(&sc->xbd_io_lock, "blkfront i/o lock", NULL, MTX_DEF);
 	xbd_initqs(sc);
 	for (i = 0; i < XBD_MAX_RING_PAGES; i++)
 		sc->xbd_ring_ref[i] = GRANT_REF_INVALID;
 
 	sc->xbd_dev = dev;
 	sc->xbd_vdevice = vdevice;
 	sc->xbd_state = XBD_STATE_DISCONNECTED;
 
 	xbd_setup_sysctl(sc);
 
 	/* Wait for backend device to publish its protocol capabilities. */
 	xenbus_set_state(dev, XenbusStateInitialising);
 
 	return (0);
 }
 
 static int
 xbd_detach(device_t dev)
 {
 	struct xbd_softc *sc = device_get_softc(dev);
 
 	DPRINTK("%s: %s removed\n", __func__, xenbus_get_node(dev));
 
 	xbd_free(sc);
 	mtx_destroy(&sc->xbd_io_lock);
 
 	return 0;
 }
 
 static int
 xbd_suspend(device_t dev)
 {
 	struct xbd_softc *sc = device_get_softc(dev);
 	int retval;
 	int saved_state;
 
 	/* Prevent new requests being issued until we fix things up. */
 	mtx_lock(&sc->xbd_io_lock);
 	saved_state = sc->xbd_state;
 	sc->xbd_state = XBD_STATE_SUSPENDED;
 
 	/* Wait for outstanding I/O to drain. */
 	retval = 0;
 	while (xbd_queue_length(sc, XBD_Q_BUSY) != 0) {
 		if (msleep(&sc->xbd_cm_q[XBD_Q_BUSY], &sc->xbd_io_lock,
 		    PRIBIO, "blkf_susp", 30 * hz) == EWOULDBLOCK) {
 			retval = EBUSY;
 			break;
 		}
 	}
 	mtx_unlock(&sc->xbd_io_lock);
 
 	if (retval != 0)
 		sc->xbd_state = saved_state;
 
 	return (retval);
 }
 
 static int
 xbd_resume(device_t dev)
 {
 	struct xbd_softc *sc = device_get_softc(dev);
 
+	if (xen_suspend_cancelled) {
+		sc->xbd_state = XBD_STATE_CONNECTED;
+		return (0);
+	}
+
 	DPRINTK("xbd_resume: %s\n", xenbus_get_node(dev));
 
 	xbd_free(sc);
 	xbd_initialize(sc);
 	return (0);
 }
 
 /**
  * Callback received when the backend's state changes.
  */
 static void
 xbd_backend_changed(device_t dev, XenbusState backend_state)
 {
 	struct xbd_softc *sc = device_get_softc(dev);
 
 	DPRINTK("backend_state=%d\n", backend_state);
 
 	switch (backend_state) {
 	case XenbusStateUnknown:
 	case XenbusStateInitialising:
 	case XenbusStateReconfigured:
 	case XenbusStateReconfiguring:
 	case XenbusStateClosed:
 		break;
 
 	case XenbusStateInitWait:
 	case XenbusStateInitialised:
 		xbd_initialize(sc);
 		break;
 
 	case XenbusStateConnected:
 		xbd_initialize(sc);
 		xbd_connect(sc);
 		break;
 
 	case XenbusStateClosing:
 		if (sc->xbd_users > 0)
 			xenbus_dev_error(dev, -EBUSY,
 			    "Device in use; refusing to close");
 		else
 			xbd_closing(dev);
 		break;	
 	}
 }
 
 /*---------------------------- NewBus Registration ---------------------------*/
 static device_method_t xbd_methods[] = { 
 	/* Device interface */ 
 	DEVMETHOD(device_probe,         xbd_probe), 
 	DEVMETHOD(device_attach,        xbd_attach), 
 	DEVMETHOD(device_detach,        xbd_detach), 
 	DEVMETHOD(device_shutdown,      bus_generic_shutdown), 
 	DEVMETHOD(device_suspend,       xbd_suspend), 
 	DEVMETHOD(device_resume,        xbd_resume), 
  
 	/* Xenbus interface */
 	DEVMETHOD(xenbus_otherend_changed, xbd_backend_changed),
 
 	{ 0, 0 } 
 }; 
 
 static driver_t xbd_driver = { 
 	"xbd", 
 	xbd_methods, 
 	sizeof(struct xbd_softc),                      
 }; 
 devclass_t xbd_devclass; 
  
 DRIVER_MODULE(xbd, xenbusb_front, xbd_driver, xbd_devclass, 0, 0); 
Index: head/sys/dev/xen/control/control.c
===================================================================
--- head/sys/dev/xen/control/control.c	(revision 314839)
+++ head/sys/dev/xen/control/control.c	(revision 314840)
@@ -1,473 +1,477 @@
 /*-
  * Copyright (c) 2010 Justin T. Gibbs, Spectra Logic Corporation
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions, and the following disclaimer,
  *    without modification.
  * 2. Redistributions in binary form must reproduce at minimum a disclaimer
  *    substantially similar to the "NO WARRANTY" disclaimer below
  *    ("Disclaimer") and any redistribution must be conditioned upon
  *    including a substantially similar Disclaimer requirement for further
  *    binary redistribution.
  *
  * NO WARRANTY
  * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
  * "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
  * LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTIBILITY AND FITNESS FOR
  * A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
  * HOLDERS OR CONTRIBUTORS BE LIABLE FOR SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
  * STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING
  * IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
  * POSSIBILITY OF SUCH DAMAGES.
  */
 
 /*-
  * PV suspend/resume support:
  *
  * Copyright (c) 2004 Christian Limpach.
  * Copyright (c) 2004-2006,2008 Kip Macy
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  * 3. All advertising materials mentioning features or use of this software
  *    must display the following acknowledgement:
  *      This product includes software developed by Christian Limpach.
  * 4. The name of the author may not be used to endorse or promote products
  *    derived from this software without specific prior written permission.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR
  * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
  * OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
  * IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT, INDIRECT,
  * INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT
  * NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
  * DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
  * THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
  * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF
  * THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
  */
 
 /*-
  * HVM suspend/resume support:
  *
  * Copyright (c) 2008 Citrix Systems, Inc.
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 /**
  * \file control.c
  *
  * \brief Device driver to repond to control domain events that impact
  *        this VM.
  */
 
 #include <sys/param.h>
 #include <sys/systm.h>
 #include <sys/kernel.h>
 #include <sys/malloc.h>
 
 #include <sys/bio.h>
 #include <sys/bus.h>
 #include <sys/conf.h>
 #include <sys/disk.h>
 #include <sys/fcntl.h>
 #include <sys/filedesc.h>
 #include <sys/kdb.h>
 #include <sys/module.h>
 #include <sys/namei.h>
 #include <sys/proc.h>
 #include <sys/reboot.h>
 #include <sys/rman.h>
 #include <sys/sched.h>
 #include <sys/taskqueue.h>
 #include <sys/types.h>
 #include <sys/vnode.h>
 #include <sys/sched.h>
 #include <sys/smp.h>
 #include <sys/eventhandler.h>
 #include <sys/timetc.h>
 
 #include <geom/geom.h>
 
 #include <machine/_inttypes.h>
 #include <machine/intr_machdep.h>
 
 #include <x86/apicvar.h>
 
 #include <vm/vm.h>
 #include <vm/vm_extern.h>
 #include <vm/vm_kern.h>
 
 #include <xen/xen-os.h>
 #include <xen/blkif.h>
 #include <xen/evtchn.h>
 #include <xen/gnttab.h>
 #include <xen/xen_intr.h>
 
 #include <xen/hvm.h>
 
 #include <xen/interface/event_channel.h>
 #include <xen/interface/grant_table.h>
 
 #include <xen/xenbus/xenbusvar.h>
 
+bool xen_suspend_cancelled;
 /*--------------------------- Forward Declarations --------------------------*/
 /** Function signature for shutdown event handlers. */
 typedef	void (xctrl_shutdown_handler_t)(void);
 
 static xctrl_shutdown_handler_t xctrl_poweroff;
 static xctrl_shutdown_handler_t xctrl_reboot;
 static xctrl_shutdown_handler_t xctrl_suspend;
 static xctrl_shutdown_handler_t xctrl_crash;
 
 /*-------------------------- Private Data Structures -------------------------*/
 /** Element type for lookup table of event name to handler. */
 struct xctrl_shutdown_reason {
 	const char		 *name;
 	xctrl_shutdown_handler_t *handler;
 };
 
 /** Lookup table for shutdown event name to handler. */
 static const struct xctrl_shutdown_reason xctrl_shutdown_reasons[] = {
 	{ "poweroff", xctrl_poweroff },
 	{ "reboot",   xctrl_reboot   },
 	{ "suspend",  xctrl_suspend  },
 	{ "crash",    xctrl_crash    },
 	{ "halt",     xctrl_poweroff },
 };
 
 struct xctrl_softc {
 	struct xs_watch    xctrl_watch;	
 };
 
 /*------------------------------ Event Handlers ------------------------------*/
 static void
 xctrl_poweroff()
 {
 	shutdown_nice(RB_POWEROFF|RB_HALT);
 }
 
 static void
 xctrl_reboot()
 {
 	shutdown_nice(0);
 }
 
 static void
 xctrl_suspend()
 {
 #ifdef SMP
 	cpuset_t cpu_suspend_map;
 #endif
-	int suspend_cancelled;
 
 	EVENTHANDLER_INVOKE(power_suspend_early);
 	stop_all_proc();
 	EVENTHANDLER_INVOKE(power_suspend);
 
 #ifdef EARLY_AP_STARTUP
 	MPASS(mp_ncpus == 1 || smp_started);
 	thread_lock(curthread);
 	sched_bind(curthread, 0);
 	thread_unlock(curthread);
 #else
 	if (smp_started) {
 		thread_lock(curthread);
 		sched_bind(curthread, 0);
 		thread_unlock(curthread);
 	}
 #endif
 	KASSERT((PCPU_GET(cpuid) == 0), ("Not running on CPU#0"));
 
 	/*
 	 * Clear our XenStore node so the toolstack knows we are
 	 * responding to the suspend request.
 	 */
 	xs_write(XST_NIL, "control", "shutdown", "");
 
 	/*
 	 * Be sure to hold Giant across DEVICE_SUSPEND/RESUME since non-MPSAFE
 	 * drivers need this.
 	 */
 	mtx_lock(&Giant);
 	if (DEVICE_SUSPEND(root_bus) != 0) {
 		mtx_unlock(&Giant);
 		printf("%s: device_suspend failed\n", __func__);
 		return;
 	}
 
 #ifdef SMP
 #ifdef EARLY_AP_STARTUP
 	/*
 	 * Suspend other CPUs. This prevents IPIs while we
 	 * are resuming, and will allow us to reset per-cpu
 	 * vcpu_info on resume.
 	 */
 	cpu_suspend_map = all_cpus;
 	CPU_CLR(PCPU_GET(cpuid), &cpu_suspend_map);
 	if (!CPU_EMPTY(&cpu_suspend_map))
 		suspend_cpus(cpu_suspend_map);
 #else
 	CPU_ZERO(&cpu_suspend_map);	/* silence gcc */
 	if (smp_started) {
 		/*
 		 * Suspend other CPUs. This prevents IPIs while we
 		 * are resuming, and will allow us to reset per-cpu
 		 * vcpu_info on resume.
 		 */
 		cpu_suspend_map = all_cpus;
 		CPU_CLR(PCPU_GET(cpuid), &cpu_suspend_map);
 		if (!CPU_EMPTY(&cpu_suspend_map))
 			suspend_cpus(cpu_suspend_map);
 	}
 #endif
 #endif
 
 	/*
 	 * Prevent any races with evtchn_interrupt() handler.
 	 */
 	disable_intr();
 	intr_suspend();
 	xen_hvm_suspend();
 
-	suspend_cancelled = HYPERVISOR_suspend(0);
+	xen_suspend_cancelled = !!HYPERVISOR_suspend(0);
 
-	xen_hvm_resume(suspend_cancelled != 0);
-	intr_resume(suspend_cancelled != 0);
+	if (!xen_suspend_cancelled) {
+		xen_hvm_resume(false);
+	}
+	intr_resume(xen_suspend_cancelled != 0);
 	enable_intr();
 
 	/*
 	 * Reset grant table info.
 	 */
-	gnttab_resume(NULL);
+	if (!xen_suspend_cancelled) {
+		gnttab_resume(NULL);
+	}
 
 #ifdef SMP
 	if (!CPU_EMPTY(&cpu_suspend_map)) {
 		/*
 		 * Now that event channels have been initialized,
 		 * resume CPUs.
 		 */
 		resume_cpus(cpu_suspend_map);
 		/* Send an IPI_BITMAP in case there are pending bitmap IPIs. */
 		lapic_ipi_vectored(IPI_BITMAP_VECTOR, APIC_IPI_DEST_ALL);
 	}
 #endif
 
 	/*
 	 * FreeBSD really needs to add DEVICE_SUSPEND_CANCEL or
 	 * similar.
 	 */
 	DEVICE_RESUME(root_bus);
 	mtx_unlock(&Giant);
 
 	/*
 	 * Warm up timecounter again and reset system clock.
 	 */
 	timecounter->tc_get_timecount(timecounter);
 	timecounter->tc_get_timecount(timecounter);
 	inittodr(time_second);
 
 #ifdef EARLY_AP_STARTUP
 	thread_lock(curthread);
 	sched_unbind(curthread);
 	thread_unlock(curthread);
 #else
 	if (smp_started) {
 		thread_lock(curthread);
 		sched_unbind(curthread);
 		thread_unlock(curthread);
 	}
 #endif
 
 	resume_all_proc();
 
 	EVENTHANDLER_INVOKE(power_resume);
 
 	if (bootverbose)
 		printf("System resumed after suspension\n");
 
 }
 
 static void
 xctrl_crash()
 {
 	panic("Xen directed crash");
 }
 
 static void
 xen_pv_shutdown_final(void *arg, int howto)
 {
 	/*
 	 * Inform the hypervisor that shutdown is complete.
 	 * This is not necessary in HVM domains since Xen
 	 * emulates ACPI in that mode and FreeBSD's ACPI
 	 * support will request this transition.
 	 */
 	if (howto & (RB_HALT | RB_POWEROFF))
 		HYPERVISOR_shutdown(SHUTDOWN_poweroff);
 	else
 		HYPERVISOR_shutdown(SHUTDOWN_reboot);
 }
 
 /*------------------------------ Event Reception -----------------------------*/
 static void
 xctrl_on_watch_event(struct xs_watch *watch, const char **vec, unsigned int len)
 {
 	const struct xctrl_shutdown_reason *reason;
 	const struct xctrl_shutdown_reason *last_reason;
 	char *result;
 	int   error;
 	int   result_len;
 	
 	error = xs_read(XST_NIL, "control", "shutdown",
 			&result_len, (void **)&result);
 	if (error != 0)
 		return;
 
 	reason = xctrl_shutdown_reasons;
 	last_reason = reason + nitems(xctrl_shutdown_reasons);
 	while (reason < last_reason) {
 
 		if (!strcmp(result, reason->name)) {
 			reason->handler();
 			break;
 		}
 		reason++;
 	}
 
 	free(result, M_XENSTORE);
 }
 
 /*------------------ Private Device Attachment Functions  --------------------*/
 /**
  * \brief Identify instances of this device type in the system.
  *
  * \param driver  The driver performing this identify action.
  * \param parent  The NewBus parent device for any devices this method adds.
  */
 static void
 xctrl_identify(driver_t *driver __unused, device_t parent)
 {
 	/*
 	 * A single device instance for our driver is always present
 	 * in a system operating under Xen.
 	 */
 	BUS_ADD_CHILD(parent, 0, driver->name, 0);
 }
 
 /**
  * \brief Probe for the existence of the Xen Control device
  *
  * \param dev  NewBus device_t for this Xen control instance.
  *
  * \return  Always returns 0 indicating success.
  */
 static int 
 xctrl_probe(device_t dev)
 {
 	device_set_desc(dev, "Xen Control Device");
 
 	return (BUS_PROBE_NOWILDCARD);
 }
 
 /**
  * \brief Attach the Xen control device.
  *
  * \param dev  NewBus device_t for this Xen control instance.
  *
  * \return  On success, 0. Otherwise an errno value indicating the
  *          type of failure.
  */
 static int
 xctrl_attach(device_t dev)
 {
 	struct xctrl_softc *xctrl;
 
 	xctrl = device_get_softc(dev);
 
 	/* Activate watch */
 	xctrl->xctrl_watch.node = "control/shutdown";
 	xctrl->xctrl_watch.callback = xctrl_on_watch_event;
 	xctrl->xctrl_watch.callback_data = (uintptr_t)xctrl;
 	xs_register_watch(&xctrl->xctrl_watch);
 
 	if (xen_pv_domain())
 		EVENTHANDLER_REGISTER(shutdown_final, xen_pv_shutdown_final, NULL,
 		                      SHUTDOWN_PRI_LAST);
 
 	return (0);
 }
 
 /**
  * \brief Detach the Xen control device.
  *
  * \param dev  NewBus device_t for this Xen control device instance.
  *
  * \return  On success, 0. Otherwise an errno value indicating the
  *          type of failure.
  */
 static int
 xctrl_detach(device_t dev)
 {
 	struct xctrl_softc *xctrl;
 
 	xctrl = device_get_softc(dev);
 
 	/* Release watch */
 	xs_unregister_watch(&xctrl->xctrl_watch);
 
 	return (0);
 }
 
 /*-------------------- Private Device Attachment Data  -----------------------*/
 static device_method_t xctrl_methods[] = { 
 	/* Device interface */ 
 	DEVMETHOD(device_identify,	xctrl_identify),
 	DEVMETHOD(device_probe,         xctrl_probe), 
 	DEVMETHOD(device_attach,        xctrl_attach), 
 	DEVMETHOD(device_detach,        xctrl_detach), 
  
 	DEVMETHOD_END
 }; 
 
 DEFINE_CLASS_0(xctrl, xctrl_driver, xctrl_methods, sizeof(struct xctrl_softc));
 devclass_t xctrl_devclass; 
  
 DRIVER_MODULE(xctrl, xenstore, xctrl_driver, xctrl_devclass, NULL, NULL);
Index: head/sys/dev/xen/netfront/netfront.c
===================================================================
--- head/sys/dev/xen/netfront/netfront.c	(revision 314839)
+++ head/sys/dev/xen/netfront/netfront.c	(revision 314840)
@@ -1,2326 +1,2340 @@
 /*-
  * Copyright (c) 2004-2006 Kip Macy
  * Copyright (c) 2015 Wei Liu <wei.liu2@citrix.com>
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include "opt_inet.h"
 #include "opt_inet6.h"
 
 #include <sys/param.h>
 #include <sys/sockio.h>
 #include <sys/limits.h>
 #include <sys/mbuf.h>
 #include <sys/malloc.h>
 #include <sys/module.h>
 #include <sys/kernel.h>
 #include <sys/socket.h>
 #include <sys/sysctl.h>
 #include <sys/taskqueue.h>
 
 #include <net/if.h>
 #include <net/if_var.h>
 #include <net/if_arp.h>
 #include <net/ethernet.h>
 #include <net/if_media.h>
 #include <net/bpf.h>
 #include <net/if_types.h>
 
 #include <netinet/in.h>
 #include <netinet/ip.h>
 #include <netinet/if_ether.h>
 #include <netinet/tcp.h>
 #include <netinet/tcp_lro.h>
 
 #include <vm/vm.h>
 #include <vm/pmap.h>
 
 #include <sys/bus.h>
 
 #include <xen/xen-os.h>
 #include <xen/hypervisor.h>
 #include <xen/xen_intr.h>
 #include <xen/gnttab.h>
 #include <xen/interface/memory.h>
 #include <xen/interface/io/netif.h>
 #include <xen/xenbus/xenbusvar.h>
 
 #include "xenbus_if.h"
 
 /* Features supported by all backends.  TSO and LRO can be negotiated */
 #define XN_CSUM_FEATURES	(CSUM_TCP | CSUM_UDP)
 
 #define NET_TX_RING_SIZE __RING_SIZE((netif_tx_sring_t *)0, PAGE_SIZE)
 #define NET_RX_RING_SIZE __RING_SIZE((netif_rx_sring_t *)0, PAGE_SIZE)
 
 #define NET_RX_SLOTS_MIN (XEN_NETIF_NR_SLOTS_MIN + 1)
 
 /*
  * Should the driver do LRO on the RX end
  *  this can be toggled on the fly, but the
  *  interface must be reset (down/up) for it
  *  to take effect.
  */
 static int xn_enable_lro = 1;
 TUNABLE_INT("hw.xn.enable_lro", &xn_enable_lro);
 
 /*
  * Number of pairs of queues.
  */
 static unsigned long xn_num_queues = 4;
 TUNABLE_ULONG("hw.xn.num_queues", &xn_num_queues);
 
 /**
  * \brief The maximum allowed data fragments in a single transmit
  *        request.
  *
  * This limit is imposed by the backend driver.  We assume here that
  * we are dealing with a Linux driver domain and have set our limit
  * to mirror the Linux MAX_SKB_FRAGS constant.
  */
 #define	MAX_TX_REQ_FRAGS (65536 / PAGE_SIZE + 2)
 
 #define RX_COPY_THRESHOLD 256
 
 #define net_ratelimit() 0
 
 struct netfront_rxq;
 struct netfront_txq;
 struct netfront_info;
 struct netfront_rx_info;
 
 static void xn_txeof(struct netfront_txq *);
 static void xn_rxeof(struct netfront_rxq *);
 static void xn_alloc_rx_buffers(struct netfront_rxq *);
 static void xn_alloc_rx_buffers_callout(void *arg);
 
 static void xn_release_rx_bufs(struct netfront_rxq *);
 static void xn_release_tx_bufs(struct netfront_txq *);
 
 static void xn_rxq_intr(struct netfront_rxq *);
 static void xn_txq_intr(struct netfront_txq *);
 static void xn_intr(void *);
 static inline int xn_count_frags(struct mbuf *m);
 static int xn_assemble_tx_request(struct netfront_txq *, struct mbuf *);
 static int xn_ioctl(struct ifnet *, u_long, caddr_t);
 static void xn_ifinit_locked(struct netfront_info *);
 static void xn_ifinit(void *);
 static void xn_stop(struct netfront_info *);
 static void xn_query_features(struct netfront_info *np);
 static int xn_configure_features(struct netfront_info *np);
 static void netif_free(struct netfront_info *info);
 static int netfront_detach(device_t dev);
 
 static int xn_txq_mq_start_locked(struct netfront_txq *, struct mbuf *);
 static int xn_txq_mq_start(struct ifnet *, struct mbuf *);
 
 static int talk_to_backend(device_t dev, struct netfront_info *info);
 static int create_netdev(device_t dev);
 static void netif_disconnect_backend(struct netfront_info *info);
 static int setup_device(device_t dev, struct netfront_info *info,
     unsigned long);
 static int xn_ifmedia_upd(struct ifnet *ifp);
 static void xn_ifmedia_sts(struct ifnet *ifp, struct ifmediareq *ifmr);
 
 static int xn_connect(struct netfront_info *);
 static void xn_kick_rings(struct netfront_info *);
 
 static int xn_get_responses(struct netfront_rxq *,
     struct netfront_rx_info *, RING_IDX, RING_IDX *,
     struct mbuf **);
 
 #define virt_to_mfn(x) (vtophys(x) >> PAGE_SHIFT)
 
 #define INVALID_P2M_ENTRY (~0UL)
 #define XN_QUEUE_NAME_LEN  8	/* xn{t,r}x_%u, allow for two digits */
 struct netfront_rxq {
 	struct netfront_info 	*info;
 	u_int			id;
 	char			name[XN_QUEUE_NAME_LEN];
 	struct mtx		lock;
 
 	int			ring_ref;
 	netif_rx_front_ring_t 	ring;
 	xen_intr_handle_t	xen_intr_handle;
 
 	grant_ref_t 		gref_head;
 	grant_ref_t 		grant_ref[NET_RX_RING_SIZE + 1];
 
 	struct mbuf		*mbufs[NET_RX_RING_SIZE + 1];
 
 	struct lro_ctrl		lro;
 
 	struct callout		rx_refill;
 };
 
 struct netfront_txq {
 	struct netfront_info 	*info;
 	u_int 			id;
 	char			name[XN_QUEUE_NAME_LEN];
 	struct mtx		lock;
 
 	int			ring_ref;
 	netif_tx_front_ring_t	ring;
 	xen_intr_handle_t 	xen_intr_handle;
 
 	grant_ref_t		gref_head;
 	grant_ref_t		grant_ref[NET_TX_RING_SIZE + 1];
 
 	struct mbuf		*mbufs[NET_TX_RING_SIZE + 1];
 	int			mbufs_cnt;
 	struct buf_ring		*br;
 
 	struct taskqueue 	*tq;
 	struct task       	defrtask;
 
 	bool			full;
 };
 
 struct netfront_info {
 	struct ifnet 		*xn_ifp;
 
 	struct mtx   		sc_lock;
 
 	u_int  num_queues;
 	struct netfront_rxq 	*rxq;
 	struct netfront_txq 	*txq;
 
 	u_int			carrier;
 	u_int			maxfrags;
 
 	device_t		xbdev;
 	uint8_t			mac[ETHER_ADDR_LEN];
 
 	int			xn_if_flags;
 
 	struct ifmedia		sc_media;
 
 	bool			xn_reset;
 };
 
 struct netfront_rx_info {
 	struct netif_rx_response rx;
 	struct netif_extra_info extras[XEN_NETIF_EXTRA_TYPE_MAX - 1];
 };
 
 #define XN_RX_LOCK(_q)         mtx_lock(&(_q)->lock)
 #define XN_RX_UNLOCK(_q)       mtx_unlock(&(_q)->lock)
 
 #define XN_TX_LOCK(_q)         mtx_lock(&(_q)->lock)
 #define XN_TX_TRYLOCK(_q)      mtx_trylock(&(_q)->lock)
 #define XN_TX_UNLOCK(_q)       mtx_unlock(&(_q)->lock)
 
 #define XN_LOCK(_sc)           mtx_lock(&(_sc)->sc_lock);
 #define XN_UNLOCK(_sc)         mtx_unlock(&(_sc)->sc_lock);
 
 #define XN_LOCK_ASSERT(_sc)    mtx_assert(&(_sc)->sc_lock, MA_OWNED);
 #define XN_RX_LOCK_ASSERT(_q)  mtx_assert(&(_q)->lock, MA_OWNED);
 #define XN_TX_LOCK_ASSERT(_q)  mtx_assert(&(_q)->lock, MA_OWNED);
 
 #define netfront_carrier_on(netif)	((netif)->carrier = 1)
 #define netfront_carrier_off(netif)	((netif)->carrier = 0)
 #define netfront_carrier_ok(netif)	((netif)->carrier)
 
 /* Access macros for acquiring freeing slots in xn_free_{tx,rx}_idxs[]. */
 
 static inline void
 add_id_to_freelist(struct mbuf **list, uintptr_t id)
 {
 
 	KASSERT(id != 0,
 		("%s: the head item (0) must always be free.", __func__));
 	list[id] = list[0];
 	list[0]  = (struct mbuf *)id;
 }
 
 static inline unsigned short
 get_id_from_freelist(struct mbuf **list)
 {
 	uintptr_t id;
 
 	id = (uintptr_t)list[0];
 	KASSERT(id != 0,
 		("%s: the head item (0) must always remain free.", __func__));
 	list[0] = list[id];
 	return (id);
 }
 
 static inline int
 xn_rxidx(RING_IDX idx)
 {
 
 	return idx & (NET_RX_RING_SIZE - 1);
 }
 
 static inline struct mbuf *
 xn_get_rx_mbuf(struct netfront_rxq *rxq, RING_IDX ri)
 {
 	int i;
 	struct mbuf *m;
 
 	i = xn_rxidx(ri);
 	m = rxq->mbufs[i];
 	rxq->mbufs[i] = NULL;
 	return (m);
 }
 
 static inline grant_ref_t
 xn_get_rx_ref(struct netfront_rxq *rxq, RING_IDX ri)
 {
 	int i = xn_rxidx(ri);
 	grant_ref_t ref = rxq->grant_ref[i];
 
 	KASSERT(ref != GRANT_REF_INVALID, ("Invalid grant reference!\n"));
 	rxq->grant_ref[i] = GRANT_REF_INVALID;
 	return (ref);
 }
 
 #define IPRINTK(fmt, args...) \
     printf("[XEN] " fmt, ##args)
 #ifdef INVARIANTS
 #define WPRINTK(fmt, args...) \
     printf("[XEN] " fmt, ##args)
 #else
 #define WPRINTK(fmt, args...)
 #endif
 #ifdef DEBUG
 #define DPRINTK(fmt, args...) \
     printf("[XEN] %s: " fmt, __func__, ##args)
 #else
 #define DPRINTK(fmt, args...)
 #endif
 
 /**
  * Read the 'mac' node at the given device's node in the store, and parse that
  * as colon-separated octets, placing result the given mac array.  mac must be
  * a preallocated array of length ETH_ALEN (as declared in linux/if_ether.h).
  * Return 0 on success, or errno on error.
  */
 static int
 xen_net_read_mac(device_t dev, uint8_t mac[])
 {
 	int error, i;
 	char *s, *e, *macstr;
 	const char *path;
 
 	path = xenbus_get_node(dev);
 	error = xs_read(XST_NIL, path, "mac", NULL, (void **) &macstr);
 	if (error == ENOENT) {
 		/*
 		 * Deal with missing mac XenStore nodes on devices with
 		 * HVM emulation (the 'ioemu' configuration attribute)
 		 * enabled.
 		 *
 		 * The HVM emulator may execute in a stub device model
 		 * domain which lacks the permission, only given to Dom0,
 		 * to update the guest's XenStore tree.  For this reason,
 		 * the HVM emulator doesn't even attempt to write the
 		 * front-side mac node, even when operating in Dom0.
 		 * However, there should always be a mac listed in the
 		 * backend tree.  Fallback to this version if our query
 		 * of the front side XenStore location doesn't find
 		 * anything.
 		 */
 		path = xenbus_get_otherend_path(dev);
 		error = xs_read(XST_NIL, path, "mac", NULL, (void **) &macstr);
 	}
 	if (error != 0) {
 		xenbus_dev_fatal(dev, error, "parsing %s/mac", path);
 		return (error);
 	}
 
 	s = macstr;
 	for (i = 0; i < ETHER_ADDR_LEN; i++) {
 		mac[i] = strtoul(s, &e, 16);
 		if (s == e || (e[0] != ':' && e[0] != 0)) {
 			free(macstr, M_XENBUS);
 			return (ENOENT);
 		}
 		s = &e[1];
 	}
 	free(macstr, M_XENBUS);
 	return (0);
 }
 
 /**
  * Entry point to this code when a new device is created.  Allocate the basic
  * structures and the ring buffers for communication with the backend, and
  * inform the backend of the appropriate details for those.  Switch to
  * Connected state.
  */
 static int
 netfront_probe(device_t dev)
 {
 
 	if (xen_hvm_domain() && xen_disable_pv_nics != 0)
 		return (ENXIO);
 
 	if (!strcmp(xenbus_get_type(dev), "vif")) {
 		device_set_desc(dev, "Virtual Network Interface");
 		return (0);
 	}
 
 	return (ENXIO);
 }
 
 static int
 netfront_attach(device_t dev)
 {
 	int err;
 
 	err = create_netdev(dev);
 	if (err != 0) {
 		xenbus_dev_fatal(dev, err, "creating netdev");
 		return (err);
 	}
 
 	SYSCTL_ADD_INT(device_get_sysctl_ctx(dev),
 	    SYSCTL_CHILDREN(device_get_sysctl_tree(dev)),
 	    OID_AUTO, "enable_lro", CTLFLAG_RW,
 	    &xn_enable_lro, 0, "Large Receive Offload");
 
 	SYSCTL_ADD_ULONG(device_get_sysctl_ctx(dev),
 	    SYSCTL_CHILDREN(device_get_sysctl_tree(dev)),
 	    OID_AUTO, "num_queues", CTLFLAG_RD,
 	    &xn_num_queues, "Number of pairs of queues");
 
 	return (0);
 }
 
 static int
 netfront_suspend(device_t dev)
 {
 	struct netfront_info *np = device_get_softc(dev);
 	u_int i;
 
 	for (i = 0; i < np->num_queues; i++) {
 		XN_RX_LOCK(&np->rxq[i]);
 		XN_TX_LOCK(&np->txq[i]);
 	}
 	netfront_carrier_off(np);
 	for (i = 0; i < np->num_queues; i++) {
 		XN_RX_UNLOCK(&np->rxq[i]);
 		XN_TX_UNLOCK(&np->txq[i]);
 	}
 	return (0);
 }
 
 /**
  * We are reconnecting to the backend, due to a suspend/resume, or a backend
  * driver restart.  We tear down our netif structure and recreate it, but
  * leave the device-layer structures intact so that this is transparent to the
  * rest of the kernel.
  */
 static int
 netfront_resume(device_t dev)
 {
 	struct netfront_info *info = device_get_softc(dev);
+	u_int i;
+
+	if (xen_suspend_cancelled) {
+		for (i = 0; i < info->num_queues; i++) {
+			XN_RX_LOCK(&info->rxq[i]);
+			XN_TX_LOCK(&info->txq[i]);
+		}
+		netfront_carrier_on(info);
+		for (i = 0; i < info->num_queues; i++) {
+			XN_RX_UNLOCK(&info->rxq[i]);
+			XN_TX_UNLOCK(&info->txq[i]);
+		}
+		return (0);
+	}
 
 	netif_disconnect_backend(info);
 	return (0);
 }
 
 static int
 write_queue_xenstore_keys(device_t dev,
     struct netfront_rxq *rxq,
     struct netfront_txq *txq,
     struct xs_transaction *xst, bool hierarchy)
 {
 	int err;
 	const char *message;
 	const char *node = xenbus_get_node(dev);
 	char *path;
 	size_t path_size;
 
 	KASSERT(rxq->id == txq->id, ("Mismatch between RX and TX queue ids"));
 	/* Split event channel support is not yet there. */
 	KASSERT(rxq->xen_intr_handle == txq->xen_intr_handle,
 	    ("Split event channels are not supported"));
 
 	if (hierarchy) {
 		path_size = strlen(node) + 10;
 		path = malloc(path_size, M_DEVBUF, M_WAITOK|M_ZERO);
 		snprintf(path, path_size, "%s/queue-%u", node, rxq->id);
 	} else {
 		path_size = strlen(node) + 1;
 		path = malloc(path_size, M_DEVBUF, M_WAITOK|M_ZERO);
 		snprintf(path, path_size, "%s", node);
 	}
 
 	err = xs_printf(*xst, path, "tx-ring-ref","%u", txq->ring_ref);
 	if (err != 0) {
 		message = "writing tx ring-ref";
 		goto error;
 	}
 	err = xs_printf(*xst, path, "rx-ring-ref","%u", rxq->ring_ref);
 	if (err != 0) {
 		message = "writing rx ring-ref";
 		goto error;
 	}
 	err = xs_printf(*xst, path, "event-channel", "%u",
 	    xen_intr_port(rxq->xen_intr_handle));
 	if (err != 0) {
 		message = "writing event-channel";
 		goto error;
 	}
 
 	free(path, M_DEVBUF);
 
 	return (0);
 
 error:
 	free(path, M_DEVBUF);
 	xenbus_dev_fatal(dev, err, "%s", message);
 
 	return (err);
 }
 
 /* Common code used when first setting up, and when resuming. */
 static int
 talk_to_backend(device_t dev, struct netfront_info *info)
 {
 	const char *message;
 	struct xs_transaction xst;
 	const char *node = xenbus_get_node(dev);
 	int err;
 	unsigned long num_queues, max_queues = 0;
 	unsigned int i;
 
 	err = xen_net_read_mac(dev, info->mac);
 	if (err != 0) {
 		xenbus_dev_fatal(dev, err, "parsing %s/mac", node);
 		goto out;
 	}
 
 	err = xs_scanf(XST_NIL, xenbus_get_otherend_path(info->xbdev),
 	    "multi-queue-max-queues", NULL, "%lu", &max_queues);
 	if (err != 0)
 		max_queues = 1;
 	num_queues = xn_num_queues;
 	if (num_queues > max_queues)
 		num_queues = max_queues;
 
 	err = setup_device(dev, info, num_queues);
 	if (err != 0)
 		goto out;
 
  again:
 	err = xs_transaction_start(&xst);
 	if (err != 0) {
 		xenbus_dev_fatal(dev, err, "starting transaction");
 		goto free;
 	}
 
 	if (info->num_queues == 1) {
 		err = write_queue_xenstore_keys(dev, &info->rxq[0],
 		    &info->txq[0], &xst, false);
 		if (err != 0)
 			goto abort_transaction_no_def_error;
 	} else {
 		err = xs_printf(xst, node, "multi-queue-num-queues",
 		    "%u", info->num_queues);
 		if (err != 0) {
 			message = "writing multi-queue-num-queues";
 			goto abort_transaction;
 		}
 
 		for (i = 0; i < info->num_queues; i++) {
 			err = write_queue_xenstore_keys(dev, &info->rxq[i],
 			    &info->txq[i], &xst, true);
 			if (err != 0)
 				goto abort_transaction_no_def_error;
 		}
 	}
 
 	err = xs_printf(xst, node, "request-rx-copy", "%u", 1);
 	if (err != 0) {
 		message = "writing request-rx-copy";
 		goto abort_transaction;
 	}
 	err = xs_printf(xst, node, "feature-rx-notify", "%d", 1);
 	if (err != 0) {
 		message = "writing feature-rx-notify";
 		goto abort_transaction;
 	}
 	err = xs_printf(xst, node, "feature-sg", "%d", 1);
 	if (err != 0) {
 		message = "writing feature-sg";
 		goto abort_transaction;
 	}
 	if ((info->xn_ifp->if_capenable & IFCAP_LRO) != 0) {
 		err = xs_printf(xst, node, "feature-gso-tcpv4", "%d", 1);
 		if (err != 0) {
 			message = "writing feature-gso-tcpv4";
 			goto abort_transaction;
 		}
 	}
 	if ((info->xn_ifp->if_capenable & IFCAP_RXCSUM) == 0) {
 		err = xs_printf(xst, node, "feature-no-csum-offload", "%d", 1);
 		if (err != 0) {
 			message = "writing feature-no-csum-offload";
 			goto abort_transaction;
 		}
 	}
 
 	err = xs_transaction_end(xst, 0);
 	if (err != 0) {
 		if (err == EAGAIN)
 			goto again;
 		xenbus_dev_fatal(dev, err, "completing transaction");
 		goto free;
 	}
 
 	return 0;
 
  abort_transaction:
 	xenbus_dev_fatal(dev, err, "%s", message);
  abort_transaction_no_def_error:
 	xs_transaction_end(xst, 1);
  free:
 	netif_free(info);
  out:
 	return (err);
 }
 
 static void
 xn_rxq_intr(struct netfront_rxq *rxq)
 {
 
 	XN_RX_LOCK(rxq);
 	xn_rxeof(rxq);
 	XN_RX_UNLOCK(rxq);
 }
 
 static void
 xn_txq_start(struct netfront_txq *txq)
 {
 	struct netfront_info *np = txq->info;
 	struct ifnet *ifp = np->xn_ifp;
 
 	XN_TX_LOCK_ASSERT(txq);
 	if (!drbr_empty(ifp, txq->br))
 		xn_txq_mq_start_locked(txq, NULL);
 }
 
 static void
 xn_txq_intr(struct netfront_txq *txq)
 {
 
 	XN_TX_LOCK(txq);
 	if (RING_HAS_UNCONSUMED_RESPONSES(&txq->ring))
 		xn_txeof(txq);
 	xn_txq_start(txq);
 	XN_TX_UNLOCK(txq);
 }
 
 static void
 xn_txq_tq_deferred(void *xtxq, int pending)
 {
 	struct netfront_txq *txq = xtxq;
 
 	XN_TX_LOCK(txq);
 	xn_txq_start(txq);
 	XN_TX_UNLOCK(txq);
 }
 
 static void
 disconnect_rxq(struct netfront_rxq *rxq)
 {
 
 	xn_release_rx_bufs(rxq);
 	gnttab_free_grant_references(rxq->gref_head);
 	gnttab_end_foreign_access(rxq->ring_ref, NULL);
 	/*
 	 * No split event channel support at the moment, handle will
 	 * be unbound in tx. So no need to call xen_intr_unbind here,
 	 * but we do want to reset the handler to 0.
 	 */
 	rxq->xen_intr_handle = 0;
 }
 
 static void
 destroy_rxq(struct netfront_rxq *rxq)
 {
 
 	callout_drain(&rxq->rx_refill);
 	free(rxq->ring.sring, M_DEVBUF);
 }
 
 static void
 destroy_rxqs(struct netfront_info *np)
 {
 	int i;
 
 	for (i = 0; i < np->num_queues; i++)
 		destroy_rxq(&np->rxq[i]);
 
 	free(np->rxq, M_DEVBUF);
 	np->rxq = NULL;
 }
 
 static int
 setup_rxqs(device_t dev, struct netfront_info *info,
 	   unsigned long num_queues)
 {
 	int q, i;
 	int error;
 	netif_rx_sring_t *rxs;
 	struct netfront_rxq *rxq;
 
 	info->rxq = malloc(sizeof(struct netfront_rxq) * num_queues,
 	    M_DEVBUF, M_WAITOK|M_ZERO);
 
 	for (q = 0; q < num_queues; q++) {
 		rxq = &info->rxq[q];
 
 		rxq->id = q;
 		rxq->info = info;
 		rxq->ring_ref = GRANT_REF_INVALID;
 		rxq->ring.sring = NULL;
 		snprintf(rxq->name, XN_QUEUE_NAME_LEN, "xnrx_%u", q);
 		mtx_init(&rxq->lock, rxq->name, "netfront receive lock",
 		    MTX_DEF);
 
 		for (i = 0; i <= NET_RX_RING_SIZE; i++) {
 			rxq->mbufs[i] = NULL;
 			rxq->grant_ref[i] = GRANT_REF_INVALID;
 		}
 
 		/* Start resources allocation */
 
 		if (gnttab_alloc_grant_references(NET_RX_RING_SIZE,
 		    &rxq->gref_head) != 0) {
 			device_printf(dev, "allocating rx gref");
 			error = ENOMEM;
 			goto fail;
 		}
 
 		rxs = (netif_rx_sring_t *)malloc(PAGE_SIZE, M_DEVBUF,
 		    M_WAITOK|M_ZERO);
 		SHARED_RING_INIT(rxs);
 		FRONT_RING_INIT(&rxq->ring, rxs, PAGE_SIZE);
 
 		error = xenbus_grant_ring(dev, virt_to_mfn(rxs),
 		    &rxq->ring_ref);
 		if (error != 0) {
 			device_printf(dev, "granting rx ring page");
 			goto fail_grant_ring;
 		}
 
 		callout_init(&rxq->rx_refill, 1);
 	}
 
 	return (0);
 
 fail_grant_ring:
 	gnttab_free_grant_references(rxq->gref_head);
 	free(rxq->ring.sring, M_DEVBUF);
 fail:
 	for (; q >= 0; q--) {
 		disconnect_rxq(&info->rxq[q]);
 		destroy_rxq(&info->rxq[q]);
 	}
 
 	free(info->rxq, M_DEVBUF);
 	return (error);
 }
 
 static void
 disconnect_txq(struct netfront_txq *txq)
 {
 
 	xn_release_tx_bufs(txq);
 	gnttab_free_grant_references(txq->gref_head);
 	gnttab_end_foreign_access(txq->ring_ref, NULL);
 	xen_intr_unbind(&txq->xen_intr_handle);
 }
 
 static void
 destroy_txq(struct netfront_txq *txq)
 {
 
 	free(txq->ring.sring, M_DEVBUF);
 	buf_ring_free(txq->br, M_DEVBUF);
 	taskqueue_drain_all(txq->tq);
 	taskqueue_free(txq->tq);
 }
 
 static void
 destroy_txqs(struct netfront_info *np)
 {
 	int i;
 
 	for (i = 0; i < np->num_queues; i++)
 		destroy_txq(&np->txq[i]);
 
 	free(np->txq, M_DEVBUF);
 	np->txq = NULL;
 }
 
 static int
 setup_txqs(device_t dev, struct netfront_info *info,
 	   unsigned long num_queues)
 {
 	int q, i;
 	int error;
 	netif_tx_sring_t *txs;
 	struct netfront_txq *txq;
 
 	info->txq = malloc(sizeof(struct netfront_txq) * num_queues,
 	    M_DEVBUF, M_WAITOK|M_ZERO);
 
 	for (q = 0; q < num_queues; q++) {
 		txq = &info->txq[q];
 
 		txq->id = q;
 		txq->info = info;
 
 		txq->ring_ref = GRANT_REF_INVALID;
 		txq->ring.sring = NULL;
 
 		snprintf(txq->name, XN_QUEUE_NAME_LEN, "xntx_%u", q);
 
 		mtx_init(&txq->lock, txq->name, "netfront transmit lock",
 		    MTX_DEF);
 
 		for (i = 0; i <= NET_TX_RING_SIZE; i++) {
 			txq->mbufs[i] = (void *) ((u_long) i+1);
 			txq->grant_ref[i] = GRANT_REF_INVALID;
 		}
 		txq->mbufs[NET_TX_RING_SIZE] = (void *)0;
 
 		/* Start resources allocation. */
 
 		if (gnttab_alloc_grant_references(NET_TX_RING_SIZE,
 		    &txq->gref_head) != 0) {
 			device_printf(dev, "failed to allocate tx grant refs\n");
 			error = ENOMEM;
 			goto fail;
 		}
 
 		txs = (netif_tx_sring_t *)malloc(PAGE_SIZE, M_DEVBUF,
 		    M_WAITOK|M_ZERO);
 		SHARED_RING_INIT(txs);
 		FRONT_RING_INIT(&txq->ring, txs, PAGE_SIZE);
 
 		error = xenbus_grant_ring(dev, virt_to_mfn(txs),
 		    &txq->ring_ref);
 		if (error != 0) {
 			device_printf(dev, "failed to grant tx ring\n");
 			goto fail_grant_ring;
 		}
 
 		txq->br = buf_ring_alloc(NET_TX_RING_SIZE, M_DEVBUF,
 		    M_WAITOK, &txq->lock);
 		TASK_INIT(&txq->defrtask, 0, xn_txq_tq_deferred, txq);
 
 		txq->tq = taskqueue_create(txq->name, M_WAITOK,
 		    taskqueue_thread_enqueue, &txq->tq);
 
 		error = taskqueue_start_threads(&txq->tq, 1, PI_NET,
 		    "%s txq %d", device_get_nameunit(dev), txq->id);
 		if (error != 0) {
 			device_printf(dev, "failed to start tx taskq %d\n",
 			    txq->id);
 			goto fail_start_thread;
 		}
 
 		error = xen_intr_alloc_and_bind_local_port(dev,
 		    xenbus_get_otherend_id(dev), /* filter */ NULL, xn_intr,
 		    &info->txq[q], INTR_TYPE_NET | INTR_MPSAFE | INTR_ENTROPY,
 		    &txq->xen_intr_handle);
 
 		if (error != 0) {
 			device_printf(dev, "xen_intr_alloc_and_bind_local_port failed\n");
 			goto fail_bind_port;
 		}
 	}
 
 	return (0);
 
 fail_bind_port:
 	taskqueue_drain_all(txq->tq);
 fail_start_thread:
 	buf_ring_free(txq->br, M_DEVBUF);
 	taskqueue_free(txq->tq);
 	gnttab_end_foreign_access(txq->ring_ref, NULL);
 fail_grant_ring:
 	gnttab_free_grant_references(txq->gref_head);
 	free(txq->ring.sring, M_DEVBUF);
 fail:
 	for (; q >= 0; q--) {
 		disconnect_txq(&info->txq[q]);
 		destroy_txq(&info->txq[q]);
 	}
 
 	free(info->txq, M_DEVBUF);
 	return (error);
 }
 
 static int
 setup_device(device_t dev, struct netfront_info *info,
     unsigned long num_queues)
 {
 	int error;
 	int q;
 
 	if (info->txq)
 		destroy_txqs(info);
 
 	if (info->rxq)
 		destroy_rxqs(info);
 
 	info->num_queues = 0;
 
 	error = setup_rxqs(dev, info, num_queues);
 	if (error != 0)
 		goto out;
 	error = setup_txqs(dev, info, num_queues);
 	if (error != 0)
 		goto out;
 
 	info->num_queues = num_queues;
 
 	/* No split event channel at the moment. */
 	for (q = 0; q < num_queues; q++)
 		info->rxq[q].xen_intr_handle = info->txq[q].xen_intr_handle;
 
 	return (0);
 
 out:
 	KASSERT(error != 0, ("Error path taken without providing an error code"));
 	return (error);
 }
 
 #ifdef INET
 /**
  * If this interface has an ipv4 address, send an arp for it. This
  * helps to get the network going again after migrating hosts.
  */
 static void
 netfront_send_fake_arp(device_t dev, struct netfront_info *info)
 {
 	struct ifnet *ifp;
 	struct ifaddr *ifa;
 
 	ifp = info->xn_ifp;
 	TAILQ_FOREACH(ifa, &ifp->if_addrhead, ifa_link) {
 		if (ifa->ifa_addr->sa_family == AF_INET) {
 			arp_ifinit(ifp, ifa);
 		}
 	}
 }
 #endif
 
 /**
  * Callback received when the backend's state changes.
  */
 static void
 netfront_backend_changed(device_t dev, XenbusState newstate)
 {
 	struct netfront_info *sc = device_get_softc(dev);
 
 	DPRINTK("newstate=%d\n", newstate);
 
 	switch (newstate) {
 	case XenbusStateInitialising:
 	case XenbusStateInitialised:
 	case XenbusStateUnknown:
 	case XenbusStateReconfigured:
 	case XenbusStateReconfiguring:
 		break;
 	case XenbusStateInitWait:
 		if (xenbus_get_state(dev) != XenbusStateInitialising)
 			break;
 		if (xn_connect(sc) != 0)
 			break;
 		/* Switch to connected state before kicking the rings. */
 		xenbus_set_state(sc->xbdev, XenbusStateConnected);
 		xn_kick_rings(sc);
 		break;
 	case XenbusStateClosing:
 		xenbus_set_state(dev, XenbusStateClosed);
 		break;
 	case XenbusStateClosed:
 		if (sc->xn_reset) {
 			netif_disconnect_backend(sc);
 			xenbus_set_state(dev, XenbusStateInitialising);
 			sc->xn_reset = false;
 		}
 		break;
 	case XenbusStateConnected:
 #ifdef INET
 		netfront_send_fake_arp(dev, sc);
 #endif
 		break;
 	}
 }
 
 /**
  * \brief Verify that there is sufficient space in the Tx ring
  *        buffer for a maximally sized request to be enqueued.
  *
  * A transmit request requires a transmit descriptor for each packet
  * fragment, plus up to 2 entries for "options" (e.g. TSO).
  */
 static inline int
 xn_tx_slot_available(struct netfront_txq *txq)
 {
 
 	return (RING_FREE_REQUESTS(&txq->ring) > (MAX_TX_REQ_FRAGS + 2));
 }
 
 static void
 xn_release_tx_bufs(struct netfront_txq *txq)
 {
 	int i;
 
 	for (i = 1; i <= NET_TX_RING_SIZE; i++) {
 		struct mbuf *m;
 
 		m = txq->mbufs[i];
 
 		/*
 		 * We assume that no kernel addresses are
 		 * less than NET_TX_RING_SIZE.  Any entry
 		 * in the table that is below this number
 		 * must be an index from free-list tracking.
 		 */
 		if (((uintptr_t)m) <= NET_TX_RING_SIZE)
 			continue;
 		gnttab_end_foreign_access_ref(txq->grant_ref[i]);
 		gnttab_release_grant_reference(&txq->gref_head,
 		    txq->grant_ref[i]);
 		txq->grant_ref[i] = GRANT_REF_INVALID;
 		add_id_to_freelist(txq->mbufs, i);
 		txq->mbufs_cnt--;
 		if (txq->mbufs_cnt < 0) {
 			panic("%s: tx_chain_cnt must be >= 0", __func__);
 		}
 		m_free(m);
 	}
 }
 
 static struct mbuf *
 xn_alloc_one_rx_buffer(struct netfront_rxq *rxq)
 {
 	struct mbuf *m;
 
 	m = m_getjcl(M_NOWAIT, MT_DATA, M_PKTHDR, MJUMPAGESIZE);
 	if (m == NULL)
 		return NULL;
 	m->m_len = m->m_pkthdr.len = MJUMPAGESIZE;
 
 	return (m);
 }
 
 static void
 xn_alloc_rx_buffers(struct netfront_rxq *rxq)
 {
 	RING_IDX req_prod;
 	int notify;
 
 	XN_RX_LOCK_ASSERT(rxq);
 
 	if (__predict_false(rxq->info->carrier == 0))
 		return;
 
 	for (req_prod = rxq->ring.req_prod_pvt;
 	     req_prod - rxq->ring.rsp_cons < NET_RX_RING_SIZE;
 	     req_prod++) {
 		struct mbuf *m;
 		unsigned short id;
 		grant_ref_t ref;
 		struct netif_rx_request *req;
 		unsigned long pfn;
 
 		m = xn_alloc_one_rx_buffer(rxq);
 		if (m == NULL)
 			break;
 
 		id = xn_rxidx(req_prod);
 
 		KASSERT(rxq->mbufs[id] == NULL, ("non-NULL xn_rx_chain"));
 		rxq->mbufs[id] = m;
 
 		ref = gnttab_claim_grant_reference(&rxq->gref_head);
 		KASSERT(ref != GNTTAB_LIST_END,
 		    ("reserved grant references exhuasted"));
 		rxq->grant_ref[id] = ref;
 
 		pfn = atop(vtophys(mtod(m, vm_offset_t)));
 		req = RING_GET_REQUEST(&rxq->ring, req_prod);
 
 		gnttab_grant_foreign_access_ref(ref,
 		    xenbus_get_otherend_id(rxq->info->xbdev), pfn, 0);
 		req->id = id;
 		req->gref = ref;
 	}
 
 	rxq->ring.req_prod_pvt = req_prod;
 
 	/* Not enough requests? Try again later. */
 	if (req_prod - rxq->ring.rsp_cons < NET_RX_SLOTS_MIN) {
 		callout_reset_curcpu(&rxq->rx_refill, hz/10,
 		    xn_alloc_rx_buffers_callout, rxq);
 		return;
 	}
 
 	wmb();		/* barrier so backend seens requests */
 
 	RING_PUSH_REQUESTS_AND_CHECK_NOTIFY(&rxq->ring, notify);
 	if (notify)
 		xen_intr_signal(rxq->xen_intr_handle);
 }
 
 static void xn_alloc_rx_buffers_callout(void *arg)
 {
 	struct netfront_rxq *rxq;
 
 	rxq = (struct netfront_rxq *)arg;
 	XN_RX_LOCK(rxq);
 	xn_alloc_rx_buffers(rxq);
 	XN_RX_UNLOCK(rxq);
 }
 
 static void
 xn_release_rx_bufs(struct netfront_rxq *rxq)
 {
 	int i,  ref;
 	struct mbuf *m;
 
 	for (i = 0; i < NET_RX_RING_SIZE; i++) {
 		m = rxq->mbufs[i];
 
 		if (m == NULL)
 			continue;
 
 		ref = rxq->grant_ref[i];
 		if (ref == GRANT_REF_INVALID)
 			continue;
 
 		gnttab_end_foreign_access_ref(ref);
 		gnttab_release_grant_reference(&rxq->gref_head, ref);
 		rxq->mbufs[i] = NULL;
 		rxq->grant_ref[i] = GRANT_REF_INVALID;
 		m_freem(m);
 	}
 }
 
 static void
 xn_rxeof(struct netfront_rxq *rxq)
 {
 	struct ifnet *ifp;
 	struct netfront_info *np = rxq->info;
 #if (defined(INET) || defined(INET6))
 	struct lro_ctrl *lro = &rxq->lro;
 #endif
 	struct netfront_rx_info rinfo;
 	struct netif_rx_response *rx = &rinfo.rx;
 	struct netif_extra_info *extras = rinfo.extras;
 	RING_IDX i, rp;
 	struct mbuf *m;
 	struct mbufq mbufq_rxq, mbufq_errq;
 	int err, work_to_do;
 
 	do {
 		XN_RX_LOCK_ASSERT(rxq);
 		if (!netfront_carrier_ok(np))
 			return;
 
 		/* XXX: there should be some sane limit. */
 		mbufq_init(&mbufq_errq, INT_MAX);
 		mbufq_init(&mbufq_rxq, INT_MAX);
 
 		ifp = np->xn_ifp;
 
 		rp = rxq->ring.sring->rsp_prod;
 		rmb();	/* Ensure we see queued responses up to 'rp'. */
 
 		i = rxq->ring.rsp_cons;
 		while ((i != rp)) {
 			memcpy(rx, RING_GET_RESPONSE(&rxq->ring, i), sizeof(*rx));
 			memset(extras, 0, sizeof(rinfo.extras));
 
 			m = NULL;
 			err = xn_get_responses(rxq, &rinfo, rp, &i, &m);
 
 			if (__predict_false(err)) {
 				if (m)
 					(void )mbufq_enqueue(&mbufq_errq, m);
 				if_inc_counter(ifp, IFCOUNTER_IQDROPS, 1);
 				continue;
 			}
 
 			m->m_pkthdr.rcvif = ifp;
 			if ( rx->flags & NETRXF_data_validated ) {
 				/* Tell the stack the checksums are okay */
 				/*
 				 * XXX this isn't necessarily the case - need to add
 				 * check
 				 */
 
 				m->m_pkthdr.csum_flags |=
 					(CSUM_IP_CHECKED | CSUM_IP_VALID | CSUM_DATA_VALID
 					    | CSUM_PSEUDO_HDR);
 				m->m_pkthdr.csum_data = 0xffff;
 			}
 			if ((rx->flags & NETRXF_extra_info) != 0 &&
 			    (extras[XEN_NETIF_EXTRA_TYPE_GSO - 1].type ==
 			    XEN_NETIF_EXTRA_TYPE_GSO)) {
 				m->m_pkthdr.tso_segsz =
 				extras[XEN_NETIF_EXTRA_TYPE_GSO - 1].u.gso.size;
 				m->m_pkthdr.csum_flags |= CSUM_TSO;
 			}
 
 			(void )mbufq_enqueue(&mbufq_rxq, m);
 			rxq->ring.rsp_cons = i;
 		}
 
 		mbufq_drain(&mbufq_errq);
 
 		/*
 		 * Process all the mbufs after the remapping is complete.
 		 * Break the mbuf chain first though.
 		 */
 		while ((m = mbufq_dequeue(&mbufq_rxq)) != NULL) {
 			if_inc_counter(ifp, IFCOUNTER_IPACKETS, 1);
 
 			/* XXX: Do we really need to drop the rx lock? */
 			XN_RX_UNLOCK(rxq);
 #if (defined(INET) || defined(INET6))
 			/* Use LRO if possible */
 			if ((ifp->if_capenable & IFCAP_LRO) == 0 ||
 			    lro->lro_cnt == 0 || tcp_lro_rx(lro, m, 0)) {
 				/*
 				 * If LRO fails, pass up to the stack
 				 * directly.
 				 */
 				(*ifp->if_input)(ifp, m);
 			}
 #else
 			(*ifp->if_input)(ifp, m);
 #endif
 
 			XN_RX_LOCK(rxq);
 		}
 
 		rxq->ring.rsp_cons = i;
 
 #if (defined(INET) || defined(INET6))
 		/*
 		 * Flush any outstanding LRO work
 		 */
 		tcp_lro_flush_all(lro);
 #endif
 
 		xn_alloc_rx_buffers(rxq);
 
 		RING_FINAL_CHECK_FOR_RESPONSES(&rxq->ring, work_to_do);
 	} while (work_to_do);
 }
 
 static void
 xn_txeof(struct netfront_txq *txq)
 {
 	RING_IDX i, prod;
 	unsigned short id;
 	struct ifnet *ifp;
 	netif_tx_response_t *txr;
 	struct mbuf *m;
 	struct netfront_info *np = txq->info;
 
 	XN_TX_LOCK_ASSERT(txq);
 
 	if (!netfront_carrier_ok(np))
 		return;
 
 	ifp = np->xn_ifp;
 
 	do {
 		prod = txq->ring.sring->rsp_prod;
 		rmb(); /* Ensure we see responses up to 'rp'. */
 
 		for (i = txq->ring.rsp_cons; i != prod; i++) {
 			txr = RING_GET_RESPONSE(&txq->ring, i);
 			if (txr->status == NETIF_RSP_NULL)
 				continue;
 
 			if (txr->status != NETIF_RSP_OKAY) {
 				printf("%s: WARNING: response is %d!\n",
 				       __func__, txr->status);
 			}
 			id = txr->id;
 			m = txq->mbufs[id];
 			KASSERT(m != NULL, ("mbuf not found in chain"));
 			KASSERT((uintptr_t)m > NET_TX_RING_SIZE,
 				("mbuf already on the free list, but we're "
 				"trying to free it again!"));
 			M_ASSERTVALID(m);
 
 			if (__predict_false(gnttab_query_foreign_access(
 			    txq->grant_ref[id]) != 0)) {
 				panic("%s: grant id %u still in use by the "
 				    "backend", __func__, id);
 			}
 			gnttab_end_foreign_access_ref(txq->grant_ref[id]);
 			gnttab_release_grant_reference(
 				&txq->gref_head, txq->grant_ref[id]);
 			txq->grant_ref[id] = GRANT_REF_INVALID;
 
 			txq->mbufs[id] = NULL;
 			add_id_to_freelist(txq->mbufs, id);
 			txq->mbufs_cnt--;
 			m_free(m);
 			/* Only mark the txq active if we've freed up at least one slot to try */
 			ifp->if_drv_flags &= ~IFF_DRV_OACTIVE;
 		}
 		txq->ring.rsp_cons = prod;
 
 		/*
 		 * Set a new event, then check for race with update of
 		 * tx_cons. Note that it is essential to schedule a
 		 * callback, no matter how few buffers are pending. Even if
 		 * there is space in the transmit ring, higher layers may
 		 * be blocked because too much data is outstanding: in such
 		 * cases notification from Xen is likely to be the only kick
 		 * that we'll get.
 		 */
 		txq->ring.sring->rsp_event =
 		    prod + ((txq->ring.sring->req_prod - prod) >> 1) + 1;
 
 		mb();
 	} while (prod != txq->ring.sring->rsp_prod);
 
 	if (txq->full &&
 	    ((txq->ring.sring->req_prod - prod) < NET_TX_RING_SIZE)) {
 		txq->full = false;
 		xn_txq_start(txq);
 	}
 }
 
 static void
 xn_intr(void *xsc)
 {
 	struct netfront_txq *txq = xsc;
 	struct netfront_info *np = txq->info;
 	struct netfront_rxq *rxq = &np->rxq[txq->id];
 
 	/* kick both tx and rx */
 	xn_rxq_intr(rxq);
 	xn_txq_intr(txq);
 }
 
 static void
 xn_move_rx_slot(struct netfront_rxq *rxq, struct mbuf *m,
     grant_ref_t ref)
 {
 	int new = xn_rxidx(rxq->ring.req_prod_pvt);
 
 	KASSERT(rxq->mbufs[new] == NULL, ("mbufs != NULL"));
 	rxq->mbufs[new] = m;
 	rxq->grant_ref[new] = ref;
 	RING_GET_REQUEST(&rxq->ring, rxq->ring.req_prod_pvt)->id = new;
 	RING_GET_REQUEST(&rxq->ring, rxq->ring.req_prod_pvt)->gref = ref;
 	rxq->ring.req_prod_pvt++;
 }
 
 static int
 xn_get_extras(struct netfront_rxq *rxq,
     struct netif_extra_info *extras, RING_IDX rp, RING_IDX *cons)
 {
 	struct netif_extra_info *extra;
 
 	int err = 0;
 
 	do {
 		struct mbuf *m;
 		grant_ref_t ref;
 
 		if (__predict_false(*cons + 1 == rp)) {
 			err = EINVAL;
 			break;
 		}
 
 		extra = (struct netif_extra_info *)
 		RING_GET_RESPONSE(&rxq->ring, ++(*cons));
 
 		if (__predict_false(!extra->type ||
 			extra->type >= XEN_NETIF_EXTRA_TYPE_MAX)) {
 			err = EINVAL;
 		} else {
 			memcpy(&extras[extra->type - 1], extra, sizeof(*extra));
 		}
 
 		m = xn_get_rx_mbuf(rxq, *cons);
 		ref = xn_get_rx_ref(rxq,  *cons);
 		xn_move_rx_slot(rxq, m, ref);
 	} while (extra->flags & XEN_NETIF_EXTRA_FLAG_MORE);
 
 	return err;
 }
 
 static int
 xn_get_responses(struct netfront_rxq *rxq,
     struct netfront_rx_info *rinfo, RING_IDX rp, RING_IDX *cons,
     struct mbuf  **list)
 {
 	struct netif_rx_response *rx = &rinfo->rx;
 	struct netif_extra_info *extras = rinfo->extras;
 	struct mbuf *m, *m0, *m_prev;
 	grant_ref_t ref = xn_get_rx_ref(rxq, *cons);
 	RING_IDX ref_cons = *cons;
 	int frags = 1;
 	int err = 0;
 	u_long ret;
 
 	m0 = m = m_prev = xn_get_rx_mbuf(rxq, *cons);
 
 	if (rx->flags & NETRXF_extra_info) {
 		err = xn_get_extras(rxq, extras, rp, cons);
 	}
 
 	if (m0 != NULL) {
 		m0->m_pkthdr.len = 0;
 		m0->m_next = NULL;
 	}
 
 	for (;;) {
 #if 0
 		DPRINTK("rx->status=%hd rx->offset=%hu frags=%u\n",
 			rx->status, rx->offset, frags);
 #endif
 		if (__predict_false(rx->status < 0 ||
 			rx->offset + rx->status > PAGE_SIZE)) {
 
 			xn_move_rx_slot(rxq, m, ref);
 			if (m0 == m)
 				m0 = NULL;
 			m = NULL;
 			err = EINVAL;
 			goto next_skip_queue;
 		}
 
 		/*
 		 * This definitely indicates a bug, either in this driver or in
 		 * the backend driver. In future this should flag the bad
 		 * situation to the system controller to reboot the backed.
 		 */
 		if (ref == GRANT_REF_INVALID) {
 			printf("%s: Bad rx response id %d.\n", __func__, rx->id);
 			err = EINVAL;
 			goto next;
 		}
 
 		ret = gnttab_end_foreign_access_ref(ref);
 		KASSERT(ret, ("Unable to end access to grant references"));
 
 		gnttab_release_grant_reference(&rxq->gref_head, ref);
 
 next:
 		if (m == NULL)
 			break;
 
 		m->m_len = rx->status;
 		m->m_data += rx->offset;
 		m0->m_pkthdr.len += rx->status;
 
 next_skip_queue:
 		if (!(rx->flags & NETRXF_more_data))
 			break;
 
 		if (*cons + frags == rp) {
 			if (net_ratelimit())
 				WPRINTK("Need more frags\n");
 			err = ENOENT;
 			printf("%s: cons %u frags %u rp %u, not enough frags\n",
 			       __func__, *cons, frags, rp);
 			break;
 		}
 		/*
 		 * Note that m can be NULL, if rx->status < 0 or if
 		 * rx->offset + rx->status > PAGE_SIZE above.
 		 */
 		m_prev = m;
 
 		rx = RING_GET_RESPONSE(&rxq->ring, *cons + frags);
 		m = xn_get_rx_mbuf(rxq, *cons + frags);
 
 		/*
 		 * m_prev == NULL can happen if rx->status < 0 or if
 		 * rx->offset + * rx->status > PAGE_SIZE above.
 		 */
 		if (m_prev != NULL)
 			m_prev->m_next = m;
 
 		/*
 		 * m0 can be NULL if rx->status < 0 or if * rx->offset +
 		 * rx->status > PAGE_SIZE above.
 		 */
 		if (m0 == NULL)
 			m0 = m;
 		m->m_next = NULL;
 		ref = xn_get_rx_ref(rxq, *cons + frags);
 		ref_cons = *cons + frags;
 		frags++;
 	}
 	*list = m0;
 	*cons += frags;
 
 	return (err);
 }
 
 /**
  * \brief Count the number of fragments in an mbuf chain.
  *
  * Surprisingly, there isn't an M* macro for this.
  */
 static inline int
 xn_count_frags(struct mbuf *m)
 {
 	int nfrags;
 
 	for (nfrags = 0; m != NULL; m = m->m_next)
 		nfrags++;
 
 	return (nfrags);
 }
 
 /**
  * Given an mbuf chain, make sure we have enough room and then push
  * it onto the transmit ring.
  */
 static int
 xn_assemble_tx_request(struct netfront_txq *txq, struct mbuf *m_head)
 {
 	struct mbuf *m;
 	struct netfront_info *np = txq->info;
 	struct ifnet *ifp = np->xn_ifp;
 	u_int nfrags;
 	int otherend_id;
 
 	/**
 	 * Defragment the mbuf if necessary.
 	 */
 	nfrags = xn_count_frags(m_head);
 
 	/*
 	 * Check to see whether this request is longer than netback
 	 * can handle, and try to defrag it.
 	 */
 	/**
 	 * It is a bit lame, but the netback driver in Linux can't
 	 * deal with nfrags > MAX_TX_REQ_FRAGS, which is a quirk of
 	 * the Linux network stack.
 	 */
 	if (nfrags > np->maxfrags) {
 		m = m_defrag(m_head, M_NOWAIT);
 		if (!m) {
 			/*
 			 * Defrag failed, so free the mbuf and
 			 * therefore drop the packet.
 			 */
 			m_freem(m_head);
 			return (EMSGSIZE);
 		}
 		m_head = m;
 	}
 
 	/* Determine how many fragments now exist */
 	nfrags = xn_count_frags(m_head);
 
 	/*
 	 * Check to see whether the defragmented packet has too many
 	 * segments for the Linux netback driver.
 	 */
 	/**
 	 * The FreeBSD TCP stack, with TSO enabled, can produce a chain
 	 * of mbufs longer than Linux can handle.  Make sure we don't
 	 * pass a too-long chain over to the other side by dropping the
 	 * packet.  It doesn't look like there is currently a way to
 	 * tell the TCP stack to generate a shorter chain of packets.
 	 */
 	if (nfrags > MAX_TX_REQ_FRAGS) {
 #ifdef DEBUG
 		printf("%s: nfrags %d > MAX_TX_REQ_FRAGS %d, netback "
 		       "won't be able to handle it, dropping\n",
 		       __func__, nfrags, MAX_TX_REQ_FRAGS);
 #endif
 		m_freem(m_head);
 		return (EMSGSIZE);
 	}
 
 	/*
 	 * This check should be redundant.  We've already verified that we
 	 * have enough slots in the ring to handle a packet of maximum
 	 * size, and that our packet is less than the maximum size.  Keep
 	 * it in here as an assert for now just to make certain that
 	 * chain_cnt is accurate.
 	 */
 	KASSERT((txq->mbufs_cnt + nfrags) <= NET_TX_RING_SIZE,
 		("%s: chain_cnt (%d) + nfrags (%d) > NET_TX_RING_SIZE "
 		 "(%d)!", __func__, (int) txq->mbufs_cnt,
                     (int) nfrags, (int) NET_TX_RING_SIZE));
 
 	/*
 	 * Start packing the mbufs in this chain into
 	 * the fragment pointers. Stop when we run out
 	 * of fragments or hit the end of the mbuf chain.
 	 */
 	m = m_head;
 	otherend_id = xenbus_get_otherend_id(np->xbdev);
 	for (m = m_head; m; m = m->m_next) {
 		netif_tx_request_t *tx;
 		uintptr_t id;
 		grant_ref_t ref;
 		u_long mfn; /* XXX Wrong type? */
 
 		tx = RING_GET_REQUEST(&txq->ring, txq->ring.req_prod_pvt);
 		id = get_id_from_freelist(txq->mbufs);
 		if (id == 0)
 			panic("%s: was allocated the freelist head!\n",
 			    __func__);
 		txq->mbufs_cnt++;
 		if (txq->mbufs_cnt > NET_TX_RING_SIZE)
 			panic("%s: tx_chain_cnt must be <= NET_TX_RING_SIZE\n",
 			    __func__);
 		txq->mbufs[id] = m;
 		tx->id = id;
 		ref = gnttab_claim_grant_reference(&txq->gref_head);
 		KASSERT((short)ref >= 0, ("Negative ref"));
 		mfn = virt_to_mfn(mtod(m, vm_offset_t));
 		gnttab_grant_foreign_access_ref(ref, otherend_id,
 		    mfn, GNTMAP_readonly);
 		tx->gref = txq->grant_ref[id] = ref;
 		tx->offset = mtod(m, vm_offset_t) & (PAGE_SIZE - 1);
 		tx->flags = 0;
 		if (m == m_head) {
 			/*
 			 * The first fragment has the entire packet
 			 * size, subsequent fragments have just the
 			 * fragment size. The backend works out the
 			 * true size of the first fragment by
 			 * subtracting the sizes of the other
 			 * fragments.
 			 */
 			tx->size = m->m_pkthdr.len;
 
 			/*
 			 * The first fragment contains the checksum flags
 			 * and is optionally followed by extra data for
 			 * TSO etc.
 			 */
 			/**
 			 * CSUM_TSO requires checksum offloading.
 			 * Some versions of FreeBSD fail to
 			 * set CSUM_TCP in the CSUM_TSO case,
 			 * so we have to test for CSUM_TSO
 			 * explicitly.
 			 */
 			if (m->m_pkthdr.csum_flags
 			    & (CSUM_DELAY_DATA | CSUM_TSO)) {
 				tx->flags |= (NETTXF_csum_blank
 				    | NETTXF_data_validated);
 			}
 			if (m->m_pkthdr.csum_flags & CSUM_TSO) {
 				struct netif_extra_info *gso =
 					(struct netif_extra_info *)
 					RING_GET_REQUEST(&txq->ring,
 							 ++txq->ring.req_prod_pvt);
 
 				tx->flags |= NETTXF_extra_info;
 
 				gso->u.gso.size = m->m_pkthdr.tso_segsz;
 				gso->u.gso.type =
 					XEN_NETIF_GSO_TYPE_TCPV4;
 				gso->u.gso.pad = 0;
 				gso->u.gso.features = 0;
 
 				gso->type = XEN_NETIF_EXTRA_TYPE_GSO;
 				gso->flags = 0;
 			}
 		} else {
 			tx->size = m->m_len;
 		}
 		if (m->m_next)
 			tx->flags |= NETTXF_more_data;
 
 		txq->ring.req_prod_pvt++;
 	}
 	BPF_MTAP(ifp, m_head);
 
 	if_inc_counter(ifp, IFCOUNTER_OPACKETS, 1);
 	if_inc_counter(ifp, IFCOUNTER_OBYTES, m_head->m_pkthdr.len);
 	if (m_head->m_flags & M_MCAST)
 		if_inc_counter(ifp, IFCOUNTER_OMCASTS, 1);
 
 	xn_txeof(txq);
 
 	return (0);
 }
 
 /* equivalent of network_open() in Linux */
 static void
 xn_ifinit_locked(struct netfront_info *np)
 {
 	struct ifnet *ifp;
 	int i;
 	struct netfront_rxq *rxq;
 
 	XN_LOCK_ASSERT(np);
 
 	ifp = np->xn_ifp;
 
 	if (ifp->if_drv_flags & IFF_DRV_RUNNING || !netfront_carrier_ok(np))
 		return;
 
 	xn_stop(np);
 
 	for (i = 0; i < np->num_queues; i++) {
 		rxq = &np->rxq[i];
 		XN_RX_LOCK(rxq);
 		xn_alloc_rx_buffers(rxq);
 		rxq->ring.sring->rsp_event = rxq->ring.rsp_cons + 1;
 		if (RING_HAS_UNCONSUMED_RESPONSES(&rxq->ring))
 			xn_rxeof(rxq);
 		XN_RX_UNLOCK(rxq);
 	}
 
 	ifp->if_drv_flags |= IFF_DRV_RUNNING;
 	ifp->if_drv_flags &= ~IFF_DRV_OACTIVE;
 	if_link_state_change(ifp, LINK_STATE_UP);
 }
 
 static void
 xn_ifinit(void *xsc)
 {
 	struct netfront_info *sc = xsc;
 
 	XN_LOCK(sc);
 	xn_ifinit_locked(sc);
 	XN_UNLOCK(sc);
 }
 
 static int
 xn_ioctl(struct ifnet *ifp, u_long cmd, caddr_t data)
 {
 	struct netfront_info *sc = ifp->if_softc;
 	struct ifreq *ifr = (struct ifreq *) data;
 	device_t dev;
 #ifdef INET
 	struct ifaddr *ifa = (struct ifaddr *)data;
 #endif
 	int mask, error = 0, reinit;
 
 	dev = sc->xbdev;
 
 	switch(cmd) {
 	case SIOCSIFADDR:
 #ifdef INET
 		XN_LOCK(sc);
 		if (ifa->ifa_addr->sa_family == AF_INET) {
 			ifp->if_flags |= IFF_UP;
 			if (!(ifp->if_drv_flags & IFF_DRV_RUNNING))
 				xn_ifinit_locked(sc);
 			arp_ifinit(ifp, ifa);
 			XN_UNLOCK(sc);
 		} else {
 			XN_UNLOCK(sc);
 #endif
 			error = ether_ioctl(ifp, cmd, data);
 #ifdef INET
 		}
 #endif
 		break;
 	case SIOCSIFMTU:
 		ifp->if_mtu = ifr->ifr_mtu;
 		ifp->if_drv_flags &= ~IFF_DRV_RUNNING;
 		xn_ifinit(sc);
 		break;
 	case SIOCSIFFLAGS:
 		XN_LOCK(sc);
 		if (ifp->if_flags & IFF_UP) {
 			/*
 			 * If only the state of the PROMISC flag changed,
 			 * then just use the 'set promisc mode' command
 			 * instead of reinitializing the entire NIC. Doing
 			 * a full re-init means reloading the firmware and
 			 * waiting for it to start up, which may take a
 			 * second or two.
 			 */
 			xn_ifinit_locked(sc);
 		} else {
 			if (ifp->if_drv_flags & IFF_DRV_RUNNING) {
 				xn_stop(sc);
 			}
 		}
 		sc->xn_if_flags = ifp->if_flags;
 		XN_UNLOCK(sc);
 		break;
 	case SIOCSIFCAP:
 		mask = ifr->ifr_reqcap ^ ifp->if_capenable;
 		reinit = 0;
 
 		if (mask & IFCAP_TXCSUM) {
 			ifp->if_capenable ^= IFCAP_TXCSUM;
 			ifp->if_hwassist ^= XN_CSUM_FEATURES;
 		}
 		if (mask & IFCAP_TSO4) {
 			ifp->if_capenable ^= IFCAP_TSO4;
 			ifp->if_hwassist ^= CSUM_TSO;
 		}
 
 		if (mask & (IFCAP_RXCSUM | IFCAP_LRO)) {
 			/* These Rx features require us to renegotiate. */
 			reinit = 1;
 
 			if (mask & IFCAP_RXCSUM)
 				ifp->if_capenable ^= IFCAP_RXCSUM;
 			if (mask & IFCAP_LRO)
 				ifp->if_capenable ^= IFCAP_LRO;
 		}
 
 		if (reinit == 0)
 			break;
 
 		/*
 		 * We must reset the interface so the backend picks up the
 		 * new features.
 		 */
 		device_printf(sc->xbdev,
 		    "performing interface reset due to feature change\n");
 		XN_LOCK(sc);
 		netfront_carrier_off(sc);
 		sc->xn_reset = true;
 		/*
 		 * NB: the pending packet queue is not flushed, since
 		 * the interface should still support the old options.
 		 */
 		XN_UNLOCK(sc);
 		/*
 		 * Delete the xenstore nodes that export features.
 		 *
 		 * NB: There's a xenbus state called
 		 * "XenbusStateReconfiguring", which is what we should set
 		 * here. Sadly none of the backends know how to handle it,
 		 * and simply disconnect from the frontend, so we will just
 		 * switch back to XenbusStateInitialising in order to force
 		 * a reconnection.
 		 */
 		xs_rm(XST_NIL, xenbus_get_node(dev), "feature-gso-tcpv4");
 		xs_rm(XST_NIL, xenbus_get_node(dev), "feature-no-csum-offload");
 		xenbus_set_state(dev, XenbusStateClosing);
 
 		/*
 		 * Wait for the frontend to reconnect before returning
 		 * from the ioctl. 30s should be more than enough for any
 		 * sane backend to reconnect.
 		 */
 		error = tsleep(sc, 0, "xn_rst", 30*hz);
 		break;
 	case SIOCADDMULTI:
 	case SIOCDELMULTI:
 		break;
 	case SIOCSIFMEDIA:
 	case SIOCGIFMEDIA:
 		error = ifmedia_ioctl(ifp, ifr, &sc->sc_media, cmd);
 		break;
 	default:
 		error = ether_ioctl(ifp, cmd, data);
 	}
 
 	return (error);
 }
 
 static void
 xn_stop(struct netfront_info *sc)
 {
 	struct ifnet *ifp;
 
 	XN_LOCK_ASSERT(sc);
 
 	ifp = sc->xn_ifp;
 
 	ifp->if_drv_flags &= ~(IFF_DRV_RUNNING | IFF_DRV_OACTIVE);
 	if_link_state_change(ifp, LINK_STATE_DOWN);
 }
 
 static void
 xn_rebuild_rx_bufs(struct netfront_rxq *rxq)
 {
 	int requeue_idx, i;
 	grant_ref_t ref;
 	netif_rx_request_t *req;
 
 	for (requeue_idx = 0, i = 0; i < NET_RX_RING_SIZE; i++) {
 		struct mbuf *m;
 		u_long pfn;
 
 		if (rxq->mbufs[i] == NULL)
 			continue;
 
 		m = rxq->mbufs[requeue_idx] = xn_get_rx_mbuf(rxq, i);
 		ref = rxq->grant_ref[requeue_idx] = xn_get_rx_ref(rxq, i);
 
 		req = RING_GET_REQUEST(&rxq->ring, requeue_idx);
 		pfn = vtophys(mtod(m, vm_offset_t)) >> PAGE_SHIFT;
 
 		gnttab_grant_foreign_access_ref(ref,
 		    xenbus_get_otherend_id(rxq->info->xbdev),
 		    pfn, 0);
 
 		req->gref = ref;
 		req->id   = requeue_idx;
 
 		requeue_idx++;
 	}
 
 	rxq->ring.req_prod_pvt = requeue_idx;
 }
 
 /* START of Xenolinux helper functions adapted to FreeBSD */
 static int
 xn_connect(struct netfront_info *np)
 {
 	int i, error;
 	u_int feature_rx_copy;
 	struct netfront_rxq *rxq;
 	struct netfront_txq *txq;
 
 	error = xs_scanf(XST_NIL, xenbus_get_otherend_path(np->xbdev),
 	    "feature-rx-copy", NULL, "%u", &feature_rx_copy);
 	if (error != 0)
 		feature_rx_copy = 0;
 
 	/* We only support rx copy. */
 	if (!feature_rx_copy)
 		return (EPROTONOSUPPORT);
 
 	/* Recovery procedure: */
 	error = talk_to_backend(np->xbdev, np);
 	if (error != 0)
 		return (error);
 
 	/* Step 1: Reinitialise variables. */
 	xn_query_features(np);
 	xn_configure_features(np);
 
 	/* Step 2: Release TX buffer */
 	for (i = 0; i < np->num_queues; i++) {
 		txq = &np->txq[i];
 		xn_release_tx_bufs(txq);
 	}
 
 	/* Step 3: Rebuild the RX buffer freelist and the RX ring itself. */
 	for (i = 0; i < np->num_queues; i++) {
 		rxq = &np->rxq[i];
 		xn_rebuild_rx_bufs(rxq);
 	}
 
 	/* Step 4: All public and private state should now be sane.  Get
 	 * ready to start sending and receiving packets and give the driver
 	 * domain a kick because we've probably just requeued some
 	 * packets.
 	 */
 	netfront_carrier_on(np);
 	wakeup(np);
 
 	return (0);
 }
 
 static void
 xn_kick_rings(struct netfront_info *np)
 {
 	struct netfront_rxq *rxq;
 	struct netfront_txq *txq;
 	int i;
 
 	for (i = 0; i < np->num_queues; i++) {
 		txq = &np->txq[i];
 		rxq = &np->rxq[i];
 		xen_intr_signal(txq->xen_intr_handle);
 		XN_TX_LOCK(txq);
 		xn_txeof(txq);
 		XN_TX_UNLOCK(txq);
 		XN_RX_LOCK(rxq);
 		xn_alloc_rx_buffers(rxq);
 		XN_RX_UNLOCK(rxq);
 	}
 }
 
 static void
 xn_query_features(struct netfront_info *np)
 {
 	int val;
 
 	device_printf(np->xbdev, "backend features:");
 
 	if (xs_scanf(XST_NIL, xenbus_get_otherend_path(np->xbdev),
 		"feature-sg", NULL, "%d", &val) != 0)
 		val = 0;
 
 	np->maxfrags = 1;
 	if (val) {
 		np->maxfrags = MAX_TX_REQ_FRAGS;
 		printf(" feature-sg");
 	}
 
 	if (xs_scanf(XST_NIL, xenbus_get_otherend_path(np->xbdev),
 		"feature-gso-tcpv4", NULL, "%d", &val) != 0)
 		val = 0;
 
 	np->xn_ifp->if_capabilities &= ~(IFCAP_TSO4|IFCAP_LRO);
 	if (val) {
 		np->xn_ifp->if_capabilities |= IFCAP_TSO4|IFCAP_LRO;
 		printf(" feature-gso-tcp4");
 	}
 
 	/*
 	 * HW CSUM offload is assumed to be available unless
 	 * feature-no-csum-offload is set in xenstore.
 	 */
 	if (xs_scanf(XST_NIL, xenbus_get_otherend_path(np->xbdev),
 		"feature-no-csum-offload", NULL, "%d", &val) != 0)
 		val = 0;
 
 	np->xn_ifp->if_capabilities |= IFCAP_HWCSUM;
 	if (val) {
 		np->xn_ifp->if_capabilities &= ~(IFCAP_HWCSUM);
 		printf(" feature-no-csum-offload");
 	}
 
 	printf("\n");
 }
 
 static int
 xn_configure_features(struct netfront_info *np)
 {
 	int err, cap_enabled;
 #if (defined(INET) || defined(INET6))
 	int i;
 #endif
 	struct ifnet *ifp;
 
 	ifp = np->xn_ifp;
 	err = 0;
 
 	if ((ifp->if_capenable & ifp->if_capabilities) == ifp->if_capenable) {
 		/* Current options are available, no need to do anything. */
 		return (0);
 	}
 
 	/* Try to preserve as many options as possible. */
 	cap_enabled = ifp->if_capenable;
 	ifp->if_capenable = ifp->if_hwassist = 0;
 
 #if (defined(INET) || defined(INET6))
 	if ((cap_enabled & IFCAP_LRO) != 0)
 		for (i = 0; i < np->num_queues; i++)
 			tcp_lro_free(&np->rxq[i].lro);
 	if (xn_enable_lro &&
 	    (ifp->if_capabilities & cap_enabled & IFCAP_LRO) != 0) {
 	    	ifp->if_capenable |= IFCAP_LRO;
 		for (i = 0; i < np->num_queues; i++) {
 			err = tcp_lro_init(&np->rxq[i].lro);
 			if (err != 0) {
 				device_printf(np->xbdev,
 				    "LRO initialization failed\n");
 				ifp->if_capenable &= ~IFCAP_LRO;
 				break;
 			}
 			np->rxq[i].lro.ifp = ifp;
 		}
 	}
 	if ((ifp->if_capabilities & cap_enabled & IFCAP_TSO4) != 0) {
 		ifp->if_capenable |= IFCAP_TSO4;
 		ifp->if_hwassist |= CSUM_TSO;
 	}
 #endif
 	if ((ifp->if_capabilities & cap_enabled & IFCAP_TXCSUM) != 0) {
 		ifp->if_capenable |= IFCAP_TXCSUM;
 		ifp->if_hwassist |= XN_CSUM_FEATURES;
 	}
 	if ((ifp->if_capabilities & cap_enabled & IFCAP_RXCSUM) != 0)
 		ifp->if_capenable |= IFCAP_RXCSUM;
 
 	return (err);
 }
 
 static int
 xn_txq_mq_start_locked(struct netfront_txq *txq, struct mbuf *m)
 {
 	struct netfront_info *np;
 	struct ifnet *ifp;
 	struct buf_ring *br;
 	int error, notify;
 
 	np = txq->info;
 	br = txq->br;
 	ifp = np->xn_ifp;
 	error = 0;
 
 	XN_TX_LOCK_ASSERT(txq);
 
 	if ((ifp->if_drv_flags & IFF_DRV_RUNNING) == 0 ||
 	    !netfront_carrier_ok(np)) {
 		if (m != NULL)
 			error = drbr_enqueue(ifp, br, m);
 		return (error);
 	}
 
 	if (m != NULL) {
 		error = drbr_enqueue(ifp, br, m);
 		if (error != 0)
 			return (error);
 	}
 
 	while ((m = drbr_peek(ifp, br)) != NULL) {
 		if (!xn_tx_slot_available(txq)) {
 			drbr_putback(ifp, br, m);
 			break;
 		}
 
 		error = xn_assemble_tx_request(txq, m);
 		/* xn_assemble_tx_request always consumes the mbuf*/
 		if (error != 0) {
 			drbr_advance(ifp, br);
 			break;
 		}
 
 		RING_PUSH_REQUESTS_AND_CHECK_NOTIFY(&txq->ring, notify);
 		if (notify)
 			xen_intr_signal(txq->xen_intr_handle);
 
 		drbr_advance(ifp, br);
 	}
 
 	if (RING_FULL(&txq->ring))
 		txq->full = true;
 
 	return (0);
 }
 
 static int
 xn_txq_mq_start(struct ifnet *ifp, struct mbuf *m)
 {
 	struct netfront_info *np;
 	struct netfront_txq *txq;
 	int i, npairs, error;
 
 	np = ifp->if_softc;
 	npairs = np->num_queues;
 
 	if (!netfront_carrier_ok(np))
 		return (ENOBUFS);
 
 	KASSERT(npairs != 0, ("called with 0 available queues"));
 
 	/* check if flowid is set */
 	if (M_HASHTYPE_GET(m) != M_HASHTYPE_NONE)
 		i = m->m_pkthdr.flowid % npairs;
 	else
 		i = curcpu % npairs;
 
 	txq = &np->txq[i];
 
 	if (XN_TX_TRYLOCK(txq) != 0) {
 		error = xn_txq_mq_start_locked(txq, m);
 		XN_TX_UNLOCK(txq);
 	} else {
 		error = drbr_enqueue(ifp, txq->br, m);
 		taskqueue_enqueue(txq->tq, &txq->defrtask);
 	}
 
 	return (error);
 }
 
 static void
 xn_qflush(struct ifnet *ifp)
 {
 	struct netfront_info *np;
 	struct netfront_txq *txq;
 	struct mbuf *m;
 	int i;
 
 	np = ifp->if_softc;
 
 	for (i = 0; i < np->num_queues; i++) {
 		txq = &np->txq[i];
 
 		XN_TX_LOCK(txq);
 		while ((m = buf_ring_dequeue_sc(txq->br)) != NULL)
 			m_freem(m);
 		XN_TX_UNLOCK(txq);
 	}
 
 	if_qflush(ifp);
 }
 
 /**
  * Create a network device.
  * @param dev  Newbus device representing this virtual NIC.
  */
 int
 create_netdev(device_t dev)
 {
 	struct netfront_info *np;
 	int err;
 	struct ifnet *ifp;
 
 	np = device_get_softc(dev);
 
 	np->xbdev         = dev;
 
 	mtx_init(&np->sc_lock, "xnsc", "netfront softc lock", MTX_DEF);
 
 	ifmedia_init(&np->sc_media, 0, xn_ifmedia_upd, xn_ifmedia_sts);
 	ifmedia_add(&np->sc_media, IFM_ETHER|IFM_MANUAL, 0, NULL);
 	ifmedia_set(&np->sc_media, IFM_ETHER|IFM_MANUAL);
 
 	err = xen_net_read_mac(dev, np->mac);
 	if (err != 0)
 		goto error;
 
 	/* Set up ifnet structure */
 	ifp = np->xn_ifp = if_alloc(IFT_ETHER);
     	ifp->if_softc = np;
     	if_initname(ifp, "xn",  device_get_unit(dev));
     	ifp->if_flags = IFF_BROADCAST | IFF_SIMPLEX | IFF_MULTICAST;
     	ifp->if_ioctl = xn_ioctl;
 
 	ifp->if_transmit = xn_txq_mq_start;
 	ifp->if_qflush = xn_qflush;
 
     	ifp->if_init = xn_ifinit;
 
     	ifp->if_hwassist = XN_CSUM_FEATURES;
 	/* Enable all supported features at device creation. */
 	ifp->if_capenable = ifp->if_capabilities =
 	    IFCAP_HWCSUM|IFCAP_TSO4|IFCAP_LRO;
 	ifp->if_hw_tsomax = 65536 - (ETHER_HDR_LEN + ETHER_VLAN_ENCAP_LEN);
 	ifp->if_hw_tsomaxsegcount = MAX_TX_REQ_FRAGS;
 	ifp->if_hw_tsomaxsegsize = PAGE_SIZE;
 
     	ether_ifattach(ifp, np->mac);
 	netfront_carrier_off(np);
 
 	return (0);
 
 error:
 	KASSERT(err != 0, ("Error path with no error code specified"));
 	return (err);
 }
 
 static int
 netfront_detach(device_t dev)
 {
 	struct netfront_info *info = device_get_softc(dev);
 
 	DPRINTK("%s\n", xenbus_get_node(dev));
 
 	netif_free(info);
 
 	return 0;
 }
 
 static void
 netif_free(struct netfront_info *np)
 {
 
 	XN_LOCK(np);
 	xn_stop(np);
 	XN_UNLOCK(np);
 	netif_disconnect_backend(np);
 	ether_ifdetach(np->xn_ifp);
 	free(np->rxq, M_DEVBUF);
 	free(np->txq, M_DEVBUF);
 	if_free(np->xn_ifp);
 	np->xn_ifp = NULL;
 	ifmedia_removeall(&np->sc_media);
 }
 
 static void
 netif_disconnect_backend(struct netfront_info *np)
 {
 	u_int i;
 
 	for (i = 0; i < np->num_queues; i++) {
 		XN_RX_LOCK(&np->rxq[i]);
 		XN_TX_LOCK(&np->txq[i]);
 	}
 	netfront_carrier_off(np);
 	for (i = 0; i < np->num_queues; i++) {
 		XN_RX_UNLOCK(&np->rxq[i]);
 		XN_TX_UNLOCK(&np->txq[i]);
 	}
 
 	for (i = 0; i < np->num_queues; i++) {
 		disconnect_rxq(&np->rxq[i]);
 		disconnect_txq(&np->txq[i]);
 	}
 }
 
 static int
 xn_ifmedia_upd(struct ifnet *ifp)
 {
 
 	return (0);
 }
 
 static void
 xn_ifmedia_sts(struct ifnet *ifp, struct ifmediareq *ifmr)
 {
 
 	ifmr->ifm_status = IFM_AVALID|IFM_ACTIVE;
 	ifmr->ifm_active = IFM_ETHER|IFM_MANUAL;
 }
 
 /* ** Driver registration ** */
 static device_method_t netfront_methods[] = {
 	/* Device interface */
 	DEVMETHOD(device_probe,         netfront_probe),
 	DEVMETHOD(device_attach,        netfront_attach),
 	DEVMETHOD(device_detach,        netfront_detach),
 	DEVMETHOD(device_shutdown,      bus_generic_shutdown),
 	DEVMETHOD(device_suspend,       netfront_suspend),
 	DEVMETHOD(device_resume,        netfront_resume),
 
 	/* Xenbus interface */
 	DEVMETHOD(xenbus_otherend_changed, netfront_backend_changed),
 
 	DEVMETHOD_END
 };
 
 static driver_t netfront_driver = {
 	"xn",
 	netfront_methods,
 	sizeof(struct netfront_info),
 };
 devclass_t netfront_devclass;
 
 DRIVER_MODULE(xe, xenbusb_front, netfront_driver, netfront_devclass, NULL,
     NULL);
Index: head/sys/xen/xen-os.h
===================================================================
--- head/sys/xen/xen-os.h	(revision 314839)
+++ head/sys/xen/xen-os.h	(revision 314840)
@@ -1,145 +1,147 @@
 /******************************************************************************
  * xen/xen-os.h
  * 
  * Random collection of macros and definition
  *
  * Copyright (c) 2003, 2004 Keir Fraser (on behalf of the Xen team)
  * All rights reserved.
  *
  * Permission is hereby granted, free of charge, to any person obtaining a copy
  * of this software and associated documentation files (the "Software"), to
  * deal in the Software without restriction, including without limitation the
  * rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
  * sell copies of the Software, and to permit persons to whom the Software is
  * furnished to do so, subject to the following conditions:
  * 
  * The above copyright notice and this permission notice shall be included in
  * all copies or substantial portions of the Software.
  * 
  * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR 
  * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, 
  * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE 
  * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER 
  * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING 
  * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER 
  * DEALINGS IN THE SOFTWARE.
  *
  * $FreeBSD$
  */
 
 #ifndef _XEN_XEN_OS_H_
 #define _XEN_XEN_OS_H_
 
 #if !defined(__XEN_INTERFACE_VERSION__)  
 #define  __XEN_INTERFACE_VERSION__ 0x00030208
 #endif  
 
 #define GRANT_REF_INVALID   0xffffffff
 
 #ifdef LOCORE
 #define __ASSEMBLY__
 #endif
 
 #include <machine/xen/xen-os.h>
 
 #include <xen/interface/xen.h>
 
 /* Everything below this point is not included by assembler (.S) files. */
 #ifndef __ASSEMBLY__
 
 extern shared_info_t *HYPERVISOR_shared_info;
 extern start_info_t *HYPERVISOR_start_info;
 
 /* XXX: we need to get rid of this and use HYPERVISOR_start_info directly */
 extern char *console_page;
 
 extern int xen_disable_pv_disks;
 extern int xen_disable_pv_nics;
 
+extern bool xen_suspend_cancelled;
+
 enum xen_domain_type {
 	XEN_NATIVE,             /* running on bare hardware    */
 	XEN_PV_DOMAIN,          /* running in a PV domain      */
 	XEN_HVM_DOMAIN,         /* running in a Xen hvm domain */
 };
 
 extern enum xen_domain_type xen_domain_type;
 
 static inline int
 xen_domain(void)
 {
 	return (xen_domain_type != XEN_NATIVE);
 }
 
 static inline int
 xen_pv_domain(void)
 {
 	return (xen_domain_type == XEN_PV_DOMAIN);
 }
 
 static inline int
 xen_hvm_domain(void)
 {
 	return (xen_domain_type == XEN_HVM_DOMAIN);
 }
 
 static inline bool
 xen_initial_domain(void)
 {
 	return (xen_domain() && HYPERVISOR_start_info != NULL &&
 	    (HYPERVISOR_start_info->flags & SIF_INITDOMAIN) != 0);
 }
 
 /*
  * Based on ofed/include/linux/bitops.h
  *
  * Those helpers are prefixed by xen_ because xen-os.h is widely included
  * and we don't want the other drivers using them.
  *
  */
 #define NBPL (NBBY * sizeof(long))
 
 static inline bool
 xen_test_bit(int bit, volatile long *addr)
 {
 	unsigned long mask = 1UL << (bit % NBPL);
 
 	return !!(atomic_load_acq_long(&addr[bit / NBPL]) & mask);
 }
 
 static inline void
 xen_set_bit(int bit, volatile long *addr)
 {
 	atomic_set_long(&addr[bit / NBPL], 1UL << (bit % NBPL));
 }
 
 static inline void
 xen_clear_bit(int bit, volatile long *addr)
 {
 	atomic_clear_long(&addr[bit / NBPL], 1UL << (bit % NBPL));
 }
 
 #undef NBPL
 
 /*
  * Functions to allocate/free unused memory in order
  * to map memory from other domains.
  */
 struct resource *xenmem_alloc(device_t dev, int *res_id, size_t size);
 int xenmem_free(device_t dev, int res_id, struct resource *res);
 
 /* Debug/emergency function, prints directly to hypervisor console */
 void xc_printf(const char *, ...) __printflike(1, 2);
 
 #ifndef xen_mb
 #define xen_mb() mb()
 #endif
 #ifndef xen_rmb
 #define xen_rmb() rmb()
 #endif
 #ifndef xen_wmb
 #define xen_wmb() wmb()
 #endif
 
 #endif /* !__ASSEMBLY__ */
 
 #endif /* _XEN_XEN_OS_H_ */
Index: head/sys/xen/xenbus/xenbusb.c
===================================================================
--- head/sys/xen/xenbus/xenbusb.c	(revision 314839)
+++ head/sys/xen/xenbus/xenbusb.c	(revision 314840)
@@ -1,970 +1,975 @@
 /******************************************************************************
  * Copyright (C) 2010 Spectra Logic Corporation
  * Copyright (C) 2008 Doug Rabson
  * Copyright (C) 2005 Rusty Russell, IBM Corporation
  * Copyright (C) 2005 Mike Wray, Hewlett-Packard
  * Copyright (C) 2005 XenSource Ltd
  * 
  * This file may be distributed separately from the Linux kernel, or
  * incorporated into other software packages, subject to the following license:
  * 
  * Permission is hereby granted, free of charge, to any person obtaining a copy
  * of this source file (the "Software"), to deal in the Software without
  * restriction, including without limitation the rights to use, copy, modify,
  * merge, publish, distribute, sublicense, and/or sell copies of the Software,
  * and to permit persons to whom the Software is furnished to do so, subject to
  * the following conditions:
  * 
  * The above copyright notice and this permission notice shall be included in
  * all copies or substantial portions of the Software.
  * 
  * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
  * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
  * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
  * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
  * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
  * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
  * IN THE SOFTWARE.
  */
 
 /**
  * \file xenbusb.c
  *
  * \brief Shared support functions for managing the NewBus buses that contain
  *        Xen front and back end device instances.
  *
  * The NewBus implementation of XenBus attaches a xenbusb_front and xenbusb_back
  * child bus to the xenstore device.  This strategy allows the small differences
  * in the handling of XenBus operations for front and back devices to be handled
  * as overrides in xenbusb_front/back.c.  Front and back specific device
  * classes are also provided so device drivers can register for the devices they
  * can handle without the need to filter within their probe routines.  The
  * net result is a device hierarchy that might look like this:
  *
  * xenstore0/
  *           xenbusb_front0/
  *                         xn0
  *                         xbd0
  *                         xbd1
  *           xenbusb_back0/
  *                        xbbd0
  *                        xnb0
  *                        xnb1
  */
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/bus.h>
 #include <sys/kernel.h>
 #include <sys/lock.h>
 #include <sys/malloc.h>
 #include <sys/module.h>
 #include <sys/sbuf.h>
 #include <sys/sysctl.h>
 #include <sys/syslog.h>
 #include <sys/systm.h>
 #include <sys/sx.h>
 #include <sys/taskqueue.h>
 
 #include <machine/xen/xen-os.h>
 #include <machine/stdarg.h>
 
 #include <xen/gnttab.h>
 #include <xen/xenstore/xenstorevar.h>
 #include <xen/xenbus/xenbusb.h>
 #include <xen/xenbus/xenbusvar.h>
 
 /*------------------------- Private Functions --------------------------------*/
 /**
  * \brief Deallocate XenBus device instance variables.
  *
  * \param ivars  The instance variable block to free.
  */
 static void
 xenbusb_free_child_ivars(struct xenbus_device_ivars *ivars)
 {
 	if (ivars->xd_otherend_watch.node != NULL) {
 		xs_unregister_watch(&ivars->xd_otherend_watch);
 		free(ivars->xd_otherend_watch.node, M_XENBUS);
 		ivars->xd_otherend_watch.node = NULL;
 	}
 
 	if (ivars->xd_local_watch.node != NULL) {
 		xs_unregister_watch(&ivars->xd_local_watch);
 		ivars->xd_local_watch.node = NULL;
 	}
 
 	if (ivars->xd_node != NULL) {
 		free(ivars->xd_node, M_XENBUS);
 		ivars->xd_node = NULL;
 	}
 	ivars->xd_node_len = 0;
 
 	if (ivars->xd_type != NULL) {
 		free(ivars->xd_type, M_XENBUS);
 		ivars->xd_type = NULL;
 	}
 
 	if (ivars->xd_otherend_path != NULL) {
 		free(ivars->xd_otherend_path, M_XENBUS);
 		ivars->xd_otherend_path = NULL;
 	}
 	ivars->xd_otherend_path_len = 0;
 
 	free(ivars, M_XENBUS);
 }
 
 /**
  * XenBus watch callback registered against the "state" XenStore
  * node of the other-end of a split device connection.
  *
  * This callback is invoked whenever the state of a device instance's
  * peer changes.
  *
  * \param watch      The xs_watch object used to register this callback
  *                   function.
  * \param vec        An array of pointers to NUL terminated strings containing
  *                   watch event data.  The vector should be indexed via the
  *                   xs_watch_type enum in xs_wire.h.
  * \param vec_size   The number of elements in vec.
  */
 static void
 xenbusb_otherend_watch_cb(struct xs_watch *watch, const char **vec,
     unsigned int vec_size __unused)
 {
 	struct xenbus_device_ivars *ivars;
 	device_t child;
 	device_t bus;
 	const char *path;
 	enum xenbus_state newstate;
 
 	ivars = (struct xenbus_device_ivars *)watch->callback_data;
 	child = ivars->xd_dev;
 	bus = device_get_parent(child);
 
 	path = vec[XS_WATCH_PATH];
 	if (ivars->xd_otherend_path == NULL
 	 || strncmp(ivars->xd_otherend_path, path, ivars->xd_otherend_path_len))
 		return;
 
 	newstate = xenbus_read_driver_state(ivars->xd_otherend_path);
 	XENBUSB_OTHEREND_CHANGED(bus, child, newstate);
 }
 
 /**
  * XenBus watch callback registered against the XenStore sub-tree
  * represnting the local half of a split device connection.
  *
  * This callback is invoked whenever any XenStore data in the subtree
  * is modified, either by us or another privledged domain.
  *
  * \param watch      The xs_watch object used to register this callback
  *                   function.
  * \param vec        An array of pointers to NUL terminated strings containing
  *                   watch event data.  The vector should be indexed via the
  *                   xs_watch_type enum in xs_wire.h.
  * \param vec_size   The number of elements in vec.
  *
  */
 static void
 xenbusb_local_watch_cb(struct xs_watch *watch, const char **vec,
     unsigned int vec_size __unused)
 {
 	struct xenbus_device_ivars *ivars;
 	device_t child;
 	device_t bus;
 	const char *path;
 
 	ivars = (struct xenbus_device_ivars *)watch->callback_data;
 	child = ivars->xd_dev;
 	bus = device_get_parent(child);
 
 	path = vec[XS_WATCH_PATH];
 	if (ivars->xd_node == NULL
 	 || strncmp(ivars->xd_node, path, ivars->xd_node_len))
 		return;
 
 	XENBUSB_LOCALEND_CHANGED(bus, child, &path[ivars->xd_node_len]);
 }
 
 /**
  * Search our internal record of configured devices (not the XenStore)
  * to determine if the XenBus device indicated by \a node is known to
  * the system.
  *
  * \param dev   The XenBus bus instance to search for device children.
  * \param node  The XenStore node path for the device to find.
  *
  * \return  The device_t of the found device if any, or NULL.
  *
  * \note device_t is a pointer type, so it can be compared against
  *       NULL for validity. 
  */
 static device_t
 xenbusb_device_exists(device_t dev, const char *node)
 {
 	device_t *kids;
 	device_t result;
 	struct xenbus_device_ivars *ivars;
 	int i, count;
 
 	if (device_get_children(dev, &kids, &count))
 		return (FALSE);
 
 	result = NULL;
 	for (i = 0; i < count; i++) {
 		ivars = device_get_ivars(kids[i]);
 		if (!strcmp(ivars->xd_node, node)) {
 			result = kids[i];
 			break;
 		}
 	}
 	free(kids, M_TEMP);
 
 	return (result);
 }
 
 static void
 xenbusb_delete_child(device_t dev, device_t child)
 {
 	struct xenbus_device_ivars *ivars;
 
 	ivars = device_get_ivars(child);
 
 	/*
 	 * We no longer care about the otherend of the
 	 * connection.  Cancel the watches now so that we
 	 * don't try to handle an event for a partially
 	 * detached child.
 	 */
 	if (ivars->xd_otherend_watch.node != NULL)
 		xs_unregister_watch(&ivars->xd_otherend_watch);
 	if (ivars->xd_local_watch.node != NULL)
 		xs_unregister_watch(&ivars->xd_local_watch);
 	
 	device_delete_child(dev, child);
 	xenbusb_free_child_ivars(ivars);
 }
 
 /**
  * \param dev    The NewBus device representing this XenBus bus.
  * \param child	 The NewBus device representing a child of dev%'s XenBus bus.
  */
 static void
 xenbusb_verify_device(device_t dev, device_t child)
 {
 	if (xs_exists(XST_NIL, xenbus_get_node(child), "") == 0) {
 
 		/*
 		 * Device tree has been removed from Xenbus.
 		 * Tear down the device.
 		 */
 		xenbusb_delete_child(dev, child);
 	}
 }
 
 /**
  * \brief Enumerate the devices on a XenBus bus and register them with
  *        the NewBus device tree.
  *
  * xenbusb_enumerate_bus() will create entries (in state DS_NOTPRESENT)
  * for nodes that appear in the XenStore, but will not invoke probe/attach
  * operations on drivers.  Probe/Attach processing must be separately
  * performed via an invocation of xenbusb_probe_children().  This is usually
  * done via the xbs_probe_children task.
  *
  * \param xbs  XenBus Bus device softc of the owner of the bus to enumerate.
  *
  * \return  On success, 0. Otherwise an errno value indicating the
  *          type of failure.
  */
 static int
 xenbusb_enumerate_bus(struct xenbusb_softc *xbs)
 {
 	const char **types;
 	u_int type_idx;
 	u_int type_count;
 	int error;
 
 	error = xs_directory(XST_NIL, xbs->xbs_node, "", &type_count, &types);
 	if (error)
 		return (error);
 
 	for (type_idx = 0; type_idx < type_count; type_idx++)
 		XENBUSB_ENUMERATE_TYPE(xbs->xbs_dev, types[type_idx]);
 
 	free(types, M_XENSTORE);
 
 	return (0);
 }
 
 /**
  * Handler for all generic XenBus device systcl nodes.
  */
 static int
 xenbusb_device_sysctl_handler(SYSCTL_HANDLER_ARGS)  
 {
 	device_t dev;
         const char *value;
 
 	dev = (device_t)arg1;
         switch (arg2) {
 	case XENBUS_IVAR_NODE:
 		value = xenbus_get_node(dev);
 		break;
 	case XENBUS_IVAR_TYPE:
 		value = xenbus_get_type(dev);
 		break;
 	case XENBUS_IVAR_STATE:
 		value = xenbus_strstate(xenbus_get_state(dev));
 		break;
 	case XENBUS_IVAR_OTHEREND_ID:
 		return (sysctl_handle_int(oidp, NULL,
 					  xenbus_get_otherend_id(dev),
 					  req));
 		/* NOTREACHED */
 	case XENBUS_IVAR_OTHEREND_PATH:
 		value = xenbus_get_otherend_path(dev);
                 break;
 	default:
 		return (EINVAL);
 	}
 	return (SYSCTL_OUT_STR(req, value));
 }
 
 /**
  * Create read-only systcl nodes for xenbusb device ivar data.
  *
  * \param dev  The XenBus device instance to register with sysctl.
  */
 static void
 xenbusb_device_sysctl_init(device_t dev)
 {
 	struct sysctl_ctx_list *ctx;
 	struct sysctl_oid      *tree;
 
 	ctx  = device_get_sysctl_ctx(dev);
 	tree = device_get_sysctl_tree(dev);
 
         SYSCTL_ADD_PROC(ctx,
 			SYSCTL_CHILDREN(tree),
 			OID_AUTO,
 			"xenstore_path",
 			CTLTYPE_STRING | CTLFLAG_RD,
 			dev,
 			XENBUS_IVAR_NODE,
 			xenbusb_device_sysctl_handler,
 			"A",
 			"XenStore path to device");
 
         SYSCTL_ADD_PROC(ctx,
 			SYSCTL_CHILDREN(tree),
 			OID_AUTO,
 			"xenbus_dev_type",
 			CTLTYPE_STRING | CTLFLAG_RD,
 			dev,
 			XENBUS_IVAR_TYPE,
 			xenbusb_device_sysctl_handler,
 			"A",
 			"XenBus device type");
 
         SYSCTL_ADD_PROC(ctx,
 			SYSCTL_CHILDREN(tree),
 			OID_AUTO,
 			"xenbus_connection_state",
 			CTLTYPE_STRING | CTLFLAG_RD,
 			dev,
 			XENBUS_IVAR_STATE,
 			xenbusb_device_sysctl_handler,
 			"A",
 			"XenBus state of peer connection");
 
         SYSCTL_ADD_PROC(ctx,
 			SYSCTL_CHILDREN(tree),
 			OID_AUTO,
 			"xenbus_peer_domid",
 			CTLTYPE_INT | CTLFLAG_RD,
 			dev,
 			XENBUS_IVAR_OTHEREND_ID,
 			xenbusb_device_sysctl_handler,
 			"I",
 			"Xen domain ID of peer");
 
         SYSCTL_ADD_PROC(ctx,
 			SYSCTL_CHILDREN(tree),
 			OID_AUTO,
 			"xenstore_peer_path",
 			CTLTYPE_STRING | CTLFLAG_RD,
 			dev,
 			XENBUS_IVAR_OTHEREND_PATH,
 			xenbusb_device_sysctl_handler,
 			"A",
 			"XenStore path to peer device");
 }
 
 /**
  * \brief Decrement the number of XenBus child devices in the
  *        connecting state by one and release the xbs_attch_ch
  *        interrupt configuration hook if the connecting count
  *        drops to zero.
  *
  * \param xbs  XenBus Bus device softc of the owner of the bus to enumerate.
  */
 static void
 xenbusb_release_confighook(struct xenbusb_softc *xbs)
 {
 	mtx_lock(&xbs->xbs_lock);
 	KASSERT(xbs->xbs_connecting_children > 0,
 		("Connecting device count error\n"));
 	xbs->xbs_connecting_children--;
 	if (xbs->xbs_connecting_children == 0
 	 && (xbs->xbs_flags & XBS_ATTACH_CH_ACTIVE) != 0) {
 		xbs->xbs_flags &= ~XBS_ATTACH_CH_ACTIVE;
 		mtx_unlock(&xbs->xbs_lock);
 		config_intrhook_disestablish(&xbs->xbs_attach_ch);
 	} else {
 		mtx_unlock(&xbs->xbs_lock);
 	}
 }
 
 /**
  * \brief Verify the existance of attached device instances and perform
  *        probe/attach processing for newly arrived devices.
  *
  * \param dev  The NewBus device representing this XenBus bus.
  *
  * \return  On success, 0. Otherwise an errno value indicating the
  *          type of failure.
  */
 static int
 xenbusb_probe_children(device_t dev)
 {
 	device_t *kids;
 	struct xenbus_device_ivars *ivars;
 	int i, count, error;
 
 	if (device_get_children(dev, &kids, &count) == 0) {
 		for (i = 0; i < count; i++) {
 			if (device_get_state(kids[i]) != DS_NOTPRESENT) {
 				/*
 				 * We already know about this one.
 				 * Make sure it's still here.
 				 */
 				xenbusb_verify_device(dev, kids[i]);
 				continue;
 			}
 
 			error = device_probe_and_attach(kids[i]);
 			if (error == ENXIO) {
 				struct xenbusb_softc *xbs;
 
 				/*
 				 * We don't have a PV driver for this device.
 				 * However, an emulated device we do support
 				 * may share this backend.  Hide the node from
 				 * XenBus until the next rescan, but leave it's
 				 * state unchanged so we don't inadvertently
 				 * prevent attachment of any emulated device.
 				 */
 				xenbusb_delete_child(dev, kids[i]);
 
 				/*
 				 * Since the XenStore state of this device
 				 * still indicates a pending attach, manually
 				 * release it's hold on the boot process.
 				 */
 				xbs = device_get_softc(dev);
 				xenbusb_release_confighook(xbs);
 
 				continue;
 			} else if (error) {
 				/*
 				 * Transition device to the closed state
 				 * so the world knows that attachment will
 				 * not occur.
 				 */
 				xenbus_set_state(kids[i], XenbusStateClosed);
 
 				/*
 				 * Remove our record of this device.
 				 * So long as it remains in the closed
 				 * state in the XenStore, we will not find
 				 * it again.  The state will only change
 				 * if the control domain actively reconfigures
 				 * this device.
 				 */
 				xenbusb_delete_child(dev, kids[i]);
 
 				continue;
 			}
 			/*
 			 * Augment default newbus provided dynamic sysctl
 			 * variables with the standard ivar contents of
 			 * XenBus devices.
 			 */
 			xenbusb_device_sysctl_init(kids[i]);
 
 			/*
 			 * Now that we have a driver managing this device
 			 * that can receive otherend state change events,
 			 * hook up a watch for them.
 			 */
 			ivars = device_get_ivars(kids[i]);
 			xs_register_watch(&ivars->xd_otherend_watch);
 			xs_register_watch(&ivars->xd_local_watch);
 		}
 		free(kids, M_TEMP);
 	}
 
 	return (0);
 }
 
 /**
  * \brief Task callback function to perform XenBus probe operations
  *        from a known safe context.
  *
  * \param arg      The NewBus device_t representing the bus instance to
  *                 on which to perform probe processing.
  * \param pending  The number of times this task was queued before it could
  *                 be run.
  */
 static void
 xenbusb_probe_children_cb(void *arg, int pending __unused)
 {
 	device_t dev = (device_t)arg;
 
 	/*
 	 * Hold Giant until the Giant free newbus changes are committed.
 	 */
 	mtx_lock(&Giant);
 	xenbusb_probe_children(dev);
 	mtx_unlock(&Giant);
 }
 
 /**
  * \brief XenStore watch callback for the root node of the XenStore
  *        subtree representing a XenBus.
  *
  * This callback performs, or delegates to the xbs_probe_children task,
  * all processing necessary to handle dynmaic device arrival and departure
  * events from a XenBus.
  *
  * \param watch  The XenStore watch object associated with this callback.
  * \param vec    The XenStore watch event data.
  * \param len	 The number of fields in the event data stream.
  */
 static void
 xenbusb_devices_changed(struct xs_watch *watch, const char **vec,
 			unsigned int len)
 {
 	struct xenbusb_softc *xbs;
 	device_t dev;
 	char *node;
 	char *type;
 	char *id;
 	char *p;
 	u_int component;
 
 	xbs = (struct xenbusb_softc *)watch->callback_data;
 	dev = xbs->xbs_dev;
 
 	if (len <= XS_WATCH_PATH) {
 		device_printf(dev, "xenbusb_devices_changed: "
 			      "Short Event Data.\n");
 		return;
 	}
 
 	node = strdup(vec[XS_WATCH_PATH], M_XENBUS);
 	p = strchr(node, '/');
 	if (p == NULL)
 		goto out;
 	*p = 0;
 	type = p + 1;
 
 	p = strchr(type, '/');
 	if (p == NULL)
 		goto out;
 	*p++ = 0;
 
 	/*
 	 * Extract the device ID.  A device ID has one or more path
 	 * components separated by the '/' character.
 	 *
 	 * e.g. "<frontend vm id>/<frontend dev id>" for backend devices.
 	 */
 	id = p;
 	for (component = 0; component < xbs->xbs_id_components; component++) {
 		p = strchr(p, '/');
 		if (p == NULL)
 			break;
 		p++;
 	}
 	if (p != NULL)
 		*p = 0;
 
 	if (*id != 0 && component >= xbs->xbs_id_components - 1) {
 		xenbusb_add_device(xbs->xbs_dev, type, id);
 		taskqueue_enqueue(taskqueue_thread, &xbs->xbs_probe_children);
 	}
 out:
 	free(node, M_XENBUS);
 }
 
 /**
  * \brief Interrupt configuration hook callback associated with xbs_attch_ch.
  *
  * Since interrupts are always functional at the time of XenBus configuration,
  * there is nothing to be done when the callback occurs.  This hook is only
  * registered to hold up boot processing while XenBus devices come online.
  * 
  * \param arg  Unused configuration hook callback argument.
  */
 static void
 xenbusb_nop_confighook_cb(void *arg __unused)
 {
 }
 
 /*--------------------------- Public Functions -------------------------------*/
 /*--------- API comments for these methods can be found in xenbusb.h ---------*/
 void
 xenbusb_identify(driver_t *driver __unused, device_t parent)
 {
 	/*
 	 * A single instance of each bus type for which we have a driver
 	 * is always present in a system operating under Xen.
 	 */
 	BUS_ADD_CHILD(parent, 0, driver->name, 0);
 }
 
 int
 xenbusb_add_device(device_t dev, const char *type, const char *id)
 {
 	struct xenbusb_softc *xbs;
 	struct sbuf *devpath_sbuf;
 	char *devpath;
 	struct xenbus_device_ivars *ivars;
 	int error;
 
 	xbs = device_get_softc(dev);
 	devpath_sbuf = sbuf_new_auto();
 	sbuf_printf(devpath_sbuf, "%s/%s/%s", xbs->xbs_node, type, id);
 	sbuf_finish(devpath_sbuf);
 	devpath = sbuf_data(devpath_sbuf);
 
 	ivars = malloc(sizeof(*ivars), M_XENBUS, M_ZERO|M_WAITOK);
 	error = ENXIO;
 
 	if (xs_exists(XST_NIL, devpath, "") != 0) {
 		device_t child;
 		enum xenbus_state state;
 		char *statepath;
 
 		child = xenbusb_device_exists(dev, devpath);
 		if (child != NULL) {
 			/*
 			 * We are already tracking this node
 			 */
 			error = 0;
 			goto out;
 		}
 			
 		state = xenbus_read_driver_state(devpath);
 		if (state != XenbusStateInitialising) {
 			/*
 			 * Device is not new, so ignore it. This can
 			 * happen if a device is going away after
 			 * switching to Closed.
 			 */
 			printf("xenbusb_add_device: Device %s ignored. "
 			       "State %d\n", devpath, state);
 			error = 0;
 			goto out;
 		}
 
 		sx_init(&ivars->xd_lock, "xdlock");
 		ivars->xd_flags = XDF_CONNECTING;
 		ivars->xd_node = strdup(devpath, M_XENBUS);
 		ivars->xd_node_len = strlen(devpath);
 		ivars->xd_type  = strdup(type, M_XENBUS);
 		ivars->xd_state = XenbusStateInitialising;
 
 		error = XENBUSB_GET_OTHEREND_NODE(dev, ivars);
 		if (error) {
 			printf("xenbus_update_device: %s no otherend id\n",
 			    devpath); 
 			goto out;
 		}
 
 		statepath = malloc(ivars->xd_otherend_path_len
 		    + strlen("/state") + 1, M_XENBUS, M_WAITOK);
 		sprintf(statepath, "%s/state", ivars->xd_otherend_path);
 		ivars->xd_otherend_watch.node = statepath;
 		ivars->xd_otherend_watch.callback = xenbusb_otherend_watch_cb;
 		ivars->xd_otherend_watch.callback_data = (uintptr_t)ivars;
 
 		ivars->xd_local_watch.node = ivars->xd_node;
 		ivars->xd_local_watch.callback = xenbusb_local_watch_cb;
 		ivars->xd_local_watch.callback_data = (uintptr_t)ivars;
 
 		mtx_lock(&xbs->xbs_lock);
 		xbs->xbs_connecting_children++;
 		mtx_unlock(&xbs->xbs_lock);
 
 		child = device_add_child(dev, NULL, -1);
 		ivars->xd_dev = child;
 		device_set_ivars(child, ivars);
 	}
 
 out:
 	sbuf_delete(devpath_sbuf);
 	if (error != 0)
 		xenbusb_free_child_ivars(ivars);
 
 	return (error);
 }
 
 int
 xenbusb_attach(device_t dev, char *bus_node, u_int id_components)
 {
 	struct xenbusb_softc *xbs;
 
 	xbs = device_get_softc(dev);
 	mtx_init(&xbs->xbs_lock, "xenbusb softc lock", NULL, MTX_DEF);
 	xbs->xbs_node = bus_node;
 	xbs->xbs_id_components = id_components;
 	xbs->xbs_dev = dev;
 
 	/*
 	 * Since XenBus buses are attached to the XenStore, and
 	 * the XenStore does not probe children until after interrupt
 	 * services are available, this config hook is used solely
 	 * to ensure that the remainder of the boot process (e.g.
 	 * mount root) is deferred until child devices are adequately
 	 * probed.  We unblock the boot process as soon as the
 	 * connecting child count in our softc goes to 0.
 	 */
 	xbs->xbs_attach_ch.ich_func = xenbusb_nop_confighook_cb;
 	xbs->xbs_attach_ch.ich_arg = dev;
 	config_intrhook_establish(&xbs->xbs_attach_ch);
 	xbs->xbs_flags |= XBS_ATTACH_CH_ACTIVE;
 	xbs->xbs_connecting_children = 1;
 
 	/*
 	 * The subtree for this bus type may not yet exist
 	 * causing initial enumeration to fail.  We still
 	 * want to return success from our attach though
 	 * so that we are ready to handle devices for this
 	 * bus when they are dynamically attached to us
 	 * by a Xen management action.
 	 */
 	(void)xenbusb_enumerate_bus(xbs);
 	xenbusb_probe_children(dev);
 
 	xbs->xbs_device_watch.node = bus_node;
 	xbs->xbs_device_watch.callback = xenbusb_devices_changed;
 	xbs->xbs_device_watch.callback_data = (uintptr_t)xbs;
 
 	TASK_INIT(&xbs->xbs_probe_children, 0, xenbusb_probe_children_cb, dev);
 
 	xs_register_watch(&xbs->xbs_device_watch);
 
 	xenbusb_release_confighook(xbs);
 
 	return (0);
 }
 
 int
 xenbusb_resume(device_t dev)
 {
 	device_t *kids;
 	struct xenbus_device_ivars *ivars;
 	int i, count, error;
 	char *statepath;
 
 	/*
 	 * We must re-examine each device and find the new path for
 	 * its backend.
 	 */
 	if (device_get_children(dev, &kids, &count) == 0) {
 		for (i = 0; i < count; i++) {
 			if (device_get_state(kids[i]) == DS_NOTPRESENT)
 				continue;
 
+			if (xen_suspend_cancelled) {
+				DEVICE_RESUME(kids[i]);
+				continue;
+			}
+
 			ivars = device_get_ivars(kids[i]);
 
 			xs_unregister_watch(&ivars->xd_otherend_watch);
 			xenbus_set_state(kids[i], XenbusStateInitialising);
 
 			/*
 			 * Find the new backend details and
 			 * re-register our watch.
 			 */
 			error = XENBUSB_GET_OTHEREND_NODE(dev, ivars);
 			if (error)
 				return (error);
 
 			statepath = malloc(ivars->xd_otherend_path_len
 			    + strlen("/state") + 1, M_XENBUS, M_WAITOK);
 			sprintf(statepath, "%s/state", ivars->xd_otherend_path);
 
 			free(ivars->xd_otherend_watch.node, M_XENBUS);
 			ivars->xd_otherend_watch.node = statepath;
 
 			DEVICE_RESUME(kids[i]);
 
 			xs_register_watch(&ivars->xd_otherend_watch);
 #if 0
 			/*
 			 * Can't do this yet since we are running in
 			 * the xenwatch thread and if we sleep here,
 			 * we will stop delivering watch notifications
 			 * and the device will never come back online.
 			 */
 			sx_xlock(&ivars->xd_lock);
 			while (ivars->xd_state != XenbusStateClosed
 			    && ivars->xd_state != XenbusStateConnected)
 				sx_sleep(&ivars->xd_state, &ivars->xd_lock,
 				    0, "xdresume", 0);
 			sx_xunlock(&ivars->xd_lock);
 #endif
 		}
 		free(kids, M_TEMP);
 	}
 
 	return (0);
 }
 
 int
 xenbusb_print_child(device_t dev, device_t child)
 {
 	struct xenbus_device_ivars *ivars = device_get_ivars(child);
 	int	retval = 0;
 
 	retval += bus_print_child_header(dev, child);
 	retval += printf(" at %s", ivars->xd_node);
 	retval += bus_print_child_footer(dev, child);
 
 	return (retval);
 }
 
 int
 xenbusb_read_ivar(device_t dev, device_t child, int index, uintptr_t *result)
 {
 	struct xenbus_device_ivars *ivars = device_get_ivars(child);
 
 	switch (index) {
 	case XENBUS_IVAR_NODE:
 		*result = (uintptr_t) ivars->xd_node;
 		return (0);
 
 	case XENBUS_IVAR_TYPE:
 		*result = (uintptr_t) ivars->xd_type;
 		return (0);
 
 	case XENBUS_IVAR_STATE:
 		*result = (uintptr_t) ivars->xd_state;
 		return (0);
 
 	case XENBUS_IVAR_OTHEREND_ID:
 		*result = (uintptr_t) ivars->xd_otherend_id;
 		return (0);
 
 	case XENBUS_IVAR_OTHEREND_PATH:
 		*result = (uintptr_t) ivars->xd_otherend_path;
 		return (0);
 	}
 
 	return (ENOENT);
 }
 
 int
 xenbusb_write_ivar(device_t dev, device_t child, int index, uintptr_t value)
 {
 	struct xenbus_device_ivars *ivars = device_get_ivars(child);
 	enum xenbus_state newstate;
 	int currstate;
 
 	switch (index) {
 	case XENBUS_IVAR_STATE:
 	{
 		int error;
 
 		newstate = (enum xenbus_state)value;
 		sx_xlock(&ivars->xd_lock);
 		if (ivars->xd_state == newstate) {
 			error = 0;
 			goto out;
 		}
 
 		error = xs_scanf(XST_NIL, ivars->xd_node, "state",
 		    NULL, "%d", &currstate);
 		if (error)
 			goto out;
 
 		do {
 			error = xs_printf(XST_NIL, ivars->xd_node, "state",
 			    "%d", newstate);
 		} while (error == EAGAIN);
 		if (error) {
 			/*
 			 * Avoid looping through xenbus_dev_fatal()
 			 * which calls xenbus_write_ivar to set the
 			 * state to closing.
 			 */
 			if (newstate != XenbusStateClosing)
 				xenbus_dev_fatal(dev, error,
 						 "writing new state");
 			goto out;
 		}
 		ivars->xd_state = newstate;
 
 		if ((ivars->xd_flags & XDF_CONNECTING) != 0
 		 && (newstate == XenbusStateClosed
 		  || newstate == XenbusStateConnected)) {
 			struct xenbusb_softc *xbs;
 
 			ivars->xd_flags &= ~XDF_CONNECTING;
 			xbs = device_get_softc(dev);
 			xenbusb_release_confighook(xbs);
 		}
 
 		wakeup(&ivars->xd_state);
 	out:
 		sx_xunlock(&ivars->xd_lock);
 		return (error);
 	}
 
 	case XENBUS_IVAR_NODE:
 	case XENBUS_IVAR_TYPE:
 	case XENBUS_IVAR_OTHEREND_ID:
 	case XENBUS_IVAR_OTHEREND_PATH:
 		/*
 		 * These variables are read-only.
 		 */
 		return (EINVAL);
 	}
 
 	return (ENOENT);
 }
 
 void
 xenbusb_otherend_changed(device_t bus, device_t child, enum xenbus_state state)
 {
 	XENBUS_OTHEREND_CHANGED(child, state);
 }
 
 void
 xenbusb_localend_changed(device_t bus, device_t child, const char *path)
 {
 
 	if (strcmp(path, "/state") != 0) {
 		struct xenbus_device_ivars *ivars;
 
 		ivars = device_get_ivars(child);
 		sx_xlock(&ivars->xd_lock);
 		ivars->xd_state = xenbus_read_driver_state(ivars->xd_node);
 		sx_xunlock(&ivars->xd_lock);
 	}
 	XENBUS_LOCALEND_CHANGED(child, path);
 }